{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"},{"sourceId":14312414,"sourceType":"datasetVersion","datasetId":9136777}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-21T03:25:42.059019Z","iopub.execute_input":"2025-12-21T03:25:42.059331Z","iopub.status.idle":"2025-12-21T03:26:11.535946Z","shell.execute_reply.started":"2025-12-21T03:25:42.0593Z","shell.execute_reply":"2025-12-21T03:26:11.534475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T03:26:34.555906Z","iopub.execute_input":"2025-12-21T03:26:34.556579Z","iopub.status.idle":"2025-12-21T03:26:34.837785Z","shell.execute_reply.started":"2025-12-21T03:26:34.556544Z","shell.execute_reply":"2025-12-21T03:26:34.836467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Thống kê số lượng mẫu theo từng class_id\n# Groupby cả class_id và class_name để dễ đọc, hàm size() dùng để đếm\nclass_stats = df.groupby(['class_id', 'class_name']).size().reset_index(name='quantity')\n\n# Sắp xếp lại theo class_id từ 0 đến 14\nclass_stats = class_stats.sort_values('class_id')\n\n# Hiển thị bảng kết quả\nprint(class_stats)\n\n# (Tuỳ chọn) Vẽ biểu đồ đơn giản để bạn dễ nhìn sự chênh lệch\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(12, 6))\nsns.barplot(data=class_stats, x='class_id', y='quantity', hue='class_name')\nplt.title('Distribution of Classes in Dataset')\nplt.xlabel('Class ID')\nplt.ylabel('Quantity')\nplt.xticks(rotation=0)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T03:26:46.944747Z","iopub.execute_input":"2025-12-21T03:26:46.945403Z","iopub.status.idle":"2025-12-21T03:26:49.728128Z","shell.execute_reply.started":"2025-12-21T03:26:46.945355Z","shell.execute_reply":"2025-12-21T03:26:49.726858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# 1. Đọc file metadata gốc\n# Thay đường dẫn này bằng đường dẫn file thực tế của bạn\ncsv_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\"\ndf = pd.read_csv(csv_path)\n\n# 2. Tách dữ liệu thành 2 nhóm\n# Nhóm A: Giữ lại TẤT CẢ các dòng có bệnh (class_id khác 14)\ndf_disease = df[df['class_id'] != 14]\n\n# Nhóm B: Lấy các dòng không có bệnh (class_id bằng 14)\ndf_normal = df[df['class_id'] == 14]\n\n# 3. Lấy 14,000 dòng đầu tiên của nhóm không bệnh\n# (Nếu bạn muốn lấy ngẫu nhiên thay vì lấy đầu tiên, hãy đổi .head(14000) thành .sample(14000))\ndf_normal_reduced = df_normal.head(14000)\n\n# 4. Gộp lại thành dataset mới\ndf_new = pd.concat([df_disease, df_normal_reduced])\n\n# 5. Xáo trộn ngẫu nhiên (Shuffle)\n# Bước này cực quan trọng để khi train model không bị học vẹt theo thứ tự (ví dụ nửa đầu toàn bệnh, nửa sau toàn sạch)\ndf_new = df_new.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# 6. Kiểm tra lại kết quả\nprint(f\"Số lượng dòng ban đầu: {len(df)}\")\nprint(f\"Số lượng dòng sau khi lọc: {len(df_new)}\")\nprint(\"-\" * 30)\nprint(\"Phân bố class mới:\")\nprint(df_new['class_id'].value_counts().sort_index())\n\n# 7. Lưu ra file CSV mới\ndf_new.to_csv(\"train_filtered_balanced.csv\", index=False)\nprint(\"\\nĐã lưu file thành công: train_filtered_balanced.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-26T08:08:03.862837Z","iopub.execute_input":"2025-12-26T08:08:03.863747Z","iopub.status.idle":"2025-12-26T08:08:04.740694Z","shell.execute_reply.started":"2025-12-26T08:08:03.863679Z","shell.execute_reply":"2025-12-26T08:08:04.739473Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Convert Img + zip\n","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd \nimport pydicom\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\nimport concurrent.futures\n\n# --- CẤU HÌNH ---\nINPUT_DIR = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\nOUTPUT_DIR = \"/kaggle/working/train_png_640\"\nFILTER_CSV = \"/kaggle/working/temp_train_1.csv\"\nIMG_SIZE = 2160\n\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# --- HÀM XỬ LÝ (GIỮ NGUYÊN CỦA BẠN) ---\ndef convert_one_image(filename):\n    if not filename.endswith('.dicom'): return\n    \n    file_path = os.path.join(INPUT_DIR, filename)\n    save_path = os.path.join(OUTPUT_DIR, filename.replace('.dicom', '.png'))\n    \n    if os.path.exists(save_path): return\n\n    try:\n        dicom = pydicom.dcmread(file_path)\n        pixel_array = dicom.pixel_array\n        \n        if \"PhotometricInterpretation\" in dicom and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n            pixel_array = np.max(pixel_array) - pixel_array\n            \n        pixel_array = pixel_array.astype(np.float32)\n        pixel_array = (pixel_array - pixel_array.min()) / (pixel_array.max() - pixel_array.min()) * 255.0\n        pixel_array = pixel_array.astype(np.uint8)\n        \n        img = cv2.resize(pixel_array, (IMG_SIZE, IMG_SIZE))\n        cv2.imwrite(save_path, img)\n    except Exception as e:\n        print(f\"Error converting {filename}: {e}\")\n\n# --- LỌC DANH SÁCH FILE ---\n\n# 1. Đọc danh sách ID cần thiết từ file CSV của bạn\nprint(f\"Đang đọc danh sách ID từ {FILTER_CSV}...\")\ndf_filter = pd.read_csv(FILTER_CSV)\nvalid_ids = set(df_filter['image_id'].unique()) # Dùng set để tra cứu cho nhanh\n\nprint(f\"Số lượng ảnh cần convert: {len(valid_ids)}\")\n\n# 2. Tạo danh sách file dựa trên valid_ids\n# Lưu ý: Trong CSV là ID (ví dụ 'abc'), nhưng tên file là 'abc.dicom'\ntarget_files = [f\"{img_id}.dicom\" for img_id in valid_ids]\n\n# 3. Kiểm tra xem file có thực sự tồn tại trong thư mục gốc không (để tránh lỗi)\n# Bước này hơi tốn thời gian một chút nhưng an toàn. \n# Nếu bạn chắc chắn file nào cũng có thì có thể bỏ qua bước check os.path.exists\nfiles_to_process = []\nfor f in target_files:\n    if os.path.exists(os.path.join(INPUT_DIR, f)):\n        files_to_process.append(f)\n    else:\n        # Trường hợp hiếm: Có ID trong CSV nhưng không tìm thấy file ảnh gốc\n        pass \n\nprint(f\"Bắt đầu xử lý {len(files_to_process)} ảnh...\")\n\n# 4. Chạy đa luồng với danh sách ĐÃ LỌC\nwith concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor:\n    list(tqdm(executor.map(convert_one_image, files_to_process), total=len(files_to_process)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-26T08:08:51.420162Z","iopub.execute_input":"2025-12-26T08:08:51.421028Z","iopub.status.idle":"2025-12-26T08:26:08.108577Z","shell.execute_reply.started":"2025-12-26T08:08:51.420999Z","shell.execute_reply":"2025-12-26T08:26:08.106627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport os\n\n# 1. Nén folder ảnh thành file zip\n# format: 'zip', root_dir: thư mục chứa folder, base_dir: folder cần nén\nprint(\"Đang nén dữ liệu...\")\nshutil.make_archive(\"/kaggle/working/dataset_output_2160\", 'zip', \"/kaggle/working/train(test)_png_1024_1\")\nprint(\"Đã nén xong: /kaggle/working/dataset_output_2160.zip\")\n\n# Sau khi dòng này chạy xong, nếu bạn đang chạy chế độ \"Save Version\", \n# file zip sẽ được lưu lại trên server Kaggle.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-26T08:48:19.939863Z","iopub.execute_input":"2025-12-26T08:48:19.940267Z","iopub.status.idle":"2025-12-26T08:50:01.290693Z","shell.execute_reply.started":"2025-12-26T08:48:19.94024Z","shell.execute_reply":"2025-12-26T08:50:01.288951Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Check shape\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport pydicom\nimport os\nfrom tqdm import tqdm\n\n# --- CẤU HÌNH ---\ncsv_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\"  # File CSV bạn đang dùng\ndicom_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\noutput_csv_path = \"train_v2(with_dims).csv\" # Tên file mới sẽ lưu\n\n# 1. Đọc file CSV\ndf = pd.read_csv(csv_path)\n\n# 2. Lấy danh sách các Image ID duy nhất (để không phải đọc 1 file dicom nhiều lần)\nunique_ids = df['image_id'].unique()\n\n# Dictionary để lưu tạm kích thước: { 'image_id': (height, width) }\nimg_dims = {}\n\nprint(f\"Đang quét kích thước của {len(unique_ids)} file DICOM...\")\n\nfor img_id in tqdm(unique_ids):\n    path = os.path.join(dicom_dir, f\"{img_id}.dicom\")\n    \n    try:\n        # CHỈ ĐỌC HEADER (Siêu nhanh)\n        dicom = pydicom.dcmread(path, stop_before_pixels=True)\n        \n        # Lưu vào dict (Rows là cao, Columns là rộng)\n        img_dims[img_id] = (dicom.Rows, dicom.Columns)\n        \n    except Exception as e:\n        print(f\"Lỗi đọc file {img_id}: {e}\")\n        img_dims[img_id] = (None, None)\n\n# 3. Map ngược lại vào DataFrame gốc\nprint(\"Đang cập nhật vào DataFrame...\")\ndf['height'] = df['image_id'].map(lambda x: img_dims.get(x, (None, None))[0])\ndf['width'] = df['image_id'].map(lambda x: img_dims.get(x, (None, None))[1])\n\n# 4. Lưu file mới\ndf.to_csv(output_csv_path, index=False)\nprint(f\"Xong! File mới đã lưu tại: {output_csv_path}\")\nprint(df[['image_id', 'height', 'width']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T03:50:28.459129Z","iopub.execute_input":"2025-12-21T03:50:28.460362Z","iopub.status.idle":"2025-12-21T03:53:42.290628Z","shell.execute_reply.started":"2025-12-21T03:50:28.460325Z","shell.execute_reply":"2025-12-21T03:53:42.288948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\nimport os\nimport random\n\n# 1. Đọc file\n# ds = pydicom.dcmread(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/000434271f63a053c4128a0ba6352c7f.dicom\")\n\n# 2. Hiển thị ảnh (Bắt buộc phải thêm cmap='gray' để nhìn đúng chất X-ray)\n# plt.imshow(ds.pixel_array, cmap='gray')\n# plt.axis('off') # Tắt trục tọa độ cho đẹp\n# plt.show()\n\nfolder_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\nn = 10  # Số lượng file cần lấy\n\n# 1. Lấy danh sách tất cả file (bỏ qua thư mục con)\nall_files = [f for f in os.listdir(folder_path) if os.path.isfile(os.path.join(folder_path, f))]\n\n# 2. Lấy ngẫu nhiên n file (dùng min để tránh lỗi nếu n > tổng số file)\nselected_files = random.sample(all_files, min(n, len(all_files)))\n\n# 3. Xử lý\nfor filename in selected_files:\n    full_path = os.path.join(folder_path, filename)\n    \n    # --- DO SOMETHING TẠI ĐÂY ---\n    dcm = pydicom.dcmread(full_path)\n    print(dcm.pixel_array.shape)\n    plt.imshow(dcm.pixel_array, cmap='gray')\n    plt.axis('off') # Tắt trục tọa độ cho đẹp\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T17:02:35.081025Z","iopub.execute_input":"2025-12-21T17:02:35.081291Z","iopub.status.idle":"2025-12-21T17:02:53.802237Z","shell.execute_reply.started":"2025-12-21T17:02:35.081277Z","shell.execute_reply":"2025-12-21T17:02:53.801596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\n# --- CẤU HÌNH ---\nINPUT_CSV = \"/kaggle/input/vindr-train-dims/train_v2(with_dims).csv\" # File gốc có cột width, height gốc\nSIZES = [640, 512, 1024] # 3 cỡ bạn cần\n\n# Đọc dữ liệu\nprint(f\"Đang đọc dữ liệu từ {INPUT_CSV}...\")\ndf_original = pd.read_csv(INPUT_CSV)\n\n# Hàm tính toán tọa độ mới\ndef recalculate_box(row, target_size):\n    # Nếu là 'No finding' (Class 14) hoặc dữ liệu rỗng -> Giữ nguyên NaN\n    if row['class_id'] == 14 or pd.isna(row['x_min']):\n        return np.nan, np.nan, np.nan, np.nan\n    \n    # Lấy kích thước gốc (Dựa vào tên cột trong ảnh bạn gửi: 'width', 'height')\n    w_orig = row['width']\n    h_orig = row['height']\n    \n    # 1. Tính Scale\n    scale = min(target_size / w_orig, target_size / h_orig)\n    \n    # 2. Tính Padding\n    nw = int(w_orig * scale)\n    nh = int(h_orig * scale)\n    pad_w = (target_size - nw) / 2\n    pad_h = (target_size - nh) / 2\n    \n    # 3. Tính tọa độ mới (Shift + Scale)\n    xmin_new = row['x_min'] * scale + pad_w\n    ymin_new = row['y_min'] * scale + pad_h\n    xmax_new = row['x_max'] * scale + pad_w\n    ymax_new = row['y_max'] * scale + pad_h\n    \n    # 4. Clip lại cho chắc chắn không văng ra ngoài ảnh\n    xmin_new = np.clip(xmin_new, 0, target_size)\n    ymin_new = np.clip(ymin_new, 0, target_size)\n    xmax_new = np.clip(xmax_new, 0, target_size)\n    ymax_new = np.clip(ymax_new, 0, target_size)\n    \n    return xmin_new, ymin_new, xmax_new, ymax_new\n\n# --- VÒNG LẶP XỬ LÝ 3 SIZE ---\nfor size in SIZES:\n    print(f\"--> Đang xử lý size {size}x{size}...\")\n    \n    # Copy từ file gốc ra để không làm hỏng dữ liệu\n    df_new = df_original.copy()\n    \n    # Áp dụng hàm tính toán (Chạy từng dòng)\n    # Tạo 4 cột mới chứa tọa độ đã resize\n    # Sử dụng apply với axis=1 để xử lý từng hàng\n    new_coords = df_new.apply(lambda row: recalculate_box(row, size), axis=1)\n    \n    # Gán ngược lại vào DataFrame\n    df_new['x_min'] = [x[0] for x in new_coords]\n    df_new['y_min'] = [x[1] for x in new_coords]\n    df_new['x_max'] = [x[2] for x in new_coords]\n    df_new['y_max'] = [x[3] for x in new_coords]\n    \n    # Thêm cột đánh dấu để đỡ nhầm sau này\n    df_new['target_size'] = size\n    \n    # Lưu file\n    filename = f\"train_{size}.csv\"\n    df_new.to_csv(filename, index=False)\n    print(f\"   Đã lưu: {filename}\")\n\nprint(\"\\n=== HOÀN TẤT! ===\")\nprint(\"Bạn đã có 3 file: train_640.csv, train_512.csv, train_1024.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T16:26:52.989164Z","iopub.execute_input":"2025-12-28T16:26:52.989577Z","iopub.status.idle":"2025-12-28T16:27:03.906206Z","shell.execute_reply.started":"2025-12-28T16:26:52.989539Z","shell.execute_reply":"2025-12-28T16:27:03.905125Z"}},"outputs":[],"execution_count":null}]}