{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n        \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-17T01:29:24.291302Z","iopub.execute_input":"2025-08-17T01:29:24.292083Z","iopub.status.idle":"2025-08-17T01:29:24.726392Z","shell.execute_reply.started":"2025-08-17T01:29:24.292051Z","shell.execute_reply":"2025-08-17T01:29:24.725495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport pandas as pd\nimport os\n\ndef process_rsna(link, root):\n    \"\"\"\n    Load dataframe từ file CSV và thêm cột link = patient_id + \"_\" + image_id\n    \n    Parameters:\n        link (str): đường dẫn file CSV\n        root (str): thư mục gốc (nếu cần để join path)\n    \n    Returns:\n        pd.DataFrame: dataframe đã thêm cột link\n    \"\"\"\n    # Load CSV\n    file_path = os.path.join(root, link)\n    df = pd.read_csv(link)\n    df = df[df['biopsy'] == 1].copy()\n    \n    # Tạo cột link = patient_id + \"_\" + image_id\n    df['name']  = df['patient_id'].astype(str) + \"_\" + df['image_id'].astype(str)\n    df['link']  = df['patient_id'].astype(str) + \"/\" + df['image_id'].astype(str)\n    df['link1'] =  \"/train_images/\"+ df['link'] + \".dcm\"\n    df['link2'] =  \"/kaggle/working/RSNA/\"+ df['patient_id'].astype(str)+ \"/\" + df['name'] + \"_\" + df['view'].astype(str)+ \"_\" + df['laterality'].astype(str) + \".png\"\n    return df\n\ndf = process_rsna(link=\"/kaggle/input/rsna-breast-cancer-detection/train.csv\", \n                  root= \"/kaggle/input/rsna-breast-cancer-detection\")\ndf ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T01:29:24.728114Z","iopub.execute_input":"2025-08-17T01:29:24.72861Z","iopub.status.idle":"2025-08-17T01:29:24.855411Z","shell.execute_reply.started":"2025-08-17T01:29:24.728573Z","shell.execute_reply":"2025-08-17T01:29:24.854205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install -q \"pydicom>=2.3\" \"pylibjpeg>=2.0\" \"pylibjpeg-libjpeg>=2.1\" \"pylibjpeg-openjpeg\" \"python-gdcm>=3.0.10\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T01:29:24.856656Z","iopub.execute_input":"2025-08-17T01:29:24.85691Z","iopub.status.idle":"2025-08-17T01:29:28.62411Z","shell.execute_reply.started":"2025-08-17T01:29:24.856891Z","shell.execute_reply":"2025-08-17T01:29:28.62314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_modality_lut, apply_voi_lut\nfrom PIL import Image\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\nfrom tqdm import tqdm\n\n# ----- utils giữ nguyên ý tưởng trước -----\ndef _to_uint8(imgf32, mono1=False):\n    imgf32 = imgf32 - np.nanmin(imgf32)\n    maxv = np.nanmax(imgf32)\n    if maxv > 0:\n        imgf32 = imgf32 / maxv * 255.0\n    img = np.clip(imgf32, 0, 255).astype(np.uint8)\n    if mono1:\n        img = 255 - img\n    return img\n\ndef _dicom_to_uint8(ds):\n    arr = ds.pixel_array.astype(np.float32)  # cần pylibjpeg/gdcm cho ảnh nén\n    try:  arr = apply_modality_lut(arr, ds).astype(np.float32)\n    except Exception:  pass\n    try:  arr = apply_voi_lut(arr, ds).astype(np.float32)\n    except Exception:  pass\n    mono1 = str(getattr(ds, \"PhotometricInterpretation\", \"\")).upper() == \"MONOCHROME1\"\n    return _to_uint8(arr, mono1)\n\ndef _resize_keep_w(img_u8: np.ndarray, target_w=2048) -> Image.Image:\n    h, w = img_u8.shape[:2]\n    if w == target_w: \n        return Image.fromarray(img_u8)\n    target_h = int(round(h * (target_w / float(w))))\n    return Image.fromarray(img_u8).resize(\n        (target_w, target_h), resample=Image.Resampling.LANCZOS\n    )\n\n# ----- worker chạy cho từng dòng -----\ndef _process_one(root, link1, link2, target_w=2048):\n    link1 = str(link1).lstrip(os.sep)\n    link2 = str(link2).lstrip(os.sep)\n    in_path  = os.path.join(root, link1)\n    out_path = str(Path(link2).with_suffix(\".png\"))  # link2 là đường dẫn tương đối/đích\n\n    if not os.path.exists(in_path):\n        return (\"MISSING\", in_path)\n\n    try:\n        ds  = pydicom.dcmread(in_path, force=True)\n        img = _dicom_to_uint8(ds)\n        pil = _resize_keep_w(img, target_w=target_w)\n        Path(out_path).parent.mkdir(parents=True, exist_ok=True)\n        pil.save(out_path)\n        return (\"OK\", out_path)\n    except Exception as e:\n        return (\"ERROR\", f\"{in_path} :: {e}\")\n\ndef Process_RSNA_parallel(df, root=\"\", max_workers=None, target_w=2048):\n    \"\"\"\n    df: có cột link1 (.dcm trong 'root') và link2 (đường đích png)\n    \"\"\"\n    # nơi đặt metadata.csv: thư mục cấp 1 của link2\n    first_folder = Path(str(df.iloc[0][\"link2\"])).parts[0] if len(df) else \"\"\n    meta_base = os.path.join(root, first_folder) if first_folder else root\n    Path(meta_base).mkdir(parents=True, exist_ok=True)\n\n    tasks = []\n    results = {\"OK\":0, \"MISSING\":0, \"ERROR\":0}\n    with ThreadPoolExecutor(max_workers=max_workers) as ex:\n        for row in df.itertuples(index=False):\n            fut = ex.submit(_process_one, root, getattr(row,\"link1\"), getattr(row,\"link2\"), target_w)\n            tasks.append(fut)\n\n        for fut in tqdm(as_completed(tasks), total=len(tasks)):\n            status, info = fut.result()\n            results[status] += 1\n            if status != \"OK\":\n                print(status, \"->\", info)\n\n    # lưu metadata sau khi chạy\n    pd.DataFrame(df).to_csv(os.path.join(meta_base, \"metadata.csv\"), index=False)\n    print(\"Summary:\", results)\n    return results           \n#/kaggle/input/rsna-breast-cancer-detection/train_images\n#/kaggle/input/rsna-breast-cancer-detection/train_images/10011/1031443799.dcm\nProcess_RSNA_parallel(df, root=\"/kaggle/input/rsna-breast-cancer-detection\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T01:29:48.360066Z","iopub.execute_input":"2025-08-17T01:29:48.360934Z","iopub.status.idle":"2025-08-17T02:13:17.458989Z","shell.execute_reply.started":"2025-08-17T01:29:48.360901Z","shell.execute_reply":"2025-08-17T02:13:17.456857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T02:13:24.86349Z","iopub.execute_input":"2025-08-17T02:13:24.864042Z","iopub.status.idle":"2025-08-17T02:13:24.889665Z","shell.execute_reply.started":"2025-08-17T02:13:24.864008Z","shell.execute_reply":"2025-08-17T02:13:24.888657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n#df = pd.DataFrame({\"a\": [1,2], \"b\": [3,4]})\nlink = \"/kaggle/working/kaggle/working/metadata.csv\"\n\n# make sure parent folder exists\nimport os\nfrom pathlib import Path\n#Path(link).parent.mkdir(parents=True, exist_ok=True)\n\ndf.to_csv(link, index=False)\nprint(\"Saved:\", link)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T02:13:34.758017Z","iopub.execute_input":"2025-08-17T02:13:34.758315Z","iopub.status.idle":"2025-08-17T02:13:34.79798Z","shell.execute_reply.started":"2025-08-17T02:13:34.758294Z","shell.execute_reply":"2025-08-17T02:13:34.797018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport os\n\n%cd /kaggle/working/kaggle\n\nfolder_to_zip = \"working\"\noutput_zip_file = \"RSNA_working.zip\"\n\nshutil.make_archive(output_zip_file.replace(\".zip\", \"\"), 'zip', folder_to_zip)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-17T02:13:57.088203Z","iopub.execute_input":"2025-08-17T02:13:57.088923Z","iopub.status.idle":"2025-08-17T02:17:04.713193Z","shell.execute_reply.started":"2025-08-17T02:13:57.08889Z","shell.execute_reply":"2025-08-17T02:17:04.712047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}