{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 데이터 개요\n\n## 데이터 구성 방법\n### 영상은 환자 ID별로 디렉토리에 구성","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"ls ../input/rsna-breast-cancer-detection/train_images | head -1 -n 4","metadata":{"execution":{"iopub.status.busy":"2023-01-08T18:56:51.247924Z","iopub.execute_input":"2023-01-08T18:56:51.248552Z","iopub.status.idle":"2023-01-08T18:56:52.665795Z","shell.execute_reply.started":"2023-01-08T18:56:51.248426Z","shell.execute_reply":"2023-01-08T18:56:52.663716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 각 디렉터리에는 여러 이미지가 들어 있습니다.","metadata":{}},{"cell_type":"code","source":"ls ../input/rsna-breast-cancer-detection/train_images/57175","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:00:46.925015Z","iopub.execute_input":"2023-01-08T19:00:46.925479Z","iopub.status.idle":"2023-01-08T19:00:47.204186Z","shell.execute_reply.started":"2023-01-08T19:00:46.925434Z","shell.execute_reply":"2023-01-08T19:00:47.202957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom matplotlib import pyplot as plt\n\nsample_sub = pd.read_csv('../input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:01:22.349759Z","iopub.execute_input":"2023-01-08T19:01:22.35026Z","iopub.status.idle":"2023-01-08T19:01:22.359665Z","shell.execute_reply.started":"2023-01-08T19:01:22.350217Z","shell.execute_reply":"2023-01-08T19:01:22.358727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:01:28.093927Z","iopub.execute_input":"2023-01-08T19:01:28.094382Z","iopub.status.idle":"2023-01-08T19:01:28.118385Z","shell.execute_reply.started":"2023-01-08T19:01:28.094342Z","shell.execute_reply":"2023-01-08T19:01:28.116811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -l ../input/rsna-breast-cancer-detection/train_images | wc -l \n# wc -l outputs one a count of one row too many when used like this! hence we need to subtract one to get the true count","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:01:41.248671Z","iopub.execute_input":"2023-01-08T19:01:41.249143Z","iopub.status.idle":"2023-01-08T19:01:43.062285Z","shell.execute_reply.started":"2023-01-08T19:01:41.2491Z","shell.execute_reply":"2023-01-08T19:01:43.057115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# In test, we have access to just a single example!","metadata":{}},{"cell_type":"code","source":"ls ../input/rsna-breast-cancer-detection/test_images/10008","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:05:50.420084Z","iopub.execute_input":"2023-01-08T19:05:50.420562Z","iopub.status.idle":"2023-01-08T19:05:50.700404Z","shell.execute_reply.started":"2023-01-08T19:05:50.420521Z","shell.execute_reply":"2023-01-08T19:05:50.6986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"그러나 노트북에서 제출하면 전체 테스트 세트가 마운트됩니다.\n\n작업할 이미지는 DICOM 형식입니다. 좀 더 자세히 살펴봅시다.","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom pathlib import Path\nimport glob","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:06:41.35546Z","iopub.execute_input":"2023-01-08T19:06:41.355928Z","iopub.status.idle":"2023-01-08T19:06:42.16757Z","shell.execute_reply.started":"2023-01-08T19:06:41.355884Z","shell.execute_reply":"2023-01-08T19:06:42.166231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"먼저, 메타데이터부터 시작하겠습니다. 각 이미지에는 교육을 위해 데이터를 분할하는 방법을 안내할 수 있는 풍부한 메타데이터가 포함되어 있습니다(이러한 정보 중 일부를 모델에 직접 제공할 수도 있습니다!)","metadata":{}},{"cell_type":"code","source":"example = '../input/rsna-breast-cancer-detection/train_images/10006/1459541791.dcm'\npydicom.dcmread(example)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:06:48.244394Z","iopub.execute_input":"2023-01-08T19:06:48.244872Z","iopub.status.idle":"2023-01-08T19:06:48.377901Z","shell.execute_reply.started":"2023-01-08T19:06:48.244832Z","shell.execute_reply":"2023-01-08T19:06:48.376733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"이제 몇 가지 이미지를 살펴보겠습니다","metadata":{}},{"cell_type":"code","source":"# source: https://www.kaggle.com/code/allunia/rsna-csf-cervical-spine-fracture-eda/notebook\ndef rescale_img_to_hu(dcm_ds):\n    \"\"\"Rescales the image to Hounsfield unit.\"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:09:43.409254Z","iopub.execute_input":"2023-01-08T19:09:43.409701Z","iopub.status.idle":"2023-01-08T19:09:43.416207Z","shell.execute_reply.started":"2023-01-08T19:09:43.409673Z","shell.execute_reply":"2023-01-08T19:09:43.415193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_images_for_patient(patient_id):\n    patient_dir = os.path.join('../input/rsna-breast-cancer-detection/train_images', str(patient_id))\n    num_images = len(glob.glob(f\"{patient_dir}/*\"))\n    print(f\"Number of images for patient: {num_images}\")\n    fig, axs = plt.subplots(2, 2, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_path in enumerate(list(Path(patient_dir).iterdir())):\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:09:50.097602Z","iopub.execute_input":"2023-01-08T19:09:50.09807Z","iopub.status.idle":"2023-01-08T19:09:50.107099Z","shell.execute_reply.started":"2023-01-08T19:09:50.098034Z","shell.execute_reply":"2023-01-08T19:09:50.10508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_images_for_patient(10006)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:09:55.692372Z","iopub.execute_input":"2023-01-08T19:09:55.692851Z","iopub.status.idle":"2023-01-08T19:10:08.875515Z","shell.execute_reply.started":"2023-01-08T19:09:55.692814Z","shell.execute_reply":"2023-01-08T19:10:08.874498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"그림에서 알 수 있듯이 이미지의 크기가 상당히 큽니다.\n\n우리는 유방이 이미지의 아주 작은 부분만을 차지하는 것을 볼 수 있다.\n\n따라서(특히 제한된 컴퓨팅 예산을 고려할 때) 정보가 포함되지 않은 이미지 부분을 잘라내는 방법을 찾는 것이 매우 좋은 방법일 것입니다! 여기서 구분선은 중요한 역할을 할 수 있다.","metadata":{}},{"cell_type":"markdown","source":"경쟁 지표 »\n우리 모델이 전달해야 할 예측 유형을 정확하게 이해하는 것은 항상 좋은 생각이다.\n\n여기서 주최자가 선택한 메트릭은 확률적 F1 점수입니다:\n\n여기에서 메트릭의 파이썬 구현을 찾을 수 있습니다\n\n우리의 모델은 해당 이미지에서 암 가능성을 출력해야 한다.\n\n그렇다면 우리가 훈련할 라벨은 무엇일까요?","metadata":{}},{"cell_type":"code","source":"train_csv = pd.read_csv('../input/rsna-breast-cancer-detection/train.csv')\ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:11:31.578978Z","iopub.execute_input":"2023-01-08T19:11:31.579433Z","iopub.status.idle":"2023-01-08T19:11:31.669147Z","shell.execute_reply.started":"2023-01-08T19:11:31.579397Z","shell.execute_reply":"2023-01-08T19:11:31.668155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv = pd.read_csv('../input/rsna-breast-cancer-detection/test.csv')\ntest_csv","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:11:40.069763Z","iopub.execute_input":"2023-01-08T19:11:40.070237Z","iopub.status.idle":"2023-01-08T19:11:40.093786Z","shell.execute_reply.started":"2023-01-08T19:11:40.0702Z","shell.execute_reply":"2023-01-08T19:11:40.092668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"예상대로 검사 파일에 암 열이 없습니다.\n\n그래서 우리는 얼마나 많은 이미지를 훈련해야 할까요?","metadata":{}},{"cell_type":"code","source":"train_csv.shape[0], train_csv.patient_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:11:47.959433Z","iopub.execute_input":"2023-01-08T19:11:47.959925Z","iopub.status.idle":"2023-01-08T19:11:47.976345Z","shell.execute_reply.started":"2023-01-08T19:11:47.959883Z","shell.execute_reply":"2023-01-08T19:11:47.974789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"11913명의 환자를 대상으로 총 54706장의 영상이 열차에 실려 있다.","metadata":{}},{"cell_type":"markdown","source":"# 레이블 분포","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,6))\nplt.subplot(1,2,1)\nax1 = sns.countplot(data=train_csv, x='cancer')\nfor container in ax1.containers:\n    ax1.bar_label(container)\nplt.title('Distribution of targets');","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:13:01.634676Z","iopub.execute_input":"2023-01-08T19:13:01.635136Z","iopub.status.idle":"2023-01-08T19:13:01.783305Z","shell.execute_reply.started":"2023-01-08T19:13:01.635096Z","shell.execute_reply":"2023-01-08T19:13:01.782555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"그 수업들은 매우 불균형해요! 대부분의 경우, 우리는 훈련에서 이 문제를 해결해야 할 것이다(아마도 암 클래스의 업샘플링, 암이 없는 경우의 다운샘플링, 클래스 가중치 등을 통해).","metadata":{}},{"cell_type":"markdown","source":"# Training a fast.ai model","metadata":{}},{"cell_type":"markdown","source":"이제 fast.ai 모델을 시험해 보는 데 필요한 모든 요소를 배치해 보겠습니다\n\n신속한 실험을 용이하게 하기 위해 열차 이미지를 PNG로 처리하여 RSNA 유방 촬영 - 이미지를 PNG로 공유했습니다(256px & 512px).\n\nfast.ai 라이브러리를 빠르게 시작하고 실행하는 데 필요한 모든 것이 여기에 있습니다.","metadata":{}},{"cell_type":"markdown","source":"# Creating dataloaders","metadata":{}},{"cell_type":"code","source":"!pip install -U pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg pydicom","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:16:36.477553Z","iopub.execute_input":"2023-01-08T19:16:36.47803Z","iopub.status.idle":"2023-01-08T19:16:50.131156Z","shell.execute_reply.started":"2023-01-08T19:16:36.477995Z","shell.execute_reply":"2023-01-08T19:16:50.129442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.data.all import *\nfrom fastai.vision.all import *\n\npath = '../input/rsna-mammography-images-as-pngs/images_as_pngs/train_images_processed'\n\ntrain_csv = pd.read_csv('../input/rsna-breast-cancer-detection/train.csv')\nfn2label = {fn: cancer_or_not for fn, cancer_or_not in zip(train_csv['image_id'].astype('str'), train_csv['cancer'])}\n\ndef label_func(path):\n    return fn2label[path.stem]\n\ndblock = DataBlock(\n    blocks    = (ImageBlock, CategoryBlock),\n    get_items = get_image_files,\n    get_y = label_func,\n    splitter  = RandomSplitter()\n)\ndsets = dblock.datasets(path)\ndls = dblock.dataloaders(path)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:23:26.016773Z","iopub.execute_input":"2023-01-08T19:23:26.017223Z","iopub.status.idle":"2023-01-08T19:23:26.138428Z","shell.execute_reply.started":"2023-01-08T19:23:26.017186Z","shell.execute_reply":"2023-01-08T19:23:26.136425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T19:22:35.471328Z","iopub.execute_input":"2023-01-08T19:22:35.471748Z","iopub.status.idle":"2023-01-08T19:22:35.496323Z","shell.execute_reply.started":"2023-01-08T19:22:35.471719Z","shell.execute_reply":"2023-01-08T19:22:35.494565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}