{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EDA notebook\ncompetition : RSNA Screening Mammography Breast Cancer Detection  \nurl : https://www.kaggle.com/competitions/rsna-breast-cancer-detection","metadata":{}},{"cell_type":"markdown","source":"reference notebook : https://www.kaggle.com/code/craigmthomas/rsna-2022-eda","metadata":{}},{"cell_type":"markdown","source":"# 0. import and Data loading","metadata":{}},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg\n!pip install japanize-matplotlib\n\nimport os\nimport copy\nimport random\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport japanize_matplotlib\nimport matplotlib.pyplot as plt\n\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\npd.set_option('display.max_columns', 50)","metadata":{"jupyter":{"outputs_hidden":true,"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:48:05.513311Z","iopub.execute_input":"2023-01-28T00:48:05.514064Z","iopub.status.idle":"2023-01-28T00:48:35.140902Z","shell.execute_reply.started":"2023-01-28T00:48:05.51397Z","shell.execute_reply":"2023-01-28T00:48:35.139856Z"},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/rsna-breast-cancer-detection/train.csv\")\ntest = pd.read_csv(\"../input/rsna-breast-cancer-detection/test.csv\")\n\nprint('train')\ndisplay(train)\nprint('test')\ndisplay(test)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:48:35.142778Z","iopub.execute_input":"2023-01-28T00:48:35.143993Z","iopub.status.idle":"2023-01-28T00:48:35.295195Z","shell.execute_reply.started":"2023-01-28T00:48:35.143951Z","shell.execute_reply":"2023-01-28T00:48:35.293867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('---train---')\nprint(f'train.shape : {train.shape[0]}rows, {train.shape[1]}cols')\nprint(f'train.patient_id.nunique() : {train.patient_id.nunique()}')\nprint(f'train.image_id.nunique() : {train.image_id.nunique()}')\nprint(f'target col : cancer')\nprint(f'columns not in test : {set(train.columns) - set(test.columns)}')\ndisplay(train.describe(include='all'))\nprint(f'train missing values : ')\nprint(train.isnull().sum())\nprint('')\nprint('---test---')\nprint(f'test.shape : {test.shape[0]}rows, {test.shape[1]}cols')\nprint(f'columns not in train : {set(test.columns)-set(train.columns)}')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:48:35.29709Z","iopub.execute_input":"2023-01-28T00:48:35.29754Z","iopub.status.idle":"2023-01-28T00:48:35.397746Z","shell.execute_reply.started":"2023-01-28T00:48:35.2975Z","shell.execute_reply":"2023-01-28T00:48:35.396623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ここまででわかったこと**  \n* trainには54706行存在しているが、患者数は11913である。１人の患者さんにつき複数のデータがある。\n* trainにはtestにはないカラムが存在している。（testにもtrainにないカラムが一つ存在しているが、submit時に使用するだけで、あまり関係ない）\n* testは１人の患者さんの４枚の画像からガンがあるかどうかを予測する。２つのview（？）で左右の胸の画像がある。よって計４画像\n* 欠損値はあるが、ageだけtestデータに含まれているので、ageの処理を考えればOK、たぶん\n  \n**これからやること**\n* 14つの特徴について、データの分布やそのデータの意味を理解する。\n* testにないカラムも調べ、予測に使えるかどうかを考える。\n","metadata":{}},{"cell_type":"markdown","source":"# 1. image data","metadata":{}},{"cell_type":"code","source":"dcm_count = 0\nfor cdir, dirs, files in os.walk('../input/rsna-breast-cancer-detection/train_images'):\n    for f in files:\n        if f.split('.')[1] == 'dcm':\n            dcm_count += 1\nprint(f'number of images .dcm : {dcm_count}')\nprint(f'number of images : {train.image_id.nunique()}')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:48:35.400561Z","iopub.execute_input":"2023-01-28T00:48:35.400965Z","iopub.status.idle":"2023-01-28T00:49:04.855019Z","shell.execute_reply.started":"2023-01-28T00:48:35.400924Z","shell.execute_reply":"2023-01-28T00:49:04.854007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"すべての画像は.dcm形式であると考えられる。そもそも.dcm形式とはなんなのか  \n  \n**.DCM**  \n>  DCMの画像フォーマットもDICOM画像フォーマットを開発し全国電機製造業者協会（NEMA）によって開発された。 NEMAは、DICOM医用画像保管、流通および分析のための標準仕様としてDCM形式を開発しました。 DCMフォーマットは、超音波画像、MRI（磁気共鳴画像）などのうちのCT（コンピュータ断層撮影）スキャンシートを含むことができる画像を保存するために使用される。これらDCMファイルの内容は、患者の氏名等の患者の詳細、および他の関連する医療データをも含むことができる。https://www.reviversoft.com/ja/file-extensions/dcm  \n\n今回はMRI画像であるため、.dcm形式の画像である。また、.dcmには患者の氏名等の患者の詳細、およびほかの関連する医療データを含んでいるため、特徴として使える可能性がある。  \npythonで.dcmを扱うにはpydicomライブラリが必要である。","metadata":{}},{"cell_type":"code","source":"file = pydicom.dcmread('/kaggle/input/rsna-breast-cancer-detection/train_images/10006/1459541791.dcm')\nprint(file)\nimg = file.pixel_array\nplt.imshow(img)\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:49:04.856304Z","iopub.execute_input":"2023-01-28T00:49:04.857036Z","iopub.status.idle":"2023-01-28T00:49:08.10544Z","shell.execute_reply.started":"2023-01-28T00:49:04.856995Z","shell.execute_reply":"2023-01-28T00:49:08.104389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_df = pd.read_csv('/kaggle/input/rsna-dicom-csv/dicom.csv').drop(columns=\"Unnamed: 0\")\ndicom_df.head()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:49:08.107114Z","iopub.execute_input":"2023-01-28T00:49:08.107813Z","iopub.status.idle":"2023-01-28T00:49:08.571478Z","shell.execute_reply.started":"2023-01-28T00:49:08.10777Z","shell.execute_reply":"2023-01-28T00:49:08.570364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"途中","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. train feature\ntrainカラムについてそれぞれ見ていく","metadata":{}},{"cell_type":"code","source":"train.corr()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:49:08.573387Z","iopub.execute_input":"2023-01-28T00:49:08.574163Z","iopub.status.idle":"2023-01-28T00:49:08.616707Z","shell.execute_reply.started":"2023-01-28T00:49:08.574118Z","shell.execute_reply":"2023-01-28T00:49:08.615475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.1 site_id","metadata":{}},{"cell_type":"markdown","source":"`site_id` - ID code for the source hospital.(ソース病院の ID コード。)\n* 病院によってがん発見率が変わる？\n* 特徴として入れるのはよくない（のか？）","metadata":{}},{"cell_type":"code","source":"p = sns.countplot(x='site_id', data=train)\np.set(title='trainデータにおけるsite_idの棒グラフ')\nplt.show()\n\n_ = train.groupby('site_id')['cancer'].value_counts()\ntmp = pd.DataFrame(index=['site_id_1', 'site_id_2'], data={'0': [_[1][0], _[2][0]], '1': [_[1][1], _[2][1]] })\ntmp['sum'] = tmp['0'] + tmp['1']\ntmp['1/sum'] = tmp['1'] / tmp['sum']\ndisplay(tmp)\n\np = sns.countplot(x='site_id', data=train, hue='cancer')\np.set(title='trainデータにおけるsite_idの棒グラフ(hue=cancer)')\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:49:08.618675Z","iopub.execute_input":"2023-01-28T00:49:08.619109Z","iopub.status.idle":"2023-01-28T00:49:09.042344Z","shell.execute_reply.started":"2023-01-28T00:49:08.61907Z","shell.execute_reply":"2023-01-28T00:49:09.041338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`site_id`はとりあえず無視する。  \n見た感じ、どちらの病院のほうががん発見率が高い、といったことはない。\n\n追記：  \nディスカッションでsite_idごとのモデルを作成していた  \nsite_idごとにモデルを分けたほうがいいのか","metadata":{}},{"cell_type":"markdown","source":"# 2.2 patient_id, image_id","metadata":{}},{"cell_type":"markdown","source":"`patient_id, image_id`に関してはただのIDであるため無視","metadata":{}},{"cell_type":"markdown","source":"# 2.3 laterality","metadata":{}},{"cell_type":"markdown","source":"`laterality` - Whether the image is of the left or right breast.（画像が左胸か右胸か）","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1, ncols=2, figsize =(20,6), tight_layout=True)\nsns.set()\n# axs = axs.flatten()\n\nsns.countplot(x='laterality', data=train, ax=axs[0])\naxs[0].set_title('laterality', fontsize=18)\n\nfor p in axs[0].patches:\n    axs[0].annotate(format(p.get_height()), \n                   (p.get_x() + p.get_width() / 2., \n                    p.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\n\nsns.countplot(x='laterality', data=train, hue='cancer', ax=axs[1])\naxs[1].set_title('laterality(hue=cancer)', fontsize=18)\nfor q in axs[1].patches:\n    axs[1].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-01-28T00:49:09.043977Z","iopub.execute_input":"2023-01-28T00:49:09.044353Z","iopub.status.idle":"2023-01-28T00:49:09.52341Z","shell.execute_reply.started":"2023-01-28T00:49:09.044305Z","shell.execute_reply":"2023-01-28T00:49:09.52235Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"左右でがんの発見に相関はなさそう","metadata":{}},{"cell_type":"markdown","source":"# 2.4 view\n`view` - The orientation of the image. The default for a screening exam is to capture two views per breast.(画像の向き。スクリーニング検査のデフォルトでは、乳房ごとに 2 つのビューをキャプチャします。)","metadata":{}},{"cell_type":"code","source":"display(train.view.value_counts())\nfig ,axs = plt.subplots(nrows=1, ncols=2, figsize=(20,8))\nsns.set()\n\nsns.countplot(x='view', data=train, ax=axs[0])\naxs[0].set(title='view')\nfor q in axs[0].patches:\n    axs[0].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\n\nsns.countplot(x='view', data=train, hue='cancer', ax=axs[1])\naxs[1].set(title='view(hue=cacer)')\n\nfor q in axs[1].patches:\n    axs[1].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\n\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:49:09.527691Z","iopub.execute_input":"2023-01-28T00:49:09.528012Z","iopub.status.idle":"2023-01-28T00:49:10.066386Z","shell.execute_reply.started":"2023-01-28T00:49:09.527983Z","shell.execute_reply":"2023-01-28T00:49:10.065369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* CCとMLOが圧倒的に多い\n* それぞれのviewの画像を見る必要がありそう\n* ML,LM,LMOに関してはがんの発見がない\n","metadata":{}},{"cell_type":"code","source":"def get_dcm_img(patient_id, image_id, tr_or_te='train_images'):\n    path = '/kaggle/input/rsna-breast-cancer-detection/'+tr_or_te+'/'+str(patient_id)+'/'+str(image_id)+'.dcm'\n    f = pydicom.dcmread(path)\n    img = f.pixel_array\n    return img.tolist()\n\nfig, axs = plt.subplots(nrows=2, ncols=3, figsize=(18,10), tight_layout=True)\naxs = axs.flatten()\nfor i, view_ in enumerate(train.view.unique().tolist()):\n    t = train[train.view==view_].iloc[0]\n    p_id = t.patient_id\n    i_id = t.image_id\n    l_or_r = t.laterality\n    cancer = t.cancer\n    img = get_dcm_img(p_id, i_id)\n    axs[i].imshow(img)\n    axs[i].set_title(f'view=[{view_}] : laterality=[{l_or_r}] : cancer=[{cancer}]')\n    axs[i].axis('off')\n    \n    \nplt.show()\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:49:10.068199Z","iopub.execute_input":"2023-01-28T00:49:10.068919Z","iopub.status.idle":"2023-01-28T00:49:38.386822Z","shell.execute_reply.started":"2023-01-28T00:49:10.06888Z","shell.execute_reply":"2023-01-28T00:49:38.385726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 画像を見ただけでは、大きな違いはわからない\n* CC,MLOと他では画像サイズが違う（？）、背景色も異なっている\n* CC,MLOには縦のラインが引かれている\n\n以下、[参考ノートブック](https://www.kaggle.com/code/craigmthomas/rsna-2022-eda#3.1---View-Feature)より\n* `MLO` - Mediolateral Oblique View\n    * ほとんどの乳房組織をキャプチャします。胸筋がビューに含まれており、適切な患者のポジショニングと全体的な画質を評価するためのガイドとして使用されます。MLO ビューは下向きに撮影されていますが、胸の中心から外側を見るように角度が付けられています。\n* `CC` - Craniocaudal View\n    * 内外側斜めビューと同様に、胸筋がビューに含まれる場合があります。これは、適切な患者のポジショニングを評価するためのガイドとして使用されます。ただし、主な違いは、Crainocaudal ビューが乳房の上からまっすぐ下を向いて撮影されることです (つまり、斜めビューのように角度が導入されていません)。\n  \nMLOビューとCCビューはどちらも標準ビューとして知られている。これらのビューは通常のスクリーニング（症状が現れる前に病気の兆候を発見するための試み）で最も一般的に使用されるビューだが、疾患プロセスが存在する場合などは他のビューが使用されることもある。４０歳未満の場合、左と右のMLOのみを実施して全体的な放射線被爆を減らすことができる。これは、ほとんどの乳房組織を適切に捉えるためである。  \n\n他のビューについて：  \n* `AT` - unknown\n    * 調査中\n* `ML` - Mediolateral View\n    * 胸の間の胸の中心から外側に向けて開く。通常、MLOビューが撮影されていないか、撮影できない場合に使用される。これは、斜めビューが利用できない場合に好まれるビュー。ほとんどの病気のプロセスは乳房の外側で発生し、したがってフィルムに近くなり、病理のより鮮明な画像が可能になる。\n* `LM` - Lateromedial View\n    * ビューが胸に向かって内側を指している腕から取られることを除いて、MLビューに似ている。このビューは、乳房の外側に病変が発生する傾向があるため、理想的ではない。\n* `LMO` - Lateromedial Oblique View\n    * 体の外側から内側を向いている点を除いて、MLO に似ている。\n\n\nそれぞれのviewについて画像を何枚か見ていく必要がありそう  \n今回は画像処理がコンペの鍵だと考えられるため、深堀したほうがいい（？）","metadata":{}},{"cell_type":"markdown","source":"# 2.4.1 CC view\n* 上下で挟んで撮影","metadata":{}},{"cell_type":"code","source":"def show_view(view_name):\n    fig, axs = plt.subplots(nrows=2, ncols=4, figsize=(15,8))\n    axs = axs.flatten()\n    tmp = train[train.view==view_name]\n    for i in range(8):\n        try:\n            n = random.randint(0, tmp.shape[0])\n            t = tmp.iloc[n]\n            p_id = t.patient_id\n            i_id = t.image_id\n            l_or_r = t.laterality\n            cancer = t.cancer\n            img = get_dcm_img(p_id, i_id)\n            axs[i].imshow(img)\n            axs[i].set_title(f'laterality=[{l_or_r}] : cancer=[{cancer}]')\n            axs[i].axis('off')\n        except:\n            pass\n    fig.suptitle(f'{view_name} (random)', fontsize=20)\nshow_view('CC')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:49:38.387975Z","iopub.execute_input":"2023-01-28T00:49:38.388361Z","iopub.status.idle":"2023-01-28T00:50:05.042851Z","shell.execute_reply.started":"2023-01-28T00:49:38.388321Z","shell.execute_reply":"2023-01-28T00:50:05.041641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.4.2 MLO view\n* 縦方向で少し斜めに挟んで撮影","metadata":{}},{"cell_type":"code","source":"show_view('MLO')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:50:05.044759Z","iopub.execute_input":"2023-01-28T00:50:05.04519Z","iopub.status.idle":"2023-01-28T00:50:34.127683Z","shell.execute_reply.started":"2023-01-28T00:50:05.045149Z","shell.execute_reply":"2023-01-28T00:50:34.126777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.4.3 AT view","metadata":{}},{"cell_type":"code","source":"show_view('AT')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:50:34.129082Z","iopub.execute_input":"2023-01-28T00:50:34.130264Z","iopub.status.idle":"2023-01-28T00:50:55.96441Z","shell.execute_reply.started":"2023-01-28T00:50:34.130218Z","shell.execute_reply":"2023-01-28T00:50:55.963301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.4.4 ML view","metadata":{}},{"cell_type":"code","source":"show_view('ML')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:50:55.966435Z","iopub.execute_input":"2023-01-28T00:50:55.966953Z","iopub.status.idle":"2023-01-28T00:51:12.692489Z","shell.execute_reply.started":"2023-01-28T00:50:55.966899Z","shell.execute_reply":"2023-01-28T00:51:12.691279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.4.5 LM view","metadata":{}},{"cell_type":"code","source":"show_view('LM')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:12.69389Z","iopub.execute_input":"2023-01-28T00:51:12.694765Z","iopub.status.idle":"2023-01-28T00:51:28.636315Z","shell.execute_reply.started":"2023-01-28T00:51:12.69471Z","shell.execute_reply":"2023-01-28T00:51:28.635313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.4.6 LMO view\n* LMOは一枚しかない","metadata":{}},{"cell_type":"code","source":"show_view('LMO')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:28.638332Z","iopub.execute_input":"2023-01-28T00:51:28.63881Z","iopub.status.idle":"2023-01-28T00:51:46.616618Z","shell.execute_reply.started":"2023-01-28T00:51:28.638758Z","shell.execute_reply":"2023-01-28T00:51:46.615468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.5 age\n`age` - 患者の年齢。","metadata":{}},{"cell_type":"code","source":"fig ,axs = plt.subplots(nrows=2, ncols=2, figsize=(15,10), tight_layout=True)\naxs = axs.flatten()\nsns.histplot(data=train, x='age', ax=axs[0])\naxs[0].set_title('age histgram(all)')\n\nsns.boxplot(data=train, y='age', ax=axs[1])\naxs[1].set_title('age boxplot(all)')\n\nsns.histplot(data=train[train.cancer==1], x='age', ax=axs[2], color='orange')\naxs[2].set_title('age histgram(cancer==1)')\n\nsns.boxplot(data=train, y='age' , ax=axs[3], x='cancer' )\naxs[3].set_title('age boxplot(hue=cancer)')\n\nplt.show()\n\nprint(train.age.describe())\nprint(f'ガンと診断された人の最低年齢：{train[train.cancer==1].age.min()}歳')\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:46.618534Z","iopub.execute_input":"2023-01-28T00:51:46.618919Z","iopub.status.idle":"2023-01-28T00:51:47.77542Z","shell.execute_reply.started":"2023-01-28T00:51:46.618882Z","shell.execute_reply":"2023-01-28T00:51:47.774288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* ガンと診断された人の年齢は高い傾向にある。\n* 特徴としては十分に使用できると思う。\n* ageには欠損値が存在するため、処理を考える必要がある。","metadata":{}},{"cell_type":"code","source":"print(f'age欠損値数：{train.age.isnull().sum()}')\nprint('ageの欠損値を含む行')\ndisplay(train[train.age.isnull()])","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:47.78021Z","iopub.execute_input":"2023-01-28T00:51:47.7829Z","iopub.status.idle":"2023-01-28T00:51:47.832021Z","shell.execute_reply.started":"2023-01-28T00:51:47.782855Z","shell.execute_reply":"2023-01-28T00:51:47.830763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"気になった点：\n* site_idはすべて１\n* machine_idはすべて49\n* viewはCCかMLO\n\n今のところは平均か中央値か、、、  \nageについてのディスカッションを探したほうがはやそう","metadata":{}},{"cell_type":"markdown","source":"# 2.6 cancer\n**目的変数**\n\n1が陽性,0が陰性（というか見つからなかった）","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1, ncols=2, figsize=(15,5))\nsns.countplot(data=train, x='cancer', ax=axs[0])\naxs[0].set_title('cancer')\n\nfor q in axs[0].patches:\n    axs[0].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\n\nsns.set()\nlabel = ['no finding', 'cancer']\naxs[1].pie(\n    train.cancer.value_counts(),\n    labels=label,\n    startangle=45,\n    autopct='%1.1f%%',\n    pctdistance=0.7,\n    radius=1.5,\n    explode=[0, 0.2],\n    textprops={ 'size': 22, 'color': 'black'}\n)\n\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:47.836726Z","iopub.execute_input":"2023-01-28T00:51:47.837322Z","iopub.status.idle":"2023-01-28T00:51:48.180376Z","shell.execute_reply.started":"2023-01-28T00:51:47.837281Z","shell.execute_reply":"2023-01-28T00:51:48.179333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* めちゃめちゃ不均衡データ\n* アンダーサンプリング or オーバーサンプリング\n* アンダーサンプリングが無難だが、画像をうまく水増しできればオーバーサンプリングも可能か\n\nほかにも不均衡データの対処法があるっぽい[url](https://qiita.com/tk-tatsuro/items/10e9dbb3f2cf030e2119)\n\nとりあえずはアンダーサンプリングでモデルの学習を行い、余裕があれば画像データを水増ししてcancerのデータを２倍、３倍にして学習してみて、結果を比較するのもありかも","metadata":{}},{"cell_type":"markdown","source":"# 2.7 biopsy\n`biopsy` - 乳房のフォローアップ生検が実施されたかどうか。  \n* trainにのみある特徴  \n* biopsy(生検)とは？\n\n> 病理医による検査のために細胞または組織を採取すること。病理医はその組織を顕微鏡で調べたり、その細胞または組織に対して他の検査を実施したりする。\n* biopsyを受けたほうがガンの発見がしやすい？","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(5,5))\nsns.countplot(data=train, x='biopsy', hue='cancer', ax=ax)\nfor q in ax.patches:\n    ax.annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\n\nax.set_title('biopsy (hue=cancer)')\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:48.182204Z","iopub.execute_input":"2023-01-28T00:51:48.182915Z","iopub.status.idle":"2023-01-28T00:51:48.422357Z","shell.execute_reply.started":"2023-01-28T00:51:48.182876Z","shell.execute_reply":"2023-01-28T00:51:48.421423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"グラフから、\n* biopsyが０（受けていない）の人で陽性と診断された人はいない\n* １（受けた）の人ではだいたい半々の割合で陽性と診断されている\n* というかbiopsyはがんの陽性を確定するために行われるらしい\n\nとりあえず放置","metadata":{}},{"cell_type":"markdown","source":"# 2.8 invasive\n`invasive` - 乳房ががんに対して陽性である場合、がんが浸潤性であることが判明したかどうか。\n* trainのみ\n* 非浸潤がん\n> がんが、最初に発生する乳管・小葉にまだとどまっている状態のものです。\nがんを取り切ることができれば、ほとんどで完治が見込まれます。ただし非浸潤がんであってもがんの範囲が広い場合は、がんを取り切るために乳房をすべて切除しなければならないこともあります。\n\n* 浸潤がん\n> がんが乳管・小葉を越えて、乳管の外の間質にまで広がっているものです。\nがんの進行度を表すステージによって治療の流れや目的は異なりますが、遠隔転移がなければ治癒を目指した治療の対象となります。\n  \nがんと診断された人のなかでも浸潤がんと非浸潤がんに分けられる。非浸潤がんなら取り除ける可能性が高い。","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1, ncols=3, figsize=(20, 5))\n\nsns.countplot(data=train, x='invasive', ax=axs[0])\nfor q in axs[0].patches:\n    axs[0].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[0].set_title('invasive')\n\nsns.countplot(data=train, x='invasive', ax=axs[1], hue='cancer')\nfor q in axs[1].patches:\n    axs[1].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[1].set_title('invasive (hue=cancer)')\n\nsns.countplot(data=train, x='cancer', ax=axs[2], hue='invasive')\nfor q in axs[2].patches:\n    axs[2].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[1].set_title('cancer (hue=invasive)')\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:48.424528Z","iopub.execute_input":"2023-01-28T00:51:48.425252Z","iopub.status.idle":"2023-01-28T00:51:48.900342Z","shell.execute_reply.started":"2023-01-28T00:51:48.425212Z","shell.execute_reply":"2023-01-28T00:51:48.89931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 当たり前かもしれないが、がんが発見されなかった人はみんな浸潤がんとは診断されていない（そもそもがんではない）  \n今のところ使う場面はない（？）  ","metadata":{}},{"cell_type":"markdown","source":"# 2.9 BIRADS\n`BIRADS` - 乳房がフォローアップを必要とした場合は 0、乳房が癌に対して陰性と評価された場合は 1、乳房が正常と評価された場合は 2。  \n`BIRADS`カラムはBI-RADSスコアを表しているらしい。  \n0 - Need additional imaging evaluation  \n1 - Negative  \n2 - Benign  \n3 - Probably Benign  \n4 - Suspicious  \n5 - Highly Suggestive of Malignancy  \n6 - Known Biopsy-Proven Malignancy  \n参考ノートブック：https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/369262","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1, ncols=2, figsize=(20,5))\nsns.countplot(data=train, x='BIRADS', ax=axs[0])\nfor q in axs[0].patches:\n    axs[0].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[0].set_title('BIRADS')\n\nsns.countplot(data=train, x='BIRADS', hue='cancer', ax=axs[1])\nfor q in axs[1].patches:\n    axs[1].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[1].set_title('BIRADS (hue=cancer)')\n    \nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:48.902361Z","iopub.execute_input":"2023-01-28T00:51:48.903095Z","iopub.status.idle":"2023-01-28T00:51:49.54634Z","shell.execute_reply.started":"2023-01-28T00:51:48.903054Z","shell.execute_reply":"2023-01-28T00:51:49.545358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* とりあえず０は異常（追加検査が必要）、１と２は正常らしい\n* ただ、０だからと言ってがんであるとは言えない\n* ちなみにcancer=1の値が少ないのはBIRADSの欠損値があるから\n\nBIRDASの予測モデルを作成して、それをtestの特徴量として使うのもあり（？）","metadata":{}},{"cell_type":"markdown","source":"# 2.10 implant\n`implant` - 患者が豊胸手術を受けたかどうか。サイト 1 は、乳房レベルではなく、患者レベルで乳房インプラント情報のみを提供します。  \n「サイト１は、乳房レベルではなく、患者レベルで乳房インプラント情報のみを提供します。」？？？  \n\n","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1, ncols=2, figsize=(20,5))\n\nsns.countplot(data=train, x='implant', ax=axs[0])\nfor q in axs[0].patches:\n    axs[0].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[0].set_title('implant')\n\nsns.countplot(data=train, x='implant', hue='cancer', ax=axs[1])\nfor q in axs[1].patches:\n    axs[1].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[1].set_title('implant (hue=cancer)')\nplt.show()\n\nfig, axs = plt.subplots(nrows=1, ncols=3, figsize=(20,5))\nfor i in range(3):\n    n = random.randint(0,1477)\n    tmp = train[train.implant==1].iloc[n]\n    p_id = tmp.patient_id\n    i_id = tmp.image_id\n    l_or_r = tmp.laterality\n    cancer = tmp.cancer\n    img = get_dcm_img(p_id, i_id)\n    axs[i].imshow(img)\n    axs[i].set_title(f'laterality=[{l_or_r}] : cancer=[{cancer}]')\n    axs[i].axis('off')\nfig.suptitle('implant==1 images', fontsize=10)\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:49.547811Z","iopub.execute_input":"2023-01-28T00:51:49.548815Z","iopub.status.idle":"2023-01-28T00:51:56.862452Z","shell.execute_reply.started":"2023-01-28T00:51:49.548776Z","shell.execute_reply":"2023-01-28T00:51:56.861433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = pd.crosstab(train['implant'], train['cancer'],margins=True)\ntmp['0/All'] = tmp[0] / tmp['All']\ntmp['1/All'] = tmp[1] / tmp['All']\ntmp = tmp.drop(index='All')\ndisplay(tmp)\nprint('implant==0の人でがんの人の割合：{}%'.format(round(tmp.loc[0,'1/All']*100,2)))\nprint('implant==1の人でがんの人の割合：{}%'.format(round(tmp.loc[1,'1/All']*100,2)))","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:56.863962Z","iopub.execute_input":"2023-01-28T00:51:56.86457Z","iopub.status.idle":"2023-01-28T00:51:56.924859Z","shell.execute_reply.started":"2023-01-28T00:51:56.864533Z","shell.execute_reply":"2023-01-28T00:51:56.92366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 豊胸手術を受けた人は、受けてない人よりがんの割合が低い\n* 決定木で分類する際はimplantありとなしの場合で結果を比較したほうがいい\n* implant==1の人は少ないため、モデルがどのように評価するかわからない","metadata":{}},{"cell_type":"markdown","source":"# 2.11 density\n`density` - 乳房組織の密度の評価。A が最も密度が低く、D が最も密度が高い。非常に密度の高い組織は、診断をより困難にする可能性があります。trainのみ  \n\n","metadata":{}},{"cell_type":"code","source":"fig, axs =plt.subplots(nrows=1, ncols=2, figsize=(20, 5))\ntrain_ = train.copy()\ntrain_['density'] = train['density'].fillna('NAN')\nsns.countplot(data=train_, x='density', ax=axs[0], order=['A','B','C','D', 'NAN'])\nfor q in axs[0].patches:\n    axs[0].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[0].set_title('density countplot')\n\nsns.countplot(data=train_, x='density', ax=axs[1],hue='cancer', order=['A','B','C','D', 'NAN'])\nfor q in axs[1].patches:\n    axs[1].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[1].set_title('density countplot(hue=cancer)')\n\n\ntmp = pd.crosstab(train_['density'], train_['cancer'])\n\nfig, axs = plt.subplots(nrows=2, ncols=3, figsize=(20,8), tight_layout=True)\naxs = axs.flatten()\nfor n,i in enumerate(['A', 'B', 'C', 'D', 'NAN']):\n    x = tmp.loc[i].tolist()\n    axs[n].pie(x, labels=['No','cancer'], autopct='%1.1f%%', explode=[0, 0.2], textprops={'weight': 'bold', 'size': 20})\n    axs[n].set_title(i, fontsize=20)\naxs[5].axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-01-28T00:51:56.926846Z","iopub.execute_input":"2023-01-28T00:51:56.927251Z","iopub.status.idle":"2023-01-28T00:51:58.295935Z","shell.execute_reply.started":"2023-01-28T00:51:56.927211Z","shell.execute_reply":"2023-01-28T00:51:58.29447Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"画像を見てみる","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1, ncols=4, figsize=(20,5))\naxs = axs.flatten()\nfor n, i in enumerate(['A', 'B', 'C', 'D']):\n    tmp = train[(train.density==i)&(train.implant==0)]\n    tmp_ = tmp.iloc[random.randint(0, tmp.shape[0])]\n    p_id = tmp_.patient_id\n    i_id = tmp_.image_id\n    l_or_r = tmp_.laterality\n    cancer = tmp_.cancer\n    img = get_dcm_img(p_id, i_id)\n    axs[n].imshow(img)\n    axs[n].set_title(f'density=[{i}] : laterality=[{l_or_r}] : cancer=[{cancer}]')\n    axs[n].axis('off')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:51:58.298395Z","iopub.execute_input":"2023-01-28T00:51:58.299506Z","iopub.status.idle":"2023-01-28T00:52:15.901615Z","shell.execute_reply.started":"2023-01-28T00:51:58.299418Z","shell.execute_reply":"2023-01-28T00:52:15.900698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* densityは欠損値が多く、一応dicomファイルにあるCompressionForce and BodyPartThicknessから補完できるみたいだけど、trainで欠損しているところはそのdicomファイルのデータも欠損しているらしい、\n* 用途としては、BIRADSスコアを予測するときなどに使えるかもしれない","metadata":{}},{"cell_type":"markdown","source":"# 2.12 machine_id\n`machine_id` - イメージング デバイスの ID コード  \n","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=2, ncols=1, figsize=(15, 10), tight_layout=True)\naxs = axs.flatten()\nsns.countplot(data=train, x='machine_id', ax=axs[0])\nfor q in axs[0].patches:\n    axs[0].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[0].set_title('machine_id countplot')\n\nsns.countplot(data=train, x='machine_id', ax=axs[1], hue='cancer')\nfor q in axs[1].patches:\n    axs[1].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[0].set_title('machine_id countplot (hue=cancer)')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-01-28T00:52:15.906959Z","iopub.execute_input":"2023-01-28T00:52:15.907608Z","iopub.status.idle":"2023-01-28T00:52:17.228622Z","shell.execute_reply.started":"2023-01-28T00:52:15.90757Z","shell.execute_reply":"2023-01-28T00:52:17.227371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"特に言うことはない（？）","metadata":{}},{"cell_type":"markdown","source":"# 2.13 difficult_negative_case\n`difficult_negative_case` - ケースが非常に困難な場合は true。trainのみ","metadata":{}},{"cell_type":"code","source":"fig, axs =plt.subplots(nrows=1, ncols=2, figsize=(20, 5))\nsns.countplot(data=train, x='difficult_negative_case', ax=axs[0])\nfor q in axs[0].patches:\n    axs[0].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[0].set_title('difficult_negative_case countplot')\n\nsns.countplot(data=train, x='difficult_negative_case', ax=axs[1],hue='cancer')\nfor q in axs[1].patches:\n    axs[1].annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)\naxs[1].set_title('difficult_negative_case countplot(hue=cancer)')","metadata":{"execution":{"iopub.status.busy":"2023-01-28T00:52:17.230243Z","iopub.execute_input":"2023-01-28T00:52:17.230868Z","iopub.status.idle":"2023-01-28T00:52:17.642371Z","shell.execute_reply.started":"2023-01-28T00:52:17.23082Z","shell.execute_reply":"2023-01-28T00:52:17.641365Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Trueの場合、がんの人はいない\n* なにか使えそうだけど、とりあえず放置","metadata":{}},{"cell_type":"markdown","source":"# 3. Tips","metadata":{}},{"cell_type":"markdown","source":"### 3.1 About Mammography\n参考：https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/369262  \nhttps://www.ncc.go.jp/jp/ncch/division/radiological_technology/radiological_diagnosis/xsenkensa/020/020.html  \n  \n#### mammographyとは\n* マンモグラフィとは女性の乳がん検診に最適な画像診断法で、乳房専用のX線検査のこと。\n* 乳房を板で圧迫し、薄く伸ばした状態で撮影する\n* 乳房全体を写し出すために、複数の方向（view）から圧迫し撮影する\n* 乳房を薄く伸ばすことで乳腺が広がり、腫瘤性の病変がより鮮明に観察可能となる。またマンモグラフィでは、乳房を触ってもしこりがわからないようなタイプの乳がんも、白い点のように見える微細石灰化病変として見つけることができる。\n* マンモグラフィは、特にこの石灰化を見つけることに有用な検査。","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. category data\n* クラメールの連関係数でカテゴリーデータの相関を調べる\n    * 0.1以上なら関係があると言えるらしい","metadata":{}},{"cell_type":"code","source":"# クラメールの連関係数\ndef cramersV(x, y):\n    \"\"\"\n    Calc Cramer's V.\n\n    Parameters\n    ----------\n    x : {numpy.ndarray, pandas.Series}\n    y : {numpy.ndarray, pandas.Series}\n    \"\"\"\n    table = np.array(pd.crosstab(x, y)).astype(np.float32)\n    n = table.sum()\n    colsum = table.sum(axis=0)\n    rowsum = table.sum(axis=1)\n    expect = np.outer(rowsum, colsum) / n\n    chisq = np.sum((table - expect) ** 2 / expect)\n    return np.sqrt(chisq / (n * (np.min(table.shape) - 1)))\n\ntarget_obj = ['laterality', 'view', 'BIRADS', 'density', 'difficult_negative_case', 'cancer']\ndf = pd.DataFrame(columns=target_obj, index=target_obj).astype('float')\nfor i in target_obj:\n    for j in target_obj:\n        df[i][j] = cramersV(train[i], train[j])\ndisplay(df)\nplt.figure(figsize=(10,7))\nsns.heatmap(df, cmap='Greens', annot=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T01:21:00.467648Z","iopub.execute_input":"2023-01-28T01:21:00.46801Z","iopub.status.idle":"2023-01-28T01:21:01.565229Z","shell.execute_reply.started":"2023-01-28T01:21:00.467979Z","shell.execute_reply":"2023-01-28T01:21:01.564308Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* `BIRADS`と`difficult_negative_case`が非常に強い相関であることがわかる  \n* `BIRADS`と`cacner`も相関がみられた","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(figsize=(10,5))\nsns.countplot(data=train, x='BIRADS', hue='difficult_negative_case', ax=axs)\nfor q in axs.patches:\n    axs.annotate(format(q.get_height()), \n                   (q.get_x() + q.get_width() / 2., \n                    q.get_height()), \n                    ha = 'center', \n                    va = 'center', \n                    xytext = (0, 9), \n                    textcoords = 'offset points',\n                    fontsize = 14)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T01:29:06.500713Z","iopub.execute_input":"2023-01-28T01:29:06.501123Z","iopub.status.idle":"2023-01-28T01:29:06.889434Z","shell.execute_reply.started":"2023-01-28T01:29:06.501089Z","shell.execute_reply":"2023-01-28T01:29:06.888255Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* difficult_negative_caseがTrueの行はすべてBIRADSが0である","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"これから\n* viewについて深堀\n* 画像処理を考える、実際にプログラムも書く\n* 相関に関することやってみる　https://istat.co.jp/sk_commentary/correlation-test/type\n* P値とか見てみる","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}