{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### here we will try everything just on test dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.image_id.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### load model"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Import Densenet from Keras\nfrom keras.applications.densenet import DenseNet121\nfrom keras.layers import Dense, GlobalAveragePooling2D\nfrom keras.models import Model\nfrom keras import backend as K","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_densenet_model():\n    # get input layer\n    img_input = tf.keras.layers.Input(shape=(256, 256,1))\n\n    # change shape for compatibility\n    img_conc = tf.keras.layers.Concatenate()([img_input, img_input, img_input])     \n    \n    # load base model - transfer learned\n    base_model = DenseNet121(weights='../input/densenet-weights-nih-coursera-ai4m/densenet.hdf5', include_top=False, input_tensor=img_conc)\n    \n    # see last layer - customise it\n    x = base_model.output\n    \n    # Add a global spatial average pooling layer\n    x_pool = GlobalAveragePooling2D()(x)\n\n    # Add a logistic layer the same size as the number of classes you're trying to predict\n    n_classes = 15\n\n    predictions = Dense(n_classes, activation=\"sigmoid\")(x_pool)\n    print(f\"Predictions have {n_classes} units, one for each class\")\n\n    # Create an updated model\n    # model = Model(inputs=in1, outputs=predictions)\n    model = Model(inputs=img_input, outputs=predictions)\n    \n    return model\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\n# import tensorflow.keras.layers as L\nimport tensorflow.keras.backend as K\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model2 = build_densenet_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! ls ../input/chest-x-ray-abnormalities-densenet-pipeline\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model2.load_weights('../input/chest-x-ray-abnormalities-densenet-pipeline/model2.1_march28.hdf5')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### now ready to predict on test images"},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# filename = '004f33259ee4aef671c2b95d54e4be68'\n# filename = '008bdde2af2462e86fd373a445d0f4cd'\n# filename = '009bc039326338823ca3aa84381f17f1'\n# filename = '013c169f9dad6f1f6485da961b9f7bf2'\nfilename = '01431a2618c0ace741e4e270a37e20b9'\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# ! ls ../input/xraynumpy/images/test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = '../input/xraynumpy/images/test/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image = np.load(path + filename + '.npy')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# model = model2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images = [image]\n\nX = np.stack(images,axis=0)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"op = model.predict(X)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"op = op[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"classes = ['Aortic enlargement',\n 'Atelectasis',\n 'Calcification',\n 'Cardiomegaly',\n 'Consolidation',\n 'ILD',\n 'Infiltration',\n 'Lung Opacity',\n 'Nodule/Mass',\n 'Other lesion',\n 'Pleural effusion',\n 'Pleural thickening',\n 'Pneumothorax',\n 'Pulmonary fibrosis',\n 'No finding']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"op_list = []\n\nfor ix, y_pred in enumerate(list(op)):\n    if y_pred > 0.5:\n        y_pred = round(y_pred,2)\n        print(ix, y_pred, classes[ix])\n        op_tag = str(ix) + ' ' + str(y_pred) + ' 0 0 1 1'\n        print('op_tag=', op_tag)\n        op_list.append(op_tag)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"op_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"op_str = ' '.join(op_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"op_str","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 0 0.576 1150 703 1419 1019 14 0 0 0 1 1\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### now run for each case"},{"metadata":{"trusted":true},"cell_type":"code","source":"def my_predict(row):\n    row_ix = row.name\n    if row_ix % 100 == 0:\n        print('done for', row_ix)\n    \n    filename = row.image_id\n    path = '../input/xraynumpy/images/test/'\n    image = np.load(path + filename + '.npy')\n    \n    images = [image]\n    X = np.stack(images,axis=0)\n    \n    \n    op = model.predict(X)\n    op = op[0]\n\n    op_list = []\n\n    for ix, y_pred in enumerate(list(op)):\n        if y_pred > 0.5:\n            y_pred = round(y_pred,2)\n#             print(ix, y_pred, classes[ix])\n            op_tag = str(ix) + ' ' + str(y_pred) + ' 0 0 1 1'\n#             print('op_tag=', op_tag)\n            op_list.append(op_tag)\n\n    op_str = ' '.join(op_list)\n    return op_str\n    \n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_predict(test.iloc[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['op_str'] = test.apply(lambda row: my_predict(row), axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_cols = ['image_id', 'op_str']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df = test[my_cols]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df.columns = ['image_id', 'PredictionString']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_filepath = str(\"submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df.to_csv(submission_filepath, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### merge v0 and v1"},{"metadata":{"trusted":true},"cell_type":"code","source":"# submission_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! ls ../input/bams-xray-results-v0-and-v1/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_v0 = pd.read_csv('../input/bams-xray-results-v0-and-v1/submission v0.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_v1 = pd.read_csv('../input/bams-xray-results-v0-and-v1/submission v1.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_v0.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_v1.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### clean some v0 predictions and see scores"},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_row(row):\n    original = row.PredictionString\n    \n#     print('original=', original)\n    original_list = original.split(' ')\n#     print('original_list=', original_list)\n    \n    assert len(original_list) % 6 == 0\n    \n    if len(original_list)  == 6:\n        return original\n    \n    shortlist = []\n    \n    for j in range(0, len(original_list), 6):\n        obj = original_list[j: j + 6]\n        conf = float(obj[1])\n        \n        if conf > 0.5:\n#             print('selecting obj=', obj)\n            obj_str = ' '.join(obj)\n            shortlist.append(obj_str)\n#         else:\n#             print('NOT selecting obj=', obj)\n\n#     print('shortlist=', shortlist)\n    shortlist_str = ' '.join(shortlist)\n    \n    return shortlist_str\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"row = submission_v0.iloc[2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"row","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clean_row(row)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_v0['clean_PredictionString'] = submission_v0.apply(lambda row: clean_row(row),\n                                                             axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_v0.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_filepath = str(\"submission_v0_clean.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_v0.to_csv(submission_filepath, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}