{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import glob\n#csv_list = sorted(glob.glob('../input/o2unet/*.csv'))\ncsv_list = sorted(glob.glob('../input/o2unet2/*.csv'))\npd.get_option(\"display.max_columns\")\npd.set_option('display.max_columns', 200)\npd.get_option(\"display.max_rows\")\npd.set_option('display.max_rows', 200)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_normalized_loss(csv_name):\n    df = pd.read_csv(csv_name)\n    df = df.loc[:,['img_idx','loss']].set_index('img_idx')\n    print(df['loss'].mean())\n    #print(os.path.basename(csv_name).split('_')[0])\n    df['loss'] = df['loss'] - df['loss'].mean()\n    return df.sort_index().add_prefix(os.path.basename(csv_name).split('_')[0] + '_')\nget_normalized_loss(csv_list[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = get_normalized_loss(csv_list[0])\nfor csv_name in csv_list[1:]:\n    df = df.join(get_normalized_loss(csv_name)) \ndf[:200]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#df['loss_avg'] = df.var(axis='columns')\ndf['loss_avg'] = df.mean(axis='columns')\ndf = df.sort_values('loss_avg')\ndf[:200]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"k=0.1\nnum_remain = int(len(df)*(1-k))\nprint('remain: ' + str(num_remain))\nprint('')\nprint('cutting line:')\nprint(df.iloc[num_remain])\ndf['loss_avg'].hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#df[df['loss_avg']>1.974052]['loss_avg'].hist()\ndf[df['loss_avg']>1.974052]['loss_avg'].hist()\nprint(len(df[df['loss_avg']>2**2]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/prostate-cancer-grade-assessment/train.csv')\nremain_df = df[:num_remain]\nremain_df.index.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_clean = train_df.loc[remain_df.index.values]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_clean['data_provider'].value_counts().plot(kind=\"bar\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_clean[:200]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_clean[-200:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_clean['isup_grade'].hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"delete_df = df[num_remain:]\ntrain_df_delete = train_df.loc[delete_df.index.values]\ntrain_df_delete['isup_grade'].hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_delete[-200:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_delete['data_provider'].value_counts().plot(kind=\"bar\")\n#train_df_clean['data_provider'][:1000].value_counts().plot(kind=\"bar\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_delete.to_csv('o2u-22017707-k01.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.loc[5880]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install imagecodecs\n!pip install tifffile","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!dpkg -l | grep pixman","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import skimage.io\nimport matplotlib.pyplot as plt\n\nidx = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(6,6))\nimg_id = train_df_delete.iloc[-idx]['image_id']\nimg_path = '../input/prostate-cancer-grade-assessment/train_images/' + img_id + '.tiff'\n\nimage = skimage.io.MultiImage(img_path)[2]\nimage = np.array(image)\nplt.imshow(image)\nidx += 1","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}