{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 221201_Explore different models\nThis kernel is a try to I understand this [other one]((https://www.kaggle.com/code/markwijkhuizen/g2net-efficientnetv2-s-generated-data-tf-training/notebook)) by [@markwijkhuizen](https://www.kaggle.com/markwijkhuizen)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport glob\nimport os\nimport sys\n\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-12-11T16:45:42.323944Z","iopub.execute_input":"2022-12-11T16:45:42.324355Z","iopub.status.idle":"2022-12-11T16:45:42.330495Z","shell.execute_reply.started":"2022-12-11T16:45:42.324328Z","shell.execute_reply":"2022-12-11T16:45:42.329433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\nVERBOSE = 1 if IS_INTERACTIVE else 2","metadata":{"execution":{"iopub.status.busy":"2022-12-11T16:45:42.333449Z","iopub.execute_input":"2022-12-11T16:45:42.333901Z","iopub.status.idle":"2022-12-11T16:45:42.344596Z","shell.execute_reply.started":"2022-12-11T16:45:42.333873Z","shell.execute_reply":"2022-12-11T16:45:42.343479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare data to train model","metadata":{}},{"cell_type":"code","source":"# Define some variables. TODO b1-p1\nTRAIN_DIR = '/kaggle/input/g2net-detecting-continuous-gravitational-waves/train'\nTRAIN_SAMPLES_DIR = '/kaggle/input/generating-continuous-gravitationalwave-sign-pub'\nTRAIN_SAMPLES_DIR_NOISE = '/kaggle/input/generating-continuous-gravitationalwave-noise-pub'\n\n# Make 360x256 Patches\nTARGET_HEIGHT = 360\nTARGET_WIDTH = 256\n\nINPUTS = ['H', 'L']\nN_INPUTS = len(INPUTS)","metadata":{"execution":{"iopub.status.busy":"2022-12-11T16:45:42.346456Z","iopub.execute_input":"2022-12-11T16:45:42.347574Z","iopub.status.idle":"2022-12-11T16:45:42.36623Z","shell.execute_reply.started":"2022-12-11T16:45:42.347512Z","shell.execute_reply":"2022-12-11T16:45:42.363856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Determine number of samples using glob\nSAMPLE_IDXS = np.arange(len(glob.glob(f'{TRAIN_SAMPLES_DIR}/train_samples/x/*')))\nSAMPLE_IDXS_NOISE = np.arange(len(glob.glob(f'{TRAIN_SAMPLES_DIR_NOISE}/train_samples/x/*')))\nSAMPLE_IDXS_VAL = np.arange(len(glob.glob(f'{TRAIN_SAMPLES_DIR}/val_samples/x/*')))\n\n# Number of samples, each sample consist of a Hanford and Livington patch\nN_SAMPLE_IDXS = len(SAMPLE_IDXS)\nN_SAMPLE_IDXS_NOISE = len(SAMPLE_IDXS_NOISE)\nN_SAMPLE_IDXS_VAL = len(SAMPLE_IDXS_VAL)\n\n# Validation targets\nTARGETS_VAL = np.load(f'{TRAIN_SAMPLES_DIR}/TARGETS_VAL.npy')\n\n# Signal to noise ratios for generated signal samples\nSNRS = np.load(f'{TRAIN_SAMPLES_DIR}/SNRS.npy')","metadata":{"execution":{"iopub.status.busy":"2022-12-11T16:45:42.368038Z","iopub.execute_input":"2022-12-11T16:45:42.369326Z","iopub.status.idle":"2022-12-11T16:45:42.46855Z","shell.execute_reply.started":"2022-12-11T16:45:42.369252Z","shell.execute_reply":"2022-12-11T16:45:42.467497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 10000 Train Signal, 5000 Train Noise and 600 Val\n# Sample Index to Number of Patches\nSAMPLE_IDX2N_PATCHES = dict([\n    (i, len(glob.glob(f'{TRAIN_SAMPLES_DIR}/train_samples/x/{i}/*.png'))) for i in tqdm(SAMPLE_IDXS)\n])\n\n# Sample Index to Number of Patches Noise\nSAMPLE_IDX2N_PATCHES_NOISE = dict([\n    (i, len(glob.glob(f'{TRAIN_SAMPLES_DIR_NOISE}/train_samples/x/{i}/*.png'))) for i in tqdm(SAMPLE_IDXS_NOISE)\n])\n\n# Validation Sample Index to Number of Patches\nSAMPLE_IDX2N_PATCHES_VAL = dict([\n    (i, len(glob.glob(f'{TRAIN_SAMPLES_DIR}/val_samples/x/{i}/*.png'))) for i in tqdm(SAMPLE_IDXS_VAL)\n])","metadata":{"execution":{"iopub.status.busy":"2022-12-11T16:45:42.470574Z","iopub.execute_input":"2022-12-11T16:45:42.471107Z","iopub.status.idle":"2022-12-11T16:45:56.509678Z","shell.execute_reply.started":"2022-12-11T16:45:42.471079Z","shell.execute_reply":"2022-12-11T16:45:56.509059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute number of samples per type\nN_TRAIN_SAMPLES = 0\nN_TRAIN_SAMPLES_SIGNAL = 0\nN_VAL_SAMPLES = 0\n\nfor idx in SAMPLE_IDXS:\n    N_TRAIN_SAMPLES += SAMPLE_IDX2N_PATCHES[idx] #10000\n    N_TRAIN_SAMPLES_SIGNAL += SAMPLE_IDX2N_PATCHES[idx] #10000\n    \nfor idx in SAMPLE_IDXS_NOISE:\n    N_TRAIN_SAMPLES += SAMPLE_IDX2N_PATCHES_NOISE[idx] #5000\n\nfor idx in SAMPLE_IDXS_VAL:\n    N_VAL_SAMPLES += SAMPLE_IDX2N_PATCHES_VAL[idx] #600\n","metadata":{"execution":{"iopub.status.busy":"2022-12-11T16:45:56.51076Z","iopub.execute_input":"2022-12-11T16:45:56.511632Z","iopub.status.idle":"2022-12-11T16:45:56.522454Z","shell.execute_reply.started":"2022-12-11T16:45:56.511604Z","shell.execute_reply":"2022-12-11T16:45:56.521286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create a copy of the data\ntrain_labels = pd.read_csv('/kaggle/input/g2net-detecting-continuous-gravitational-waves/train_labels.csv')  \nN_SAMPLES = len(train_labels)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-11T16:45:56.523742Z","iopub.execute_input":"2022-12-11T16:45:56.523999Z","iopub.status.idle":"2022-12-11T16:45:56.540027Z","shell.execute_reply.started":"2022-12-11T16:45:56.523974Z","shell.execute_reply":"2022-12-11T16:45:56.539263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating the model using the Sequential API & Compiling the model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport sys\nimport gc\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 16\nBATCH_SIZE_FINE_TUNE = 8\nEPOCHS = 5\nEPOCHS_FINE_TUNE = 1\nNOISE_RATIO = 0.333\n\nLR_MAX = 4e-4","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:34:01.180407Z","iopub.execute_input":"2022-12-11T17:34:01.180759Z","iopub.status.idle":"2022-12-11T17:34:01.187021Z","shell.execute_reply.started":"2022-12-11T17:34:01.180733Z","shell.execute_reply":"2022-12-11T17:34:01.185669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sys.path.append('/kaggle/input/efficientnetv2-pretrained-imagenet21k-weights/brain_automl/')\nsys.path.append('/kaggle/input/efficientnetv2-pretrained-imagenet21k-weights/brain_automl/efficientnetv2/')\nimport effnetv2_model","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:31:05.520858Z","iopub.execute_input":"2022-12-11T17:31:05.521176Z","iopub.status.idle":"2022-12-11T17:31:05.526168Z","shell.execute_reply.started":"2022-12-11T17:31:05.521153Z","shell.execute_reply":"2022-12-11T17:31:05.524553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    # Input spectogram 360x256\n    inputs = tf.keras.layers.Input(shape=[TARGET_HEIGHT, TARGET_WIDTH], dtype=tf.uint8)\n    \n    # Create 3-channel image\n    x = tf.expand_dims(inputs, axis=-1)\n    x = tf.tile(x, [1,1,1,3])\n    x = tf.cast(x, tf.float32)\n    # Imagenet normalization\n    x = tf.keras.applications.imagenet_utils.preprocess_input(x, mode='torch')\n    # Gaussion noise\n    x = tf.keras.layers.GaussianNoise(0.10)(x)\n\n    # EfficientNetV2-S backbone\n    cnn = effnetv2_model.get_model(f'efficientnetv2-s', include_top=False, weights='imagenet21k-ft1k')\n\n    x = cnn(x)\n    # Apply ReLu activation, there should not be a negative (noise) filter\n    # Noise filters do not make sense...\n    x = tf.keras.layers.LeakyReLU()(x)\n    # Dropout for regularization\n    x = tf.keras.layers.Dropout(0.30)(x)\n\n    # Output is a single neuron with sigmoid activation initialized with HE weights\n    output = tf.keras.layers.Dense(1, activation='sigmoid', kernel_initializer='he_uniform')(x)\n    \n    # Model\n    model = tf.keras.models.Model(inputs=inputs, outputs=output)\n    \n    # LOSS\n    loss = tf.keras.losses.BinaryCrossentropy(from_logits=False)\n\n    # OPTIMIZER\n    optimizer = tfa.optimizers.AdamW(learning_rate=LR_MAX, weight_decay=LR_MAX*1e-2, epsilon=1e-7)\n\n    # METRICS\n    metrics = [\n        tf.keras.metrics.BinaryAccuracy(),\n        tf.keras.metrics.Precision(),\n        tf.keras.metrics.Recall(),\n        tf.keras.metrics.AUC(),\n    ]\n\n    # Compile Model\n    model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:31:17.696411Z","iopub.execute_input":"2022-12-11T17:31:17.696817Z","iopub.status.idle":"2022-12-11T17:31:17.709029Z","shell.execute_reply.started":"2022-12-11T17:31:17.69679Z","shell.execute_reply":"2022-12-11T17:31:17.708061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\ngc.collect()\n\nmodel = get_model()","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:34:03.76675Z","iopub.execute_input":"2022-12-11T17:34:03.767313Z","iopub.status.idle":"2022-12-11T17:34:20.669246Z","shell.execute_reply.started":"2022-12-11T17:34:03.767268Z","shell.execute_reply":"2022-12-11T17:34:20.667184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training an evaluating the model","metadata":{}},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:38:33.184507Z","iopub.execute_input":"2022-12-11T17:38:33.184886Z","iopub.status.idle":"2022-12-11T17:38:33.499731Z","shell.execute_reply.started":"2022-12-11T17:38:33.184859Z","shell.execute_reply":"2022-12-11T17:38:33.49759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Samnple Index to Patch File Paths\nSAMPLE_IDX2FILE_PATHS = dict([\n    (i, glob.glob(f'{TRAIN_SAMPLES_DIR}/train_samples/x/{i}/*.png')) for i in tqdm(SAMPLE_IDXS)\n])\n\n# Samnple Index to Patch File Paths\nSAMPLE_IDX_NOISE2FILE_PATHS = dict([\n    (i, glob.glob(f'{TRAIN_SAMPLES_DIR_NOISE}/train_samples/x/{i}/*.png')) for i in tqdm(SAMPLE_IDXS_NOISE)\n])\n\n# Samnple Index to Patch File Paths VAL\nSAMPLE_IDX2FILE_PATHS_VAL = dict([\n    (i, glob.glob(f'{TRAIN_SAMPLES_DIR}/val_samples/x/{i}/*.png')) for i in tqdm(SAMPLE_IDXS_VAL)\n])","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:37:36.865271Z","iopub.execute_input":"2022-12-11T17:37:36.865663Z","iopub.status.idle":"2022-12-11T17:37:48.516563Z","shell.execute_reply.started":"2022-12-11T17:37:36.865635Z","shell.execute_reply":"2022-12-11T17:37:48.515079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_TRAIN_STEPS = N_TRAIN_SAMPLES // BATCH_SIZE + 1\nN_VAL_STEPS = N_VAL_SAMPLES\n","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:36:59.616056Z","iopub.execute_input":"2022-12-11T17:36:59.616452Z","iopub.status.idle":"2022-12-11T17:36:59.623584Z","shell.execute_reply.started":"2022-12-11T17:36:59.616425Z","shell.execute_reply":"2022-12-11T17:36:59.621441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation set will iterate sequentially over all samples\ndef get_val_dataset(val=True, signal_only=False):\n    while True:\n        if val:\n            for index in SAMPLE_IDXS_VAL:\n                # Load target\n                y = np.expand_dims(TARGETS_VAL[index], axis=0)\n\n                for file_path in SAMPLE_IDX2FILE_PATHS_VAL[index]:\n                    X = cv2.imread(file_path, -1)\n                    X = np.expand_dims(X, axis=0)\n\n                    yield X, y\n        else:\n            for index in SAMPLE_IDXS_NOISE:\n                # Load target\n                y = np.array([[0]], dtype=np.int8)\n\n                for file_path in SAMPLE_IDX_NOISE2FILE_PATHS[index]:\n                    X = cv2.imread(file_path, -1)\n                    X = np.expand_dims(X, axis=0)\n\n                    yield X, y\n                    \n            for index in SAMPLE_IDXS:\n                # Load target\n                y = np.array([[1]], dtype=np.int8)\n\n                for file_path in SAMPLE_IDX2FILE_PATHS[index]:\n                    X = cv2.imread(file_path, -1)\n                    X = np.expand_dims(X, axis=0)\n\n                    yield X, y","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:36:32.929252Z","iopub.execute_input":"2022-12-11T17:36:32.929662Z","iopub.status.idle":"2022-12-11T17:36:32.940488Z","shell.execute_reply.started":"2022-12-11T17:36:32.929628Z","shell.execute_reply":"2022-12-11T17:36:32.938722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation Baseline, sanity check, should be random guessing\n_ = model.evaluate(\n        get_val_dataset(val=True),\n        steps=N_VAL_STEPS,\n        verbose=VERBOSE,\n    )","metadata":{"execution":{"iopub.status.busy":"2022-12-11T17:38:36.404757Z","iopub.execute_input":"2022-12-11T17:38:36.405093Z","iopub.status.idle":"2022-12-11T17:46:09.037853Z","shell.execute_reply.started":"2022-12-11T17:38:36.405067Z","shell.execute_reply":"2022-12-11T17:46:09.036417Z"},"trusted":true},"execution_count":null,"outputs":[]}]}