{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install mediapipe --q\n!pip install -q git+https://github.com/hoyso48/tf-utils@main","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:43:09.969642Z","iopub.execute_input":"2023-08-18T23:43:09.97052Z","iopub.status.idle":"2023-08-18T23:43:17.658591Z","shell.execute_reply.started":"2023-08-18T23:43:09.970478Z","shell.execute_reply":"2023-08-18T23:43:17.657316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport json\nimport os\nimport sys\nfrom glob import glob\nfrom pathlib import Path\nfrom IPython.display import Video,display\nfrom tqdm.auto import tqdm\nfrom PIL import Image\ntqdm.pandas()\nfrom multiprocessing import Pool\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport tensorflow.keras.mixed_precision as mixed_precision\nimport sklearn\nfrom tf_utils.schedules import OneCycleLR, ListedLR\nfrom tf_utils.callbacks import Snapshot, SWA\nfrom tf_utils.learners import FGM, AWP","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-18T23:43:17.660729Z","iopub.execute_input":"2023-08-18T23:43:17.661143Z","iopub.status.idle":"2023-08-18T23:44:01.403278Z","shell.execute_reply.started":"2023-08-18T23:43:17.661107Z","shell.execute_reply":"2023-08-18T23:44:01.40231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rootin = '/kaggle/input/wlasl-processed'\nrootout= '/kaggle/working/'\nvideo_path = os.path.join(rootin,'videos')\nnslt_100_json =  os.path.join(rootin,'nslt_100.json')","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.404463Z","iopub.execute_input":"2023-08-18T23:44:01.404968Z","iopub.status.idle":"2023-08-18T23:44:01.409788Z","shell.execute_reply.started":"2023-08-18T23:44:01.404939Z","shell.execute_reply":"2023-08-18T23:44:01.409008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import cv2\n# import pandas as pd\n# import mediapipe as mp\n# import pyarrow as pa\n# import pyarrow.parquet as pq\n# import numpy as np\n\n# mp_holistic = mp.solutions.holistic\n# mp_drawing = mp.solutions.drawing_utils\n\n# def process_video(video_path):\n#     cap = cv2.VideoCapture(video_path)\n#     with mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.5) as holistic:\n#         features = []\n#         frame_number = 0\n#         while cap.isOpened():\n#             ret, frame = cap.read()\n#             if not ret:\n#                 break\n\n#             rgb_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n#             results = holistic.process(rgb_frame)\n\n#             pose_landmarks = results.pose_landmarks.landmark if results.pose_landmarks else []\n#             left_hand_landmarks = results.left_hand_landmarks.landmark if results.left_hand_landmarks else []\n#             right_hand_landmarks = results.right_hand_landmarks.landmark if results.right_hand_landmarks else []\n#             face_landmarks = results.face_landmarks.landmark if results.face_landmarks else []\n\n#             for landmark_index in range(33):  # Assuming 33 pose landmarks\n#                 landmark = pose_landmarks[landmark_index] if landmark_index < len(pose_landmarks) else None\n#                 features.append({\n#                     'frame': frame_number,\n#                     'row_id': f'{frame_number}-pose-{landmark_index}',\n#                     'type': 'pose',\n#                     'landmark_index': landmark_index,\n#                     'x': landmark.x if landmark and landmark.visibility > 0 else np.nan,\n#                     'y': landmark.y if landmark and landmark.visibility > 0 else np.nan,\n#                     'z': landmark.z if landmark and landmark.visibility > 0 else np.nan\n#                 })\n\n#             for landmark_index in range(21):  # Assuming 21 hand landmarks\n#                 left_hand_landmark = left_hand_landmarks[landmark_index] if landmark_index < len(left_hand_landmarks) else None\n#                 right_hand_landmark = right_hand_landmarks[landmark_index] if landmark_index < len(right_hand_landmarks) else None\n#                 features.append({\n#                     'frame': frame_number,\n#                     'row_id': f'{frame_number}-left_hand-{landmark_index}',\n#                     'type': 'left_hand',\n#                     'landmark_index': landmark_index,\n#                     'x': left_hand_landmark.x if left_hand_landmark else np.nan,\n#                     'y': left_hand_landmark.y if left_hand_landmark else np.nan,\n#                     'z': left_hand_landmark.z if left_hand_landmark else np.nan\n#                 })\n\n#                 features.append({\n#                     'frame': frame_number,\n#                     'row_id': f'{frame_number}-right_hand-{landmark_index}',\n#                     'type': 'right_hand',\n#                     'landmark_index': landmark_index,\n#                     'x': right_hand_landmark.x if right_hand_landmark else np.nan,\n#                     'y': right_hand_landmark.y if right_hand_landmark else np.nan,\n#                     'z': right_hand_landmark.z if right_hand_landmark else np.nan\n#                 })\n\n#             for landmark_index in range(468):  # Assuming 468 face landmarks\n#                 landmark = face_landmarks[landmark_index] if landmark_index < len(face_landmarks) else None\n#                 features.append({\n#                     'frame': frame_number,\n#                     'row_id': f'{frame_number}-face-{landmark_index}',\n#                     'type': 'face',\n#                     'landmark_index': landmark_index,\n#                     'x': landmark.x if landmark else np.nan,\n#                     'y': landmark.y if landmark else np.nan,\n#                     'z': landmark.z if landmark else np.nan\n#                 })\n\n#             frame_number += 1\n\n#         cap.release()\n#         return features\n\n# def save_features_to_parquet(features, output_file):\n#     df = pd.DataFrame(features)\n#     table = pa.Table.from_pandas(df)\n#     pq.write_table(table, output_file)\n#     del features\n# def run(video_file):\n#     check_file = os.path.join('/kaggle/input/wlasl-landmarks/wlasl', f\"{os.path.splitext(os.path.basename(video_file))[0]}.parquet\")\n\n#     if video_file.endswith(\".mp4\") and not os.path.exists(check_file):\n\n#         video_file = os.path.join(input_directory, video_file)\n#         output_file = os.path.join(output_directory, f\"{os.path.splitext(os.path.basename(video_file))[0]}.parquet\")\n\n#         features = process_video(video_file)\n#         save_features_to_parquet(features, output_file)\n# input_directory = video_path\n# output_directory = rootout+\"/landmarks_files\"\n# os.makedirs(output_directory,exist_ok = True)\n# # for video_file in tqdm(os.listdir(input_directory)):\n# # pool = Pool()\n# # pool.map(run, tqdm())\n# # pool.close()\n# #     break\n# from multiprocessing import Pool\n# if __name__ == '__main__':\n#     with Pool(processes=8) as p:\n#         with tqdm(total=len(os.listdir(input_directory))) as pbar:\n#             for _ in p.imap_unordered(run, os.listdir(input_directory)):\n#                 pbar.update()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.411966Z","iopub.execute_input":"2023-08-18T23:44:01.41226Z","iopub.status.idle":"2023-08-18T23:44:01.425229Z","shell.execute_reply.started":"2023-08-18T23:44:01.412233Z","shell.execute_reply":"2023-08-18T23:44:01.424456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cp -r /kaggle/input/wlasl-landmarks/wlasl/* {output_directory}","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.42627Z","iopub.execute_input":"2023-08-18T23:44:01.426546Z","iopub.status.idle":"2023-08-18T23:44:01.438322Z","shell.execute_reply.started":"2023-08-18T23:44:01.426521Z","shell.execute_reply":"2023-08-18T23:44:01.437478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.concat([pd.read_json(\"/kaggle/input/wlasl-processed/WLASL_v0.3.json\")['gloss'],pd.DataFrame(pd.read_json(\"/kaggle/input/wlasl-processed/WLASL_v0.3.json\")['instances'].to_dict()[0])],axis = 1).dropna()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.439453Z","iopub.execute_input":"2023-08-18T23:44:01.439726Z","iopub.status.idle":"2023-08-18T23:44:01.451466Z","shell.execute_reply.started":"2023-08-18T23:44:01.439701Z","shell.execute_reply":"2023-08-18T23:44:01.450613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_videos_ids(json_list):\n#     \"\"\"\n#     function to check if the video id is available in the dataset\n#     and return the viedos ids of the current instance\n    \n#     input: instance json list\n#     output: list of videos_ids\n    \n#     \"\"\"\n#     videos_list = []    \n#     for ins in json_list:\n#         video_id = ins['video_id']\n#         if os.path.exists(f'{rootin}/videos/{video_id}.mp4'):\n#             videos_list.append(video_id)\n#     return videos_list\n# def get_json_features(json_list):\n#     \"\"\"\n#     function to check if the video id is available in the dataset\n#     and return the viedos ids and url or any other featrue of the current instance\n    \n#     input: instance json list\n#     output: list of videos_ids\n    \n#     \"\"\"\n#     videos_ids = []\n#     videos_urls = []\n#     videos_bbox = []\n#     videos_fps = []\n#     videos_frame_end = []\n#     videos_frame_start = []\n#     videos_signer_id = []\n#     videos_source = []\n#     videos_split = []\n#     videos_variation_id = []\n#     for ins in json_list:\n      \n#         video_id = ins['video_id']\n#         video_url = ins['url']\n#         video_bbox = ins['bbox']\n#         video_fps = ins['fps']\n#         video_frame_end = ins['frame_end']\n#         video_frame_start = ins['frame_start']\n#         video_signer_id = ins['signer_id']\n#         video_source = ins['source']\n#         video_split = ins['split']\n#         video_variation_id = ins['variation_id']\n#         if os.path.exists(f'{rootin}/videos/{video_id}.mp4'):\n#             videos_ids.append(video_id)\n#             videos_urls.append(video_url)\n#             videos_bbox.append(video_bbox)\n#             videos_fps.append(video_fps)\n#             videos_frame_end.append(video_frame_end)\n#             videos_frame_start.append(video_frame_start)\n#             videos_signer_id.append(video_signer_id)\n#             videos_source.append(video_source)\n#             videos_split.append(video_split)\n#             videos_variation_id.append(video_variation_id)\n#     return videos_ids, videos_urls, videos_bbox, videos_fps, videos_frame_end, videos_frame_start, videos_signer_id, videos_source, videos_split,videos_variation_id\n# with open('/kaggle/input/wlasl-processed/WLASL_v0.3.json', 'r') as data_file:\n#     json_data = data_file.read()\n# wlasl_df = pd.read_json(rootin + '/WLASL_v0.3.json')\n# instance_json = json.loads(json_data)\n# get_videos_ids(instance_json[0]['instances'])[0]\n# wlasl_df['videos_ids'] = wlasl_df['instances'].progress_apply(get_videos_ids)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.452703Z","iopub.execute_input":"2023-08-18T23:44:01.453089Z","iopub.status.idle":"2023-08-18T23:44:01.463589Z","shell.execute_reply.started":"2023-08-18T23:44:01.45306Z","shell.execute_reply":"2023-08-18T23:44:01.462779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# features_df = pd.DataFrame(columns=['gloss', 'video_id', 'urls', 'bbox', 'fps', 'frame_end', 'frame_start','signer_id', 'source', 'split', 'variation_id'])\n# for row in tqdm(wlasl_df.iterrows(),total = wlasl_df.shape[0]):\n#     ids, urls, bbox, fps, frame_end, frame_start,signer_id, source, split, variation_id = get_json_features(row[1][1])\n#     word = [row[1][0]] * len(ids)\n#     df = pd.DataFrame(list(zip(word, ids, urls, bbox, fps, frame_end, frame_start, signer_id, source, split, variation_id)), columns=features_df.columns)\n#     features_df = features_df.append(df, ignore_index=True)\n# features_df['video_id'] = features_df['video_id'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.464651Z","iopub.execute_input":"2023-08-18T23:44:01.465014Z","iopub.status.idle":"2023-08-18T23:44:01.47562Z","shell.execute_reply.started":"2023-08-18T23:44:01.464987Z","shell.execute_reply":"2023-08-18T23:44:01.474852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# features_df.to_csv(\"main.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.476698Z","iopub.execute_input":"2023-08-18T23:44:01.477308Z","iopub.status.idle":"2023-08-18T23:44:01.484414Z","shell.execute_reply.started":"2023-08-18T23:44:01.477277Z","shell.execute_reply":"2023-08-18T23:44:01.483587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nROWS_PER_FRAME = 543\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path).sort_values(['frame','type','landmark_index'])\n    data = data[data_columns]\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\ndef encode_row(row):\n    coordinates = load_relevant_data_subset(row.path)\n    coordinates_encoded = coordinates.tobytes()\n#     participant_id = int(row.participant_id)\n#     sequence_id = int(row.sequence_id)\n    sign = int(row.label)\n\n    record_bytes = tf.train.Example(features=tf.train.Features(feature={\n                'coordinates': tf.train.Feature(bytes_list=tf.train.BytesList(value=[coordinates_encoded])),\n#                 'participant_id': tf.train.Feature(int64_list=tf.train.Int64List(value=[participant_id])),\n#                 'sequence_id':tf.train.Feature(int64_list=tf.train.Int64List(value=[sequence_id])),\n                'sign':tf.train.Feature(int64_list=tf.train.Int64List(value=[sign])),\n                })).SerializeToString()\n    return record_bytes\n\ndef process_chunk(row):\n    \n    tfrecord_name = f\"{output_tf_records}/{str(row['video_id']).zfill(5)}.tfrecord\"\n    if not os.path.exists(tfrecord_name):\n        options = tf.io.TFRecordOptions(compression_type='GZIP', compression_level=9)\n        with tf.io.TFRecordWriter(tfrecord_name, options=options) as file_writer:\n            record_bytes = encode_row(row)\n            file_writer.write(record_bytes)\n            del record_bytes\n            file_writer.close()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.488335Z","iopub.execute_input":"2023-08-18T23:44:01.488642Z","iopub.status.idle":"2023-08-18T23:44:01.50169Z","shell.execute_reply.started":"2023-08-18T23:44:01.488615Z","shell.execute_reply":"2023-08-18T23:44:01.500883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM2CHAR = dict(pd.read_csv(\"/kaggle/input/wlasl-processed/wlasl_class_list.txt\",sep = '\\t',header = None).replace('losear','lose').values.tolist())\nCHAR2NUM = {v:k for k,v in NUM2CHAR.items()}","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.502682Z","iopub.execute_input":"2023-08-18T23:44:01.502977Z","iopub.status.idle":"2023-08-18T23:44:01.531446Z","shell.execute_reply.started":"2023-08-18T23:44:01.502952Z","shell.execute_reply":"2023-08-18T23:44:01.530666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/wlasl-landmarks/main.csv\").drop(\"Unnamed: 0\",axis = 1)\ndf['path'] = df['video_id'].progress_apply(lambda x: os.path.join(\"/kaggle/input/wlasl-landmarks/landmarks_files\",str(x).zfill(5)+\".parquet\"))\ndf['label'] = df['gloss'].progress_apply(lambda x: CHAR2NUM.get(x))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.532478Z","iopub.execute_input":"2023-08-18T23:44:01.532755Z","iopub.status.idle":"2023-08-18T23:44:01.739882Z","shell.execute_reply.started":"2023-08-18T23:44:01.532729Z","shell.execute_reply":"2023-08-18T23:44:01.738986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# output_tf_records = '/kaggle/working/landmarks_tf_records'\n# os.makedirs(output_tf_records,exist_ok = True)\n# for idx,row in tqdm(df.iterrows(),total = df.shape[0]):\n#     process_chunk(row)\n# #     break","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.741016Z","iopub.execute_input":"2023-08-18T23:44:01.741334Z","iopub.status.idle":"2023-08-18T23:44:01.745249Z","shell.execute_reply.started":"2023-08-18T23:44:01.741306Z","shell.execute_reply":"2023-08-18T23:44:01.744358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ['KAGGLE_USERNAME'] = 'glitchr'\nos.environ['KAGGLE_KEY']  = '1f780208f5233dc5ac7fc4e130217992'","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.746451Z","iopub.execute_input":"2023-08-18T23:44:01.74677Z","iopub.status.idle":"2023-08-18T23:44:01.75499Z","shell.execute_reply.started":"2023-08-18T23:44:01.746741Z","shell.execute_reply":"2023-08-18T23:44:01.75421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import shutil\n# shutil.make_archive('landmarks_files', format='zip',\n#                     root_dir='/kaggle/input/wlasl-landmarks', base_dir='landmarks_files')","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.756003Z","iopub.execute_input":"2023-08-18T23:44:01.756349Z","iopub.status.idle":"2023-08-18T23:44:01.76728Z","shell.execute_reply.started":"2023-08-18T23:44:01.756322Z","shell.execute_reply":"2023-08-18T23:44:01.766431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm -rf /kaggle/working/landmarks_files.zip","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.768404Z","iopub.execute_input":"2023-08-18T23:44:01.768711Z","iopub.status.idle":"2023-08-18T23:44:01.777044Z","shell.execute_reply.started":"2023-08-18T23:44:01.768684Z","shell.execute_reply":"2023-08-18T23:44:01.776121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cp -r /kaggle/input/wlasl-landmarks/main.csv /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.778205Z","iopub.execute_input":"2023-08-18T23:44:01.778507Z","iopub.status.idle":"2023-08-18T23:44:01.785802Z","shell.execute_reply.started":"2023-08-18T23:44:01.77848Z","shell.execute_reply":"2023-08-18T23:44:01.785036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !kaggle datasets init -p /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.786727Z","iopub.execute_input":"2023-08-18T23:44:01.787Z","iopub.status.idle":"2023-08-18T23:44:01.794091Z","shell.execute_reply.started":"2023-08-18T23:44:01.786975Z","shell.execute_reply":"2023-08-18T23:44:01.793227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import json\n# json.load(open(\"/kaggle/working/dataset-metadata.json\",'r'))","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.795217Z","iopub.execute_input":"2023-08-18T23:44:01.795532Z","iopub.status.idle":"2023-08-18T23:44:01.80286Z","shell.execute_reply.started":"2023-08-18T23:44:01.795505Z","shell.execute_reply":"2023-08-18T23:44:01.801997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta_data = {\n#     'title': 'wlasl-data-feature-extracted',\n#     'id': 'glitchr/wlasl-data-feature-extracted',\n#     'licenses': [{'name': 'CC0-1.0'}]\n# }","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.80385Z","iopub.execute_input":"2023-08-18T23:44:01.804253Z","iopub.status.idle":"2023-08-18T23:44:01.811352Z","shell.execute_reply.started":"2023-08-18T23:44:01.804225Z","shell.execute_reply":"2023-08-18T23:44:01.810583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# json.dump(meta_data,open(\"/kaggle/working/dataset-metadata.json\",'w'),indent = 4)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.812396Z","iopub.execute_input":"2023-08-18T23:44:01.812751Z","iopub.status.idle":"2023-08-18T23:44:01.81936Z","shell.execute_reply.started":"2023-08-18T23:44:01.812723Z","shell.execute_reply":"2023-08-18T23:44:01.818484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# json.load(open(\"/kaggle/working/dataset-metadata.json\",'r'))","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.820421Z","iopub.execute_input":"2023-08-18T23:44:01.820772Z","iopub.status.idle":"2023-08-18T23:44:01.826918Z","shell.execute_reply.started":"2023-08-18T23:44:01.820744Z","shell.execute_reply":"2023-08-18T23:44:01.826105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cp -r /kaggle/input/wlasl-landmarks/* /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.828027Z","iopub.execute_input":"2023-08-18T23:44:01.828395Z","iopub.status.idle":"2023-08-18T23:44:01.834669Z","shell.execute_reply.started":"2023-08-18T23:44:01.828365Z","shell.execute_reply":"2023-08-18T23:44:01.833844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !kaggle datasets create -p /kaggle/working/ --dir-mode 'zip'","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:01.835732Z","iopub.execute_input":"2023-08-18T23:44:01.836039Z","iopub.status.idle":"2023-08-18T23:44:01.842723Z","shell.execute_reply.started":"2023-08-18T23:44:01.836012Z","shell.execute_reply":"2023-08-18T23:44:01.841896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\nMAX_LEN = 58\nCROP_LEN = MAX_LEN\nNUM_CLASSES  = 2000\nPAD = -100.\nNOSE=[\n    1,2,98,327\n]\nLNOSE = [98]\nRNOSE = [327]\nLIP = [ 0, \n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\nLLIP = [84,181,91,146,61,185,40,39,37,87,178,88,95,78,191,80,81,82]\nRLIP = [314,405,321,375,291,409,270,269,267,317,402,318,324,308,415,310,311,312]\n\nPOSE = [500, 502, 504, 501, 503, 505, 512, 513]\nLPOSE = [513,505,503,501]\nRPOSE = [512,504,502,500]\n\nREYE = [\n    33, 7, 163, 144, 145, 153, 154, 155, 133,\n    246, 161, 160, 159, 158, 157, 173,\n]\nLEYE = [\n    263, 249, 390, 373, 374, 380, 381, 382, 362,\n    466, 388, 387, 386, 385, 384, 398,\n]\n\nLHAND = np.arange(468, 489).tolist()\nRHAND = np.arange(522, 543).tolist()\n\nPOINT_LANDMARKS = LIP + LHAND + RHAND + NOSE + REYE + LEYE #+POSE\n\nNUM_NODES = len(POINT_LANDMARKS)\nCHANNELS = 6*NUM_NODES\nCHANNELS = 2*NUM_NODES\n\n\nprint(NUM_NODES)\nprint(CHANNELS)\n\ndef interp1d_(x, target_len, method='random'):\n    length = tf.shape(x)[1]\n    target_len = tf.maximum(1,target_len)\n    if method == 'random':\n        if tf.random.uniform(()) < 0.33:\n            x = tf.image.resize(x, (target_len,tf.shape(x)[1]),'bilinear')\n        else:\n            if tf.random.uniform(()) < 0.5:\n                x = tf.image.resize(x, (target_len,tf.shape(x)[1]),'bicubic')\n            else:\n                x = tf.image.resize(x, (target_len,tf.shape(x)[1]),'nearest')\n    else:\n        x = tf.image.resize(x, (target_len,tf.shape(x)[1]),method)\n    return x\n\ndef tf_nan_mean(x, axis=0, keepdims=False):\n    return tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis, keepdims=keepdims) / tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis, keepdims=keepdims)\n\ndef tf_nan_std(x, center=None, axis=0, keepdims=False):\n    if center is None:\n        center = tf_nan_mean(x, axis=axis,  keepdims=True)\n    d = x - center\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis, keepdims=keepdims))\n\nclass Preprocess(tf.keras.layers.Layer):\n    def __init__(self, max_len=MAX_LEN, point_landmarks=POINT_LANDMARKS, **kwargs):\n        super().__init__(**kwargs)\n        self.max_len = max_len\n        self.point_landmarks = point_landmarks\n\n    def call(self, inputs):\n        if tf.rank(inputs) == 3:\n            x = inputs[None,...]\n        else:\n            x = inputs\n        \n        mean = tf_nan_mean(tf.gather(x, [17], axis=2), axis=[1,2], keepdims=True)\n        mean = tf.where(tf.math.is_nan(mean), tf.constant(0.5,x.dtype), mean)\n        x = tf.gather(x, self.point_landmarks, axis=2) #N,T,P,C\n        std = tf_nan_std(x, center=mean, axis=[1,2], keepdims=True)\n        \n        x = (x - mean)/std\n\n        if self.max_len is not None:\n            x = x[:,:self.max_len]\n        length = tf.shape(x)[1]\n        x = x[...,:2]\n\n#         dx = tf.cond(tf.shape(x)[1]>1,lambda:tf.pad(x[:,1:] - x[:,:-1], [[0,0],[0,1],[0,0],[0,0]]),lambda:tf.zeros_like(x))\n\n#         dx2 = tf.cond(tf.shape(x)[1]>2,lambda:tf.pad(x[:,2:] - x[:,:-2], [[0,0],[0,2],[0,0],[0,0]]),lambda:tf.zeros_like(x))\n\n#         x = tf.concat([\n#             tf.reshape(x, (-1,length,2*len(self.point_landmarks))),\n#             tf.reshape(dx, (-1,length,2*len(self.point_landmarks))),\n#             tf.reshape(dx2, (-1,length,2*len(self.point_landmarks))),\n#         ], axis = -1)\n        x =  tf.reshape(x, (-1,length,2*len(self.point_landmarks)))\n\n        x = tf.where(tf.math.is_nan(x),tf.constant(0.,x.dtype),x)\n        \n        return x\ndef decode_tfrec(record_bytes):\n    features = tf.io.parse_single_example(record_bytes, {\n        'coordinates': tf.io.FixedLenFeature([], tf.string),\n        'sign': tf.io.FixedLenFeature([], tf.int64),\n    })\n    out = {}\n    out['coordinates']  = tf.reshape(tf.io.decode_raw(features['coordinates'], tf.float32), (-1,ROWS_PER_FRAME,3))\n    out['sign'] = features['sign']\n    return out\n\ndef filter_nans_tf(x, ref_point=POINT_LANDMARKS):\n    mask = tf.math.logical_not(tf.reduce_all(tf.math.is_nan(tf.gather(x,ref_point,axis=1)), axis=[-2,-1]))\n    x = tf.boolean_mask(x, mask, axis=0)\n    return x\n\ndef preprocess(x, augment=False, max_len=MAX_LEN,NUM_C = NUM_CLASSES):\n    coord = x['coordinates']\n    coord = filter_nans_tf(coord)\n    if augment:\n        coord = augment_fn(coord, max_len=max_len)\n    coord = tf.ensure_shape(coord, (None,ROWS_PER_FRAME,3))\n    \n    return tf.cast(Preprocess(max_len=max_len)(coord)[0],tf.float32), tf.one_hot(x['sign'], NUM_C)\n\ndef flip_lr(x):\n    x,y,z = tf.unstack(x, axis=-1)\n    x = 1-x\n    new_x = tf.stack([x,y,z], -1)\n    new_x = tf.transpose(new_x, [1,0,2])\n    lhand = tf.gather(new_x, LHAND, axis=0)\n    rhand = tf.gather(new_x, RHAND, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LHAND)[...,None], rhand)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RHAND)[...,None], lhand)\n    llip = tf.gather(new_x, LLIP, axis=0)\n    rlip = tf.gather(new_x, RLIP, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LLIP)[...,None], rlip)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RLIP)[...,None], llip)\n    lpose = tf.gather(new_x, LPOSE, axis=0)\n    rpose = tf.gather(new_x, RPOSE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LPOSE)[...,None], rpose)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RPOSE)[...,None], lpose)\n    leye = tf.gather(new_x, LEYE, axis=0)\n    reye = tf.gather(new_x, REYE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LEYE)[...,None], reye)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(REYE)[...,None], leye)\n    lnose = tf.gather(new_x, LNOSE, axis=0)\n    rnose = tf.gather(new_x, RNOSE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LNOSE)[...,None], rnose)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RNOSE)[...,None], lnose)\n    new_x = tf.transpose(new_x, [1,0,2])\n    return new_x\n\ndef resample(x, rate=(0.8,1.2)):\n    rate = tf.random.uniform((), rate[0], rate[1])\n    length = tf.shape(x)[0]\n    new_size = tf.cast(rate*tf.cast(length,tf.float32), tf.int32)\n    new_x = interp1d_(x, new_size)\n    return new_x\n\ndef spatial_random_affine(xyz,\n    scale  = (0.8,1.2),\n    shear = (-0.15,0.15),\n    shift  = (-0.1,0.1),\n    degree = (-30,30),\n):\n    center = tf.constant([0.5,0.5])\n    if scale is not None:\n        scale = tf.random.uniform((),*scale)\n        xyz = scale*xyz\n\n    if shear is not None:\n        xy = xyz[...,:2]\n        z = xyz[...,2:]\n        shear_x = shear_y = tf.random.uniform((),*shear)\n        if tf.random.uniform(()) < 0.5:\n            shear_x = 0.\n        else:\n            shear_y = 0.\n        shear_mat = tf.identity([\n            [1.,shear_x],\n            [shear_y,1.]\n        ])\n        xy = xy @ shear_mat\n        center = center + [shear_y, shear_x]\n        xyz = tf.concat([xy,z], axis=-1)\n\n    if degree is not None:\n        xy = xyz[...,:2]\n        z = xyz[...,2:]\n        xy -= center\n        degree = tf.random.uniform((),*degree)\n        radian = degree/180*np.pi\n        c = tf.math.cos(radian)\n        s = tf.math.sin(radian)\n        rotate_mat = tf.identity([\n            [c,s],\n            [-s, c],\n        ])\n        xy = xy @ rotate_mat\n        xy = xy + center\n        xyz = tf.concat([xy,z], axis=-1)\n\n    if shift is not None:\n        shift = tf.random.uniform((),*shift)\n        xyz = xyz + shift\n\n    return xyz\n\ndef temporal_crop(x, length=MAX_LEN):\n    l = tf.shape(x)[0]\n    offset = tf.random.uniform((), 0, tf.clip_by_value(l-length,1,length), dtype=tf.int32)\n    x = x[offset:offset+length]\n    return x\n\ndef temporal_mask(x, size=(0.2,0.4), mask_value=float('nan')):\n    l = tf.shape(x)[0]\n    mask_size = tf.random.uniform((), *size)\n    mask_size = tf.cast(tf.cast(l, tf.float32) * mask_size, tf.int32)\n    mask_offset = tf.random.uniform((), 0, tf.clip_by_value(l-mask_size,1,l), dtype=tf.int32)\n    x = tf.tensor_scatter_nd_update(x,tf.range(mask_offset, mask_offset+mask_size)[...,None],tf.fill([mask_size,543,3],mask_value))\n    return x\n\ndef spatial_mask(x, size=(0.2,0.4), mask_value=float('nan')):\n    mask_offset_y = tf.random.uniform(())\n    mask_offset_x = tf.random.uniform(())\n    mask_size = tf.random.uniform((), *size)\n    mask_x = (mask_offset_x<x[...,0]) & (x[...,0] < mask_offset_x + mask_size)\n    mask_y = (mask_offset_y<x[...,1]) & (x[...,1] < mask_offset_y + mask_size)\n    mask = mask_x & mask_y\n    x = tf.where(mask[...,None], mask_value, x)\n    return x\n\ndef augment_fn(x, always=False, max_len=None):\n    if tf.random.uniform(())<0.8 or always:\n        x = resample(x, (0.5,1.5))\n    if tf.random.uniform(())<0.5 or always:\n        x = flip_lr(x)\n    if max_len is not None:\n        x = temporal_crop(x, max_len)\n    if tf.random.uniform(())<0.75 or always:\n        x = spatial_random_affine(x)\n    if tf.random.uniform(())<0.5 or always:\n        x = temporal_mask(x)\n    if tf.random.uniform(())<0.5 or always:\n        x = spatial_mask(x)\n    return x\n\ndef get_tfrec_dataset(tfrecords, NUM_C,batch_size=64, max_len=MAX_LEN, drop_remainder=False, augment=False, shuffle=False, repeat=False):\n    # Initialize dataset with TFRecords\n    ds = tf.data.TFRecordDataset(tfrecords, num_parallel_reads=tf.data.AUTOTUNE, compression_type='GZIP')\n    ds = ds.map(decode_tfrec, tf.data.AUTOTUNE)\n    ds = ds.map(lambda x: preprocess(x, augment=augment, max_len=max_len,NUM_C = NUM_C), tf.data.AUTOTUNE)\n\n    if repeat: \n        ds = ds.repeat()\n        \n    if shuffle:\n        ds = ds.shuffle(shuffle)\n        options = tf.data.Options()\n        options.experimental_deterministic = (False)\n        ds = ds.with_options(options)\n    \n    if batch_size:\n        ds = ds.padded_batch(batch_size, padding_values=PAD, padded_shapes=([max_len,CHANNELS],[NUM_C]), drop_remainder=drop_remainder)\n\n    ds = ds.prefetch(tf.data.AUTOTUNE)\n        \n    return ds\nTRAIN_FILENAMES = '/kaggle/input/wlasl-data-feature-extracted/landmarks_tf_records/00335.tfrecord'\nds = get_tfrec_dataset(TRAIN_FILENAMES,NUM_C = 100, augment=True, batch_size=1024)\nfor x in ds:\n    temp_train = x\n    break\nds","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:24:27.036004Z","iopub.execute_input":"2023-08-19T00:24:27.036574Z","iopub.status.idle":"2023-08-19T00:24:30.146126Z","shell.execute_reply.started":"2023-08-19T00:24:27.036536Z","shell.execute_reply":"2023-08-19T00:24:30.144796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\nimport matplotlib.animation as animation\nfrom matplotlib.animation import FuncAnimation\nimport matplotlib.pyplot as plt\ndef filter_nans(frames):\n    return frames[~np.isnan(frames).all(axis=(-2,-1))]\n\nds = tf.data.TFRecordDataset(TRAIN_FILENAMES, num_parallel_reads=tf.data.AUTOTUNE, compression_type='GZIP')\nds = ds.map(decode_tfrec, tf.data.AUTOTUNE)\nprint(ds)\nfor x in ds:\n    temp = x['coordinates'].numpy()\n    if not len(filter_nans(temp[:,LHAND])) == 0:\n        break\n    \nedges = [(0,1),(1,2),(2,3),(3,4),(0,5),(0,17),(5,6),(6,7),(7,8),(5,9),(9,10),(10,11),(11,12),\n         (9,13),(13,14),(14,15),(15,16),(13,17),(17,18),(18,19),(19,20)]\n\nfig, ax = plt.subplots()\n\ndef plot_frame(frame, edges=[], idxs=[]):\n        \n    frame[np.isnan(frame)] = 0\n    x = list(frame[...,0])\n    y = list(frame[...,1])\n    if len(idxs) == 0:\n        idxs = list(range(len(x)))\n    ax.clear()\n    ax.scatter(x, y, color='dodgerblue')\n    for i in range(len(x)):\n        ax.text(x[i], y[i], idxs[i])\n        \n    for edge in edges:\n        ax.plot([x[edge[0]], x[edge[1]]], [y[edge[0]], y[edge[1]]], color='salmon')\n    ax.set_xticks([])\n    ax.set_yticks([])\n    ax.set_xticklabels([])\n    ax.set_yticklabels([])\n\ndef animate_frames(frames, edges=[], idxs=[]):\n    anim = FuncAnimation(fig, lambda frame: plot_frame(frame, edges, idxs), frames=frames, interval=100)\n    return HTML(anim.to_jshtml())","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:24:30.147876Z","iopub.execute_input":"2023-08-19T00:24:30.148208Z","iopub.status.idle":"2023-08-19T00:24:30.402099Z","shell.execute_reply.started":"2023-08-19T00:24:30.148163Z","shell.execute_reply":"2023-08-19T00:24:30.401023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"animate_frames(filter_nans(temp[:,LHAND]),edges=edges)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:24:30.403331Z","iopub.execute_input":"2023-08-19T00:24:30.403638Z","iopub.status.idle":"2023-08-19T00:24:30.415798Z","shell.execute_reply.started":"2023-08-19T00:24:30.403604Z","shell.execute_reply":"2023-08-19T00:24:30.414902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"animate_frames(filter_nans(temp[:,POINT_LANDMARKS]))","metadata":{"execution":{"iopub.status.busy":"2023-08-18T21:36:35.281206Z","iopub.execute_input":"2023-08-18T21:36:35.281482Z","iopub.status.idle":"2023-08-18T21:36:48.69747Z","shell.execute_reply.started":"2023-08-18T21:36:35.281458Z","shell.execute_reply":"2023-08-18T21:36:48.696437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ECA(tf.keras.layers.Layer):\n    def __init__(self, kernel_size=5, **kwargs):\n        super().__init__(**kwargs)\n        self.supports_masking = True\n        self.kernel_size = kernel_size\n        self.conv = tf.keras.layers.Conv1D(1, kernel_size=kernel_size, strides=1, padding=\"same\", use_bias=False)\n\n    def call(self, inputs, mask=None):\n        nn = tf.keras.layers.GlobalAveragePooling1D()(inputs, mask=mask)\n        nn = tf.expand_dims(nn, -1)\n        nn = self.conv(nn)\n        nn = tf.squeeze(nn, -1)\n        nn = tf.nn.sigmoid(nn)\n        nn = nn[:,None,:]\n        return inputs * nn\n\nclass LateDropout(tf.keras.layers.Layer):\n    def __init__(self, rate, noise_shape=None, start_step=0, **kwargs):\n        super().__init__(**kwargs)\n        self.supports_masking = True\n        self.rate = rate\n        self.start_step = start_step\n        self.dropout = tf.keras.layers.Dropout(rate, noise_shape=noise_shape)\n      \n    def build(self, input_shape):\n        super().build(input_shape)\n        agg = tf.VariableAggregation.ONLY_FIRST_REPLICA\n        self._train_counter = tf.Variable(0, dtype=\"int64\", aggregation=agg, trainable=False)\n\n    def call(self, inputs, training=False):\n        x = tf.cond(self._train_counter < self.start_step, lambda:inputs, lambda:self.dropout(inputs, training=training))\n        if training:\n            self._train_counter.assign_add(1)\n        return x\n\nclass CausalDWConv1D(tf.keras.layers.Layer):\n    def __init__(self, \n        kernel_size=17,\n        dilation_rate=1,\n        use_bias=False,\n        depthwise_initializer='glorot_uniform',\n        name='', **kwargs):\n        super().__init__(name=name,**kwargs)\n        self.causal_pad = tf.keras.layers.ZeroPadding1D((dilation_rate*(kernel_size-1),0),name=name + '_pad')\n        self.dw_conv = tf.keras.layers.DepthwiseConv1D(\n                            kernel_size,\n                            strides=1,\n                            dilation_rate=dilation_rate,\n                            padding='valid',\n                            use_bias=use_bias,\n                            depthwise_initializer=depthwise_initializer,\n                            name=name + '_dwconv')\n        self.supports_masking = True\n        \n    def call(self, inputs):\n        x = self.causal_pad(inputs)\n        x = self.dw_conv(x)\n        return x\n\ndef Conv1DBlock(channel_size,\n          kernel_size,\n          dilation_rate=1,\n          drop_rate=0.0,\n          expand_ratio=2,\n          se_ratio=0.25,\n          activation='swish',\n          name=None):\n    '''\n    efficient conv1d block, @hoyso48\n    '''\n    if name is None:\n        name = str(tf.keras.backend.get_uid(\"mbblock\"))\n    # Expansion phase\n    def apply(inputs):\n        channels_in = tf.keras.backend.int_shape(inputs)[-1]\n        channels_expand = channels_in * expand_ratio\n\n        skip = inputs\n\n        x = tf.keras.layers.Dense(\n            channels_expand,\n            use_bias=True,\n            activation=activation,\n            name=name + '_expand_conv')(inputs)\n\n        # Depthwise Convolution\n        x = CausalDWConv1D(kernel_size,\n            dilation_rate=dilation_rate,\n            use_bias=False,\n            name=name + '_dwconv')(x)\n\n        x = tf.keras.layers.BatchNormalization(momentum=0.95, name=name + '_bn')(x)\n\n        x  = ECA()(x)\n\n        x = tf.keras.layers.Dense(\n            channel_size,\n            use_bias=True,\n            name=name + '_project_conv')(x)\n\n        if drop_rate > 0:\n            x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1), name=name + '_drop')(x)\n\n        if (channels_in == channel_size):\n            x = tf.keras.layers.add([x, skip], name=name + '_add')\n        return x\n\n    return apply\nclass MultiHeadSelfAttention(tf.keras.layers.Layer):\n    def __init__(self, dim=256, num_heads=4, dropout=0, **kwargs):\n        super().__init__(**kwargs)\n        self.dim = dim\n        self.scale = self.dim ** -0.5\n        self.num_heads = num_heads\n        self.qkv = tf.keras.layers.Dense(3 * dim, use_bias=False)\n        self.drop1 = tf.keras.layers.Dropout(dropout)\n        self.proj = tf.keras.layers.Dense(dim, use_bias=False)\n        self.supports_masking = True\n\n    def call(self, inputs, mask=None):\n        qkv = self.qkv(inputs)\n        qkv = tf.keras.layers.Permute((2, 1, 3))(tf.keras.layers.Reshape((-1, self.num_heads, self.dim * 3 // self.num_heads))(qkv))\n        q, k, v = tf.split(qkv, [self.dim // self.num_heads] * 3, axis=-1)\n\n        attn = tf.matmul(q, k, transpose_b=True) * self.scale\n\n        if mask is not None:\n            mask = mask[:, None, None, :]\n\n        attn = tf.keras.layers.Softmax(axis=-1)(attn, mask=mask)\n        attn = self.drop1(attn)\n\n        x = attn @ v\n        x = tf.keras.layers.Reshape((-1, self.dim))(tf.keras.layers.Permute((2, 1, 3))(x))\n        x = self.proj(x)\n        return x\n\n\ndef TransformerBlock(dim=256, num_heads=4, expand=4, attn_dropout=0.2, drop_rate=0.2, activation='swish'):\n    def apply(inputs):\n        x = inputs\n        x = tf.keras.layers.BatchNormalization(momentum=0.95)(x)\n        x = MultiHeadSelfAttention(dim=dim,num_heads=num_heads,dropout=attn_dropout)(x)\n        x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1))(x)\n        x = tf.keras.layers.Add()([inputs, x])\n        attn_out = x\n\n        x = tf.keras.layers.BatchNormalization(momentum=0.95)(x)\n        x = tf.keras.layers.Dense(dim*expand, use_bias=False, activation=activation)(x)\n        x = tf.keras.layers.Dense(dim, use_bias=False)(x)\n        x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1))(x)\n        x = tf.keras.layers.Add()([attn_out, x])\n        return x\n    return apply\ndef get_model(NC,max_len=MAX_LEN, dropout_step=0, dim=192):\n    inp = tf.keras.Input((max_len,CHANNELS))\n    x = tf.keras.layers.Masking(mask_value=PAD,input_shape=(max_len,CHANNELS))(inp)\n    ksize = 17\n    x = tf.keras.layers.Dense(dim, use_bias=False,name='stem_conv')(x)\n    x = tf.keras.layers.BatchNormalization(momentum=0.95,name='stem_bn')(x)\n\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = TransformerBlock(dim,expand=2)(x)\n\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = TransformerBlock(dim,expand=2)(x)\n\n    if dim == 384: #for the 4x sized model\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = TransformerBlock(dim,expand=2)(x)\n\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = TransformerBlock(dim,expand=2)(x)\n\n    x = tf.keras.layers.Dense(dim*2,activation=None,name='top_conv')(x)\n    x = tf.keras.layers.GlobalAveragePooling1D()(x)\n    x = LateDropout(0.8, start_step=dropout_step)(x)\n    x = tf.keras.layers.Dense(NC,name='classifier')(x)\n    return tf.keras.Model(inp, x)\n\nmodel = get_model(100)\n# model.summary()\ny = model(temp_train[0])\ntf.keras.losses.CategoricalCrossentropy(from_logits=True)(temp_train[1],y)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:24:30.417895Z","iopub.execute_input":"2023-08-19T00:24:30.41823Z","iopub.status.idle":"2023-08-19T00:24:31.889276Z","shell.execute_reply.started":"2023-08-19T00:24:30.418201Z","shell.execute_reply":"2023-08-19T00:24:31.888045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x in model.layers:\n    if not x.supports_masking:\n        print(x.supports_masking, x.name)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:24:31.891307Z","iopub.execute_input":"2023-08-19T00:24:31.891633Z","iopub.status.idle":"2023-08-19T00:24:31.896532Z","shell.execute_reply.started":"2023-08-19T00:24:31.891604Z","shell.execute_reply":"2023-08-19T00:24:31.895548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['tfr_path'] = df['video_id'].progress_apply(lambda x: os.path.join(\"/kaggle/input/wlasl-data-feature-extracted/landmarks_tf_records\",str(x).zfill(5)+\".tfrecord\"))","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:24:32.042161Z","iopub.execute_input":"2023-08-19T00:24:32.042476Z","iopub.status.idle":"2023-08-19T00:24:32.116787Z","shell.execute_reply.started":"2023-08-19T00:24:32.042448Z","shell.execute_reply":"2023-08-19T00:24:32.115851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_100 = pd.read_json(\"/kaggle/input/wlasl-processed/nslt_100.json\").T.reset_index().rename(columns = {\"index\":\"video_id\"})\ntemp_df = df[df['video_id'].isin(df_100['video_id'])].copy()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:24:32.313879Z","iopub.execute_input":"2023-08-19T00:24:32.314158Z","iopub.status.idle":"2023-08-19T00:24:32.904405Z","shell.execute_reply.started":"2023-08-19T00:24:32.314131Z","shell.execute_reply":"2023-08-19T00:24:32.903226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_strategy(device='TPU-VM'):\n    if \"TPU\" in device:\n        tpu = 'local' if device=='TPU-VM' else None\n        print(\"connecting to TPU...\")\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu=tpu)\n        strategy = tf.distribute.TPUStrategy(tpu)\n\n    if device == \"GPU\"  or device==\"CPU\":\n        ngpu = len(tf.config.experimental.list_physical_devices('GPU'))\n        if ngpu>1:\n            print(\"Using multi GPU\")\n            strategy = tf.distribute.MirroredStrategy()\n        elif ngpu==1:\n            print(\"Using single GPU\")\n            strategy = tf.distribute.get_strategy()\n        else:\n            print(\"Using CPU\")\n            strategy = tf.distribute.get_strategy()\n\n    if device == \"GPU\":\n        print(\"Num GPUs Available: \", ngpu)\n\n    AUTO     = tf.data.experimental.AUTOTUNE\n    REPLICAS = strategy.num_replicas_in_sync\n    print(f'REPLICAS: {REPLICAS}')\n    \n    return strategy, REPLICAS\ntry:\n    STRATEGY, N_REPLICAS = get_strategy()\nexcept:\n    STRATEGY, N_REPLICAS = get_strategy('GPU')    ","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:44:11.617757Z","iopub.execute_input":"2023-08-18T23:44:11.618212Z","iopub.status.idle":"2023-08-18T23:44:17.379972Z","shell.execute_reply.started":"2023-08-18T23:44:11.618168Z","shell.execute_reply":"2023-08-18T23:44:17.378397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    \ndef train_fold(CFG, train_files, valid_files=None, strategy=STRATEGY, summary=True):\n    seed_everything(CFG.seed)\n    tf.keras.backend.clear_session()\n    gc.collect()\n    tf.config.optimizer.set_jit(True)\n        \n    if CFG.fp16:\n        try:\n            policy = mixed_precision.Policy('mixed_bfloat16')\n            mixed_precision.set_global_policy(policy)\n        except:\n            policy = mixed_precision.Policy('mixed_float16')\n            mixed_precision.set_global_policy(policy)\n    else:\n        policy = mixed_precision.Policy('float32')\n        mixed_precision.set_global_policy(policy)\n\n    if valid_files is not None:\n        train_ds = get_tfrec_dataset(train_files,NUM_C = CFG.NUM_CLASSES, batch_size=CFG.batch_size, max_len=CFG.max_len, drop_remainder=True, augment=True, repeat=True, shuffle=len(train_files)//2)\n        valid_ds = get_tfrec_dataset(valid_files, NUM_C = CFG.NUM_CLASSES,batch_size=CFG.batch_size, max_len=CFG.max_len, drop_remainder=False, repeat=False, shuffle=False)\n    else:\n        train_ds = get_tfrec_dataset(train_files,NUM_C = CFG.NUM_CLASSES, batch_size=CFG.batch_size, max_len=CFG.max_len, drop_remainder=True, augment=True, repeat=True, shuffle=len(train_files)//2)\n        valid_ds = None\n        valid_files = []\n    \n    num_train = len(train_files)\n    num_valid = len(valid_files)\n    steps_per_epoch = num_train//CFG.batch_size\n    seed_everything(CFG.seed)\n    with strategy.scope():\n        dropout_step = CFG.dropout_start_epoch * steps_per_epoch\n        model = get_model(NC = CFG.NUM_CLASSES,max_len=CFG.max_len, dropout_step=dropout_step, dim=CFG.dim)\n\n        schedule = OneCycleLR(CFG.lr, CFG.epoch, warmup_epochs=CFG.epoch*CFG.warmup, steps_per_epoch=steps_per_epoch, resume_epoch=CFG.resume, decay_epochs=CFG.epoch, lr_min=CFG.lr_min, decay_type=CFG.decay_type, warmup_type='linear')\n        decay_schedule = OneCycleLR(CFG.lr*CFG.weight_decay, CFG.epoch, warmup_epochs=CFG.epoch*CFG.warmup, steps_per_epoch=steps_per_epoch, resume_epoch=CFG.resume, decay_epochs=CFG.epoch, lr_min=CFG.lr_min*CFG.weight_decay, decay_type=CFG.decay_type, warmup_type='linear')\n                \n        awp_step = CFG.awp_start_epoch * steps_per_epoch\n#         if CFG.fgm:\n#             model = FGM(model.input, model.output, delta=CFG.awp_lambda, eps=0., start_step=awp_step)\n#         elif CFG.awp:\n#             model = AWP(model.input, model.output, delta=CFG.awp_lambda, eps=0., start_step=awp_step)\n\n        opt = tfa.optimizers.RectifiedAdam(learning_rate=schedule, weight_decay=decay_schedule, sma_threshold=4, clipvalue=1.)\n        opt = tfa.optimizers.Lookahead(opt,sync_period=5)\n\n        model.compile(\n            optimizer=opt,\n            loss=[tf.keras.losses.CategoricalCrossentropy(from_logits=True)], #[tf.keras.losses.CategoricalCrossentropy(from_logits=True)],\n            metrics=[\n                [\n                tf.keras.metrics.CategoricalAccuracy(),\n                ],\n            ],\n            steps_per_execution=steps_per_epoch,\n        )\n    \n    if summary:\n        print()\n        model.summary()\n        print()\n        print(train_ds, valid_ds)\n        print()\n        schedule.plot()\n        print()\n        init=False\n    print(f'-----------------')\n    print(f'train:{num_train} valid:{num_valid}')\n    print()\n    \n    if CFG.resume:\n        print(f'resume from epoch{CFG.resume}')\n        model.load_weights(f'{CFG.output_dir}/{CFG.comment}-last.h5')\n        if train_ds is not None:\n            model.evaluate(train_ds.take(steps_per_epoch))\n        if valid_ds is not None:\n            model.evaluate(valid_ds)\n\n    logger = tf.keras.callbacks.CSVLogger(f'{CFG.output_dir}/{CFG.comment}-logs.csv')\n    sv_loss = tf.keras.callbacks.ModelCheckpoint(f'{CFG.output_dir}/{CFG.comment}-best.h5', monitor='val_loss', verbose=1,\n                save_weights_only=True, mode='min', save_freq='epoch')\n    snap = Snapshot(f'{CFG.output_dir}/{CFG.comment}', CFG.snapshot_epochs)\n    swa = SWA(f'{CFG.output_dir}/{CFG.comment}', CFG.swa_epochs, strategy=strategy, train_ds=train_ds, valid_ds=valid_ds, valid_steps=-(num_valid//-CFG.batch_size))\n    callbacks = []\n    if CFG.save_output:\n        callbacks.append(logger)\n        callbacks.append(snap)\n        callbacks.append(swa)\n\n#         if valid_files is not None:\n#             callbacks.append(sv_loss)\n        \n    history = model.fit(\n        train_ds,\n        epochs=CFG.epoch-CFG.resume,\n        steps_per_epoch=steps_per_epoch,\n        callbacks=callbacks,\n        validation_data=valid_ds,\n        verbose=CFG.verbose,\n        validation_steps=-(num_valid//-CFG.batch_size)\n    )\n\n    return model, history\nfrom sklearn.model_selection import train_test_split\ndef train_folds(CFG,df, strategy=STRATEGY, summary=True):\n    train_files = df[df['split'].isin(['train','val'])]['tfr_path'].tolist()\n#     valid_files = df[df['split'].isin(['val'])]['tfr_path'].tolist()\n    valid_files = None\n\n    return train_fold(CFG, train_files, valid_files = valid_files, strategy=strategy, summary=summary)\nimport random, gc\nclass CFG:\n    n_splits = 5\n    save_output = True\n    output_dir = '/kaggle/working'\n    \n    seed = 42\n    verbose = 2 #0) silent 1) progress bar 2) one line per epoch\n    \n    max_len = MAX_LEN\n    replicas = N_REPLICAS\n    lr = 5e-4 * 8\n    weight_decay = 0.1\n    lr_min = 1e-6\n    epoch = 600 \n    warmup = 0\n    batch_size = 4 * replicas\n    snapshot_epochs = []\n    swa_epochs = [] #list(range(epoch//2,epoch+1))\n    NUM_CLASSES = 100\n    fp16 = True\n    fgm = False\n    awp = True\n    awp_lambda = 0.2\n    awp_start_epoch = 15\n    dropout_start_epoch = 15\n    resume = 0\n    decay_type = 'cosine'\n    dim = 192\n    comment = f'islr-fp16-192-8-seed{seed}'\n","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:33:10.883112Z","iopub.execute_input":"2023-08-19T00:33:10.883621Z","iopub.status.idle":"2023-08-19T00:33:10.919015Z","shell.execute_reply.started":"2023-08-19T00:33:10.88358Z","shell.execute_reply":"2023-08-19T00:33:10.917928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_folds(CFG,temp_df)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:33:11.509487Z","iopub.execute_input":"2023-08-19T00:33:11.510257Z","iopub.status.idle":"2023-08-19T00:42:05.05005Z","shell.execute_reply.started":"2023-08-19T00:33:11.510224Z","shell.execute_reply":"2023-08-19T00:42:05.048472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = temp_df[temp_df['split']=='test']\ntest_files = temp_df[temp_df['split']=='test']['tfr_path'].tolist()\nprint(f\" Test Files length: {len(test_files)}\")\nNUM_C = 100\nbatch_size= 32\ntest_ds = get_tfrec_dataset(test_files, NUM_C = NUM_C, batch_size=batch_size, max_len=MAX_LEN, drop_remainder=False, repeat=False, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:42:35.750822Z","iopub.execute_input":"2023-08-19T00:42:35.751949Z","iopub.status.idle":"2023-08-19T00:42:35.917811Z","shell.execute_reply.started":"2023-08-19T00:42:35.751908Z","shell.execute_reply":"2023-08-19T00:42:35.916458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_xy(ds):\n    X,Y = [],[]\n    for x,y in ds:\n        X.append(x)\n        Y.append(y.numpy().argmax(-1))\n    return np.vstack(X),np.hstack(Y)\nxtest,ytest = get_xy(test_ds)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:42:36.107282Z","iopub.execute_input":"2023-08-19T00:42:36.108357Z","iopub.status.idle":"2023-08-19T00:42:36.291769Z","shell.execute_reply.started":"2023-08-19T00:42:36.108318Z","shell.execute_reply":"2023-08-19T00:42:36.29034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model  = get_model(CFG.NUM_CLASSES,dim = 192)\nbest_model.load_weights('/kaggle/working/islr-fp16-192-8-seed42-last.h5')","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:42:37.940668Z","iopub.execute_input":"2023-08-19T00:42:37.941789Z","iopub.status.idle":"2023-08-19T00:42:39.243241Z","shell.execute_reply.started":"2023-08-19T00:42:37.941743Z","shell.execute_reply":"2023-08-19T00:42:39.241829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = best_model.predict(xtest).argmax(-1)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:42:39.245206Z","iopub.execute_input":"2023-08-19T00:42:39.245506Z","iopub.status.idle":"2023-08-19T00:42:41.1637Z","shell.execute_reply.started":"2023-08-19T00:42:39.245479Z","shell.execute_reply":"2023-08-19T00:42:41.16231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score,ConfusionMatrixDisplay,confusion_matrix\nprint(f\"Test Accuracy: {(accuracy_score(ytest,y_pred)*100):.2f} %\")","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:42:41.165725Z","iopub.execute_input":"2023-08-19T00:42:41.166134Z","iopub.status.idle":"2023-08-19T00:42:41.173808Z","shell.execute_reply.started":"2023-08-19T00:42:41.1661Z","shell.execute_reply":"2023-08-19T00:42:41.172595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:42:46.290055Z","iopub.execute_input":"2023-08-19T00:42:46.290485Z","iopub.status.idle":"2023-08-19T00:42:51.962497Z","shell.execute_reply.started":"2023-08-19T00:42:46.290451Z","shell.execute_reply":"2023-08-19T00:42:51.960972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\ncm = confusion_matrix(ytest, y_pred,labels = test_df['label'].tolist())\n\nplt.figure(figsize = (15,15))\nsns.heatmap(cm, annot=True,)\nplt.savefig(\"Confusion Matrix (WLASL-100).png\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-19T00:42:51.964609Z","iopub.execute_input":"2023-08-19T00:42:51.964955Z","iopub.status.idle":"2023-08-19T00:43:25.952797Z","shell.execute_reply.started":"2023-08-19T00:42:51.964923Z","shell.execute_reply":"2023-08-19T00:43:25.95161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}