{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom skimage.transform import resize\nimport json\nfrom tqdm.notebook import tqdm\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-08-13T00:29:23.398117Z","iopub.execute_input":"2023-08-13T00:29:23.39916Z","iopub.status.idle":"2023-08-13T00:29:32.973548Z","shell.execute_reply.started":"2023-08-13T00:29:23.399094Z","shell.execute_reply":"2023-08-13T00:29:32.972363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inpdir = \"/kaggle/input/asl-fingerspelling\"\ndf = pd.read_csv(f'{inpdir}/train.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2023-08-13T00:29:32.976091Z","iopub.execute_input":"2023-08-13T00:29:32.976978Z","iopub.status.idle":"2023-08-13T00:29:33.186734Z","shell.execute_reply.started":"2023-08-13T00:29:32.976934Z","shell.execute_reply":"2023-08-13T00:29:33.185938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"phrase_bytes\"] = df[\"phrase\"].map(lambda x: x.encode(\"utf-8\"))\ndisplay(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-13T00:29:33.19102Z","iopub.execute_input":"2023-08-13T00:29:33.191649Z","iopub.status.idle":"2023-08-13T00:29:33.240118Z","shell.execute_reply.started":"2023-08-13T00:29:33.191619Z","shell.execute_reply":"2023-08-13T00:29:33.239036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-08-13T00:29:33.242555Z","iopub.execute_input":"2023-08-13T00:29:33.24291Z","iopub.status.idle":"2023-08-13T00:29:49.080455Z","shell.execute_reply.started":"2023-08-13T00:29:33.242882Z","shell.execute_reply":"2023-08-13T00:29:49.079213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NOSE=[\n    1,2,98,327\n]\nLNOSE = [98]\nRNOSE = [327]\nLIP = [ 0, \n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\nLLIP = [84,181,91,146,61,185,40,39,37,87,178,88,95,78,191,80,81,82]\nRLIP = [314,405,321,375,291,409,270,269,267,317,402,318,324,308,415,310,311,312]\n\nPOSE = [490,491,492,493,494,495,496,497,498,499,500,501,502,503,504,505,506,507,508,509,510,511,512,513,514,515,516,517,518,519,520,521,522]\nPOSE = [i-1 for i in POSE]\nREYE = [\n    33, 7, 163, 144, 145, 153, 154, 155, 133,\n    246, 161, 160, 159, 158, 157, 173,\n]\nLEYE = [\n    263, 249, 390, 373, 374, 380, 381, 382, 362,\n    466, 388, 387, 386, 385, 384, 398,\n]\n\nLHAND = np.arange(468, 489).tolist()\nRHAND = np.arange(522, 543).tolist()\n\nPOINT_LANDMARKS = LIP + LHAND + RHAND + NOSE + REYE + LEYE + POSE\n\ndef idx_to_cols(idx_array):\n    sample_cols = data.iloc[:,[i+1 for i in idx_array]].columns\n    patterns = [s[1:] for s in sample_cols]\n    matched_cols = [s for s in data.columns if s[1:] in patterns]\n    return matched_cols\n\nSEL_COLS = idx_to_cols(POINT_LANDMARKS)","metadata":{"execution":{"iopub.status.busy":"2023-08-13T00:30:25.80439Z","iopub.execute_input":"2023-08-13T00:30:25.804804Z","iopub.status.idle":"2023-08-13T00:30:25.858511Z","shell.execute_reply.started":"2023-08-13T00:30:25.804759Z","shell.execute_reply":"2023-08-13T00:30:25.857268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx_to_cols(POSE)","metadata":{"execution":{"iopub.status.busy":"2023-08-13T00:30:26.957866Z","iopub.execute_input":"2023-08-13T00:30:26.958276Z","iopub.status.idle":"2023-08-13T00:30:26.974315Z","shell.execute_reply.started":"2023-08-13T00:30:26.958244Z","shell.execute_reply":"2023-08-13T00:30:26.973188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(SEL_COLS)","metadata":{"execution":{"iopub.status.busy":"2023-07-25T21:03:00.382078Z","iopub.execute_input":"2023-07-25T21:03:00.382559Z","iopub.status.idle":"2023-07-25T21:03:00.40647Z","shell.execute_reply.started":"2023-07-25T21:03:00.382526Z","shell.execute_reply":"2023-07-25T21:03:00.405553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_duplicate(phrase,sequence_id):\n    dup_df = df[(df['phrase']==phrase)&(df['sequence_id']<sequence_id)]\n    if dup_df.empty:\n        return None\n    else:\n        ans = []\n        for i,row in dup_df.iterrows():\n            file_id = row['file_id']\n            seq_id = row['sequence_id']\n            pqfile = f\"{inpdir}/train_landmarks/{file_id}.parquet\"\n            data = load_relevant_data_subset(pqfile)\n            frames = data.iloc[data.index==seq_id]\n            ans.append(frames)\n        return ans","metadata":{"execution":{"iopub.status.busy":"2023-07-25T21:03:00.407849Z","iopub.execute_input":"2023-07-25T21:03:00.408196Z","iopub.status.idle":"2023-07-25T21:03:00.423435Z","shell.execute_reply.started":"2023-07-25T21:03:00.408159Z","shell.execute_reply":"2023-07-25T21:03:00.422233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_relevant_data_subset(pq_path):\n    return pd.read_parquet(pq_path, columns=SEL_COLS)\n\nstat = None\ncnt = 0\n\nfor file_id in tqdm(df.file_id.unique()):\n    pqfile = f\"{inpdir}/train_landmarks/{file_id}.parquet\"\n    if not os.path.isdir(\"tfds-v2\"): os.mkdir(\"tfds-v2\")\n    tffile = f\"tfds-v2/{file_id}.tfrecord\"\n    seq_refs = df.loc[df.file_id == file_id]\n    seqs = load_relevant_data_subset(pqfile)\n    with tf.io.TFRecordWriter(tffile) as file_writer:\n        for seq_id, phrase_bytes,phrase in zip(seq_refs.sequence_id, seq_refs.phrase_bytes,seq_refs.phrase):\n            frames = seqs.iloc[seqs.index == seq_id]\n            frames = pd.DataFrame(data = frames, columns=frames.columns)\n            features = {COL: tf.train.Feature(float_list=tf.train.FloatList(value=frames[COL])) for COL in SEL_COLS}\n            features[\"phrase\"] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[phrase_bytes]))\n            record_bytes = tf.train.Example(features=tf.train.Features(feature=features)).SerializeToString()\n            file_writer.write(record_bytes)\n            ","metadata":{"execution":{"iopub.status.busy":"2023-07-25T21:03:00.442988Z","iopub.execute_input":"2023-07-25T21:03:00.443557Z"},"trusted":true},"execution_count":null,"outputs":[]}]}