{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport tensorflow as tf\nimport json\nimport shutil\nimport sys","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-04T11:53:02.439461Z","iopub.execute_input":"2023-06-04T11:53:02.439884Z","iopub.status.idle":"2023-06-04T11:53:12.868196Z","shell.execute_reply.started":"2023-06-04T11:53:02.439854Z","shell.execute_reply":"2023-06-04T11:53:12.867315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! python --version\ntf.__version__","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:12.86975Z","iopub.execute_input":"2023-06-04T11:53:12.871134Z","iopub.status.idle":"2023-06-04T11:53:13.148806Z","shell.execute_reply.started":"2023-06-04T11:53:12.8711Z","shell.execute_reply":"2023-06-04T11:53:13.146632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 500)  ","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.150496Z","iopub.execute_input":"2023-06-04T11:53:13.150911Z","iopub.status.idle":"2023-06-04T11:53:13.157342Z","shell.execute_reply.started":"2023-06-04T11:53:13.150874Z","shell.execute_reply":"2023-06-04T11:53:13.155493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/asl-fingerspelling/' #'/kaggle/input/asl-signs/'\ndf_train = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nprint(len(df_train))","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.161061Z","iopub.execute_input":"2023-06-04T11:53:13.161476Z","iopub.status.idle":"2023-06-04T11:53:13.304134Z","shell.execute_reply.started":"2023-06-04T11:53:13.16144Z","shell.execute_reply":"2023-06-04T11:53:13.302796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    character_map = json.load(f)\nrev_character_map = {j:i for i,j in character_map.items()}\nprint(character_map)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.305958Z","iopub.execute_input":"2023-06-04T11:53:13.306415Z","iopub.status.idle":"2023-06-04T11:53:13.315484Z","shell.execute_reply.started":"2023-06-04T11:53:13.306377Z","shell.execute_reply":"2023-06-04T11:53:13.313975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"number_of_characters = len(character_map)\nnumber_of_characters","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.316874Z","iopub.execute_input":"2023-06-04T11:53:13.317164Z","iopub.status.idle":"2023-06-04T11:53:13.329873Z","shell.execute_reply.started":"2023-06-04T11:53:13.317141Z","shell.execute_reply":"2023-06-04T11:53:13.327358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# write inference_args.json\nsel_columns= ['frame','x_left_hand_0', 'x_left_hand_1', 'x_left_hand_2', 'x_left_hand_3',\n       'x_left_hand_4', 'x_left_hand_5', 'x_left_hand_6', 'x_left_hand_7',\n       'x_left_hand_8', 'x_left_hand_9', 'x_left_hand_10', 'x_left_hand_11',\n       'x_left_hand_12', 'x_left_hand_13', 'x_left_hand_14', 'x_left_hand_15',\n       'x_left_hand_16', 'x_left_hand_17', 'x_left_hand_18', 'x_left_hand_19','x_left_hand_20',   \n        'y_left_hand_0', 'y_left_hand_1', 'y_left_hand_2', 'y_left_hand_3',\n       'y_left_hand_4', 'y_left_hand_5', 'y_left_hand_6', 'y_left_hand_7',\n       'y_left_hand_8', 'y_left_hand_9', 'y_left_hand_10', 'y_left_hand_11',\n       'y_left_hand_12', 'y_left_hand_13', 'y_left_hand_14', 'y_left_hand_15',\n       'y_left_hand_16', 'y_left_hand_17', 'y_left_hand_18', 'y_left_hand_19','y_left_hand_20',                \n       'x_right_hand_0', 'x_right_hand_1', 'x_right_hand_2', 'x_right_hand_3',\n       'x_right_hand_4', 'x_right_hand_5', 'x_right_hand_6', 'x_right_hand_7',\n       'x_right_hand_8', 'x_right_hand_9', 'x_right_hand_10',\n       'x_right_hand_11', 'x_right_hand_12', 'x_right_hand_13',\n       'x_right_hand_14', 'x_right_hand_15', 'x_right_hand_16',\n       'x_right_hand_17', 'x_right_hand_18', 'x_right_hand_19','x_right_hand_20',\n        'y_right_hand_0', 'y_right_hand_1', 'y_right_hand_2', 'y_right_hand_3',\n       'y_right_hand_4', 'y_right_hand_5', 'y_right_hand_6', 'y_right_hand_7',\n       'y_right_hand_8', 'y_right_hand_9', 'y_right_hand_10',\n       'y_right_hand_11', 'y_right_hand_12', 'y_right_hand_13',\n       'y_right_hand_14', 'y_right_hand_15', 'y_right_hand_16',\n       'y_right_hand_17', 'y_right_hand_18', 'y_right_hand_19',\n       'y_right_hand_20',\n            ]\n\nwith open ('/kaggle/working/inference_args.json', \"w\") as f:\n    json.dump(({\"selected_columns\": sel_columns }), f) # 85 cols incl frame  sel_columns1 for frame only\n# for submission zip - json for selected columns    \nwith open (\"/kaggle/working/inference_args.json\", \"r\") as f:\n    slc = json.load(f)\nuse_columns = slc[\"selected_columns\"]    \njson.dumps(slc,skipkeys=True)        ","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.332365Z","iopub.execute_input":"2023-06-04T11:53:13.33303Z","iopub.status.idle":"2023-06-04T11:53:13.351296Z","shell.execute_reply.started":"2023-06-04T11:53:13.332986Z","shell.execute_reply":"2023-06-04T11:53:13.34996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# per Evaluation how data is loaded\ndef load_relevant_data_subset(pq_path):\n    return pd.read_parquet(pq_path, columns=use_columns)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.355731Z","iopub.execute_input":"2023-06-04T11:53:13.356619Z","iopub.status.idle":"2023-06-04T11:53:13.361351Z","shell.execute_reply.started":"2023-06-04T11:53:13.356523Z","shell.execute_reply":"2023-06-04T11:53:13.360358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/asl-fingerspelling/'\nrow =13005   # example from train to use in preprocessing\nprint(PATH + df_train.path[row])\npq_path = PATH + df_train.path[row]\nlandmark = load_relevant_data_subset(pq_path)\nprint(landmark.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.362555Z","iopub.execute_input":"2023-06-04T11:53:13.362938Z","iopub.status.idle":"2023-06-04T11:53:13.901403Z","shell.execute_reply.started":"2023-06-04T11:53:13.362909Z","shell.execute_reply":"2023-06-04T11:53:13.900231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq = landmark.loc[[df_train.sequence_id[row]]]\nprint(seq.shape)\nseq.head(6)  # view rows with NaNs","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.904448Z","iopub.execute_input":"2023-06-04T11:53:13.904842Z","iopub.status.idle":"2023-06-04T11:53:13.993739Z","shell.execute_reply.started":"2023-06-04T11:53:13.90481Z","shell.execute_reply":"2023-06-04T11:53:13.992255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"number_of_features =len(sel_columns)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:13.995308Z","iopub.execute_input":"2023-06-04T11:53:13.995686Z","iopub.status.idle":"2023-06-04T11:53:14.003745Z","shell.execute_reply.started":"2023-06-04T11:53:13.995653Z","shell.execute_reply":"2023-06-04T11:53:14.001549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for ASL-F  basic testing\nclass FeatureGen(tf.keras.layers.Layer):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n    \n    def call(self, x_in):\n        #print(x_in.shape) #print to check shape of input for tests\n        x0 = x_in\n        xlh = tf.slice(x0, [0,1], [-1,42 ])  # lh slice  x y (n.b. 0 is frame so use 1)\n        xrh = tf.slice(x0, [0,43], [-1,42 ])   # rh slice x y\n\n        # remove NaN rows\n        xrhv = tf.boolean_mask(xrh, tf.reduce_all(~tf.math.is_nan(xrh), axis=1))\n        #x = tf.boolean_mask(x0, tf.reduce_all(~tf.math.is_nan(x0), axis=1))  # axis = -1 seems to look at NaNs in rows and cols so lose all\n        xlhv =tf.boolean_mask(xlh, tf.reduce_all(~tf.math.is_nan(xlh), axis=1))\n        x = tf.cond(tf.math.is_nan(tf.math.zero_fraction(xlhv)), lambda: tf.identity(xrhv), lambda: tf.identity(xlhv))\n        \n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:14.005078Z","iopub.execute_input":"2023-06-04T11:53:14.00539Z","iopub.status.idle":"2023-06-04T11:53:14.019511Z","shell.execute_reply.started":"2023-06-04T11:53:14.005361Z","shell.execute_reply":"2023-06-04T11:53:14.017409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#testf = tf.random.normal([3,85])\nFeatureGen()(seq) #(testf)  # example preprocessing to see NaN rows and hand removed","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:14.021918Z","iopub.execute_input":"2023-06-04T11:53:14.022352Z","iopub.status.idle":"2023-06-04T11:53:14.266046Z","shell.execute_reply.started":"2023-06-04T11:53:14.022313Z","shell.execute_reply":"2023-06-04T11:53:14.264553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example basic model using preprocessing layer\n# based on # see https://www.kaggle.com/code/longyikim/finger-first\nclass MyModel(tf.keras.layers.Layer):\n    def __init__(self):\n        super().__init__()\n        self.prep_inputs = FeatureGen()\n\n    def call(self, x):\n        x_in = self.prep_inputs(tf.cast(x, dtype=tf.float32))\n        S, _ = x_in.shape\n        \n        xc = tf.constant([\n         [0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 1., 0., 0., 0., 0., 0., 0.,\n          0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0.,\n          0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0.,\n          0., 0., 0., 0., 0., 0., 0., 0.], # 10 +      \n         [0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0.,\n          0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0.,\n          0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0., 0.,\n          0., 0., 0., 0., 0., 0., 0., 0.], # 0 space          \n                         ])\n        out =  xc \n        return out\n    \ninputs = tf.keras.Input(shape=(number_of_features), name=\"inputs\")\nmmodel = MyModel()\nx = tf.expand_dims(inputs,0)\nmmodel(inputs)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:14.267623Z","iopub.execute_input":"2023-06-04T11:53:14.267997Z","iopub.status.idle":"2023-06-04T11:53:14.56075Z","shell.execute_reply.started":"2023-06-04T11:53:14.267965Z","shell.execute_reply":"2023-06-04T11:53:14.55968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_inference_model(model):\n    inputs = tf.keras.Input(shape=(number_of_features), name=\"inputs\")\n    \n\n    out = model(inputs)\n\n    # explicitly name the final (identity) layer for the submission format\n    out = tf.keras.layers.Activation(\"linear\", name=\"outputs\")(out)\n\n    inference_model = tf.keras.Model(inputs=inputs, outputs=out)\n    inference_model.compile(loss=\"sparse_categorical_crossentropy\",\n                            metrics=\"accuracy\")\n    return inference_model\n\ninference_model = get_inference_model(mmodel)\ninference_model.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:14.562022Z","iopub.execute_input":"2023-06-04T11:53:14.562291Z","iopub.status.idle":"2023-06-04T11:53:14.673379Z","shell.execute_reply.started":"2023-06-04T11:53:14.562269Z","shell.execute_reply":"2023-06-04T11:53:14.670024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test model \ninference_model(tf.random.normal([23,85])).shape  ","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:14.675815Z","iopub.execute_input":"2023-06-04T11:53:14.676319Z","iopub.status.idle":"2023-06-04T11:53:14.717699Z","shell.execute_reply.started":"2023-06-04T11:53:14.676272Z","shell.execute_reply":"2023-06-04T11:53:14.716085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_model(seq )#.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:14.719269Z","iopub.execute_input":"2023-06-04T11:53:14.719641Z","iopub.status.idle":"2023-06-04T11:53:14.744414Z","shell.execute_reply.started":"2023-06-04T11:53:14.719611Z","shell.execute_reply":"2023-06-04T11:53:14.74285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert model\nconverter = tf.lite.TFLiteConverter.from_keras_model(inference_model)\n\n\nconverter.optimizations = [tf.lite.Optimize.DEFAULT]\nconverter.experimental_new_converter=True\nconverter.target_spec.supported_ops = [tf.lite.OpsSet.TFLITE_BUILTINS,\n                                       tf.lite.OpsSet.SELECT_TF_OPS]\n\ntflite_model = converter.convert()\n\nmodel_path = \"model.tflite\"\n\n# Save the model.\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:14.747024Z","iopub.execute_input":"2023-06-04T11:53:14.747429Z","iopub.status.idle":"2023-06-04T11:53:16.507756Z","shell.execute_reply.started":"2023-06-04T11:53:14.747398Z","shell.execute_reply":"2023-06-04T11:53:16.506237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /tmp/sub\n!cp /kaggle/working/inference_args.json\t /tmp/sub\n!cp /kaggle/working/model.tflite /tmp/sub\n\n!ls /tmp/sub","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:16.509163Z","iopub.execute_input":"2023-06-04T11:53:16.509492Z","iopub.status.idle":"2023-06-04T11:53:17.608928Z","shell.execute_reply.started":"2023-06-04T11:53:16.509461Z","shell.execute_reply":"2023-06-04T11:53:17.606766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make the submission zip \narchn = 'submission'\nshutil.make_archive(archn, 'zip', '/tmp/sub/')","metadata":{"execution":{"iopub.status.busy":"2023-06-04T11:53:17.61107Z","iopub.execute_input":"2023-06-04T11:53:17.611574Z","iopub.status.idle":"2023-06-04T11:53:17.621576Z","shell.execute_reply.started":"2023-06-04T11:53:17.611524Z","shell.execute_reply":"2023-06-04T11:53:17.620737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}