{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-15T12:23:24.261143Z","iopub.execute_input":"2023-05-15T12:23:24.261588Z","iopub.status.idle":"2023-05-15T12:23:24.277177Z","shell.execute_reply.started":"2023-05-15T12:23:24.261544Z","shell.execute_reply":"2023-05-15T12:23:24.276178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --pre torcharrow -f https://download.pytorch.org/whl/nightly/cpu/torch_nightly.html","metadata":{"execution":{"iopub.status.busy":"2023-05-15T12:18:01.316364Z","iopub.execute_input":"2023-05-15T12:18:01.316855Z","iopub.status.idle":"2023-05-15T12:18:45.522782Z","shell.execute_reply.started":"2023-05-15T12:18:01.316821Z","shell.execute_reply":"2023-05-15T12:18:45.521267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd \nimport torchdata\nimport torcharrow","metadata":{"execution":{"iopub.status.busy":"2023-05-15T12:23:46.752Z","iopub.execute_input":"2023-05-15T12:23:46.752423Z","iopub.status.idle":"2023-05-15T12:23:51.975205Z","shell.execute_reply.started":"2023-05-15T12:23:46.752392Z","shell.execute_reply":"2023-05-15T12:23:51.973847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train data\ntrain = pd.read_csv(\"/kaggle/input/asl-fingerspelling/train.csv\")\ntrain","metadata":{"execution":{"iopub.status.busy":"2023-05-15T12:25:00.631756Z","iopub.execute_input":"2023-05-15T12:25:00.632665Z","iopub.status.idle":"2023-05-15T12:25:00.847066Z","shell.execute_reply.started":"2023-05-15T12:25:00.632619Z","shell.execute_reply":"2023-05-15T12:25:00.845783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"supplemental_metdata = pd.read_csv(\"/kaggle/input/asl-fingerspelling/supplemental_metadata.csv\")\nsupplemental_metdata","metadata":{"execution":{"iopub.status.busy":"2023-05-15T12:26:31.943544Z","iopub.execute_input":"2023-05-15T12:26:31.944017Z","iopub.status.idle":"2023-05-15T12:26:32.083763Z","shell.execute_reply.started":"2023-05-15T12:26:31.943983Z","shell.execute_reply":"2023-05-15T12:26:32.082836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"supplemental_metdata.drop([\"file_id\" , \"sequence_id\" , \"participant_id\"] , axis = 1 , inplace = True)\nsupplemental_metdata","metadata":{"execution":{"iopub.status.busy":"2023-05-15T12:26:46.74956Z","iopub.execute_input":"2023-05-15T12:26:46.750029Z","iopub.status.idle":"2023-05-15T12:26:46.769203Z","shell.execute_reply.started":"2023-05-15T12:26:46.749994Z","shell.execute_reply":"2023-05-15T12:26:46.767655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#this problem is technically to build a model that has video as input and text as output\n#One lead that I found is to use the combination of Convolutional Neural Networks (CNNs) \n# and Recurrent Neural Networks (RNNs), specifically an Encoder-Decoder framework\n\n#The code below is the code on this structure. It won't work, but at least there is\n#something for us to do more research on\nimport tensorflow as tf\nfrom tensorflow.keras.layers import LSTM, Dense, Conv3D, TimeDistributed\nimport pyarrow.parquet as pq\nimport glob\n\n# Define the CNN encoder\ndef create_encoder():\n    model = tf.keras.Sequential()\n    # Add Conv3D layers and other necessary layers\n    # Modify the architecture according to your requirements\n    return model\n\n# Define the RNN decoder\ndef create_decoder(vocab_size):\n    model = tf.keras.Sequential()\n    # Add LSTM layers and other necessary layers\n    # Modify the architecture according to your requirements\n    return model\n\n# Create the video-to-text model\ndef create_video_to_text_model(encoder, decoder, max_caption_length, vocab_size):\n    video_input = tf.keras.Input(shape=(None, height, width, channels))  # Adjust input shape based on your video frames\n    encoded_sequence = encoder(video_input)\n    decoder_output = decoder(encoded_sequence)\n\n    # Define the model architecture\n    model = tf.keras.Model(inputs=video_input, outputs=decoder_output)\n\n    # Compile the model and define the loss function and optimizer\n    model.compile(loss=\"categorical_crossentropy\", optimizer=\"adam\")\n\n    return model\n\n# Create the encoder and decoder models\nencoder_model = create_encoder()\ndecoder_model = create_decoder(vocab_size)\n\n# Create the video-to-text model\nvideo_to_text_model = create_video_to_text_model(encoder_model, decoder_model, max_caption_length, vocab_size)\n\n# Train the model using your video and language data\nvideo_data_parquet_files = glob.glob(\"path/to/video_data_*.parquet\")\n\n# Read video data from multiple .parquet files\nvideo_data = []\nfor file in video_data_parquet_files:\n    video_data_table = pq.read_table(file)\n    video_data.append(video_data_table.to_pandas())\n\nvideo_data_concatenated = pd.concat(video_data, axis=0)\n\ncaption_data = ...\nvideo_to_text_model.fit(video_data_concatenated, caption_data, epochs=num_epochs, batch_size=batch_size)\n\n# Generate text captions for new videos\nnew_video_parquet_files = glob.glob(\"path/to/new_video_*.parquet\")\n\n# Read new video data from multiple .parquet files\nnew_video = []\nfor file in new_video_parquet_files:\n    new_video_table = pq.read_table(file)\n    new_video.append(new_video_table.to_pandas())\n\nnew_video_concatenated = pd.concat(new_video, axis=0)\n\npredicted_caption = video_to_text_model.predict(new_video_concatenated)\n","metadata":{},"execution_count":null,"outputs":[]}]}