{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":52950,"databundleVersionId":5973250,"sourceType":"competition"}],"dockerImageVersionId":30527,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfor i in pd.read_parquet('/kaggle/input/asl-fingerspelling/supplemental_landmarks/1032110484.parquet').columns:\n    print(i)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-11T08:25:04.704618Z","iopub.execute_input":"2023-08-11T08:25:04.705097Z","iopub.status.idle":"2023-08-11T08:25:07.749444Z","shell.execute_reply.started":"2023-08-11T08:25:04.705061Z","shell.execute_reply":"2023-08-11T08:25:07.748012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet').reset_index()\ndf","metadata":{"execution":{"iopub.status.busy":"2023-08-15T14:02:35.890355Z","iopub.execute_input":"2023-08-15T14:02:35.890714Z","iopub.status.idle":"2023-08-15T14:02:53.18028Z","shell.execute_reply.started":"2023-08-15T14:02:35.89069Z","shell.execute_reply":"2023-08-15T14:02:53.178981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['x_right_hand_0',\n'x_right_hand_1',\n'x_right_hand_2',\n'x_right_hand_3',\n'x_right_hand_4',\n'x_right_hand_5',\n'x_right_hand_6',\n'x_right_hand_7',\n'x_right_hand_8',\n'x_right_hand_9',\n'x_right_hand_10',\n'x_right_hand_11',\n'x_right_hand_12',\n'x_right_hand_13',\n'x_right_hand_14',\n'x_right_hand_15',\n'x_right_hand_16',\n'x_right_hand_17',\n'x_right_hand_18',\n'x_right_hand_19',\n'x_right_hand_20']]","metadata":{"execution":{"iopub.status.busy":"2023-08-15T14:07:41.657596Z","iopub.execute_input":"2023-08-15T14:07:41.658006Z","iopub.status.idle":"2023-08-15T14:07:41.703035Z","shell.execute_reply.started":"2023-08-15T14:07:41.657972Z","shell.execute_reply":"2023-08-15T14:07:41.701533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nwith open('/kaggle/input/asl-fingerspelling/character_to_prediction_index.json', 'r') as json_file:\n    data = json.load(json_file)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-11T07:31:41.881844Z","iopub.execute_input":"2023-08-11T07:31:41.882281Z","iopub.status.idle":"2023-08-11T07:31:41.889936Z","shell.execute_reply.started":"2023-08-11T07:31:41.882246Z","shell.execute_reply":"2023-08-11T07:31:41.888915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-08-11T07:35:03.887614Z","iopub.execute_input":"2023-08-11T07:35:03.888089Z","iopub.status.idle":"2023-08-11T07:35:03.897598Z","shell.execute_reply.started":"2023-08-11T07:35:03.888045Z","shell.execute_reply":"2023-08-11T07:35:03.89621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/asl-fingerspelling/supplemental_metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-11T07:49:15.975464Z","iopub.execute_input":"2023-08-11T07:49:15.976151Z","iopub.status.idle":"2023-08-11T07:49:16.152029Z","shell.execute_reply.started":"2023-08-11T07:49:15.976098Z","shell.execute_reply":"2023-08-11T07:49:16.151116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-21T12:55:53.495169Z","iopub.execute_input":"2023-08-21T12:55:53.495591Z","iopub.status.idle":"2023-08-21T12:55:53.681004Z","shell.execute_reply.started":"2023-08-21T12:55:53.49556Z","shell.execute_reply":"2023-08-21T12:55:53.679504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-08-21T13:01:56.371135Z","iopub.execute_input":"2023-08-21T13:01:56.371624Z","iopub.status.idle":"2023-08-21T13:02:00.717177Z","shell.execute_reply.started":"2023-08-21T13:01:56.371585Z","shell.execute_reply":"2023-08-21T13:02:00.715949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom minisom import MiniSom\nfrom sklearn.cluster import DBSCAN\nfrom tqdm import tqdm\n\ncol = [\n    'x_right_hand_0', 'x_right_hand_1', 'x_right_hand_2', 'x_right_hand_3', 'x_right_hand_4',\n    'x_right_hand_5', 'x_right_hand_6', 'x_right_hand_7', 'x_right_hand_8', 'x_right_hand_9',\n    'x_right_hand_10', 'x_right_hand_11', 'x_right_hand_12', 'x_right_hand_13', 'x_right_hand_14',\n    'x_right_hand_15', 'x_right_hand_16', 'x_right_hand_17', 'x_right_hand_18', 'x_right_hand_19',\n    'x_right_hand_20',\n    'x_left_hand_0', 'x_left_hand_1', 'x_left_hand_2', 'x_left_hand_3', 'x_left_hand_4',\n    'x_left_hand_5', 'x_left_hand_6', 'x_left_hand_7', 'x_left_hand_8', 'x_left_hand_9',\n    'x_left_hand_10', 'x_left_hand_11', 'x_left_hand_12', 'x_left_hand_13', 'x_left_hand_14',\n    'x_left_hand_15', 'x_left_hand_16', 'x_left_hand_17', 'x_left_hand_18', 'x_left_hand_19',\n    'x_left_hand_20',\n    'y_right_hand_0', 'y_right_hand_1', 'y_right_hand_2', 'y_right_hand_3', 'y_right_hand_4',\n    'y_right_hand_5', 'y_right_hand_6', 'y_right_hand_7', 'y_right_hand_8', 'y_right_hand_9',\n    'y_right_hand_10', 'y_right_hand_11', 'y_right_hand_12', 'y_right_hand_13', 'y_right_hand_14',\n    'y_right_hand_15', 'y_right_hand_16', 'y_right_hand_17', 'y_right_hand_18', 'y_right_hand_19',\n    'y_right_hand_20',\n    'y_left_hand_0', 'y_left_hand_1', 'y_left_hand_2', 'y_left_hand_3', 'y_left_hand_4',\n    'y_left_hand_5', 'y_left_hand_6', 'y_left_hand_7', 'y_left_hand_8', 'y_left_hand_9',\n    'y_left_hand_10', 'y_left_hand_11', 'y_left_hand_12', 'y_left_hand_13', 'y_left_hand_14',\n    'y_left_hand_15', 'y_left_hand_16', 'y_left_hand_17', 'y_left_hand_18', 'y_left_hand_19',\n    'y_left_hand_20',\n    'z_right_hand_0', 'z_right_hand_1', 'z_right_hand_2', 'z_right_hand_3', 'z_right_hand_4',\n    'z_right_hand_5', 'z_right_hand_6', 'z_right_hand_7', 'z_right_hand_8', 'z_right_hand_9',\n    'z_right_hand_10', 'z_right_hand_11', 'z_right_hand_12', 'z_right_hand_13', 'z_right_hand_14',\n    'z_right_hand_15', 'z_right_hand_16', 'z_right_hand_17', 'z_right_hand_18', 'z_right_hand_19',\n    'z_right_hand_20',\n    'z_left_hand_0', 'z_left_hand_1', 'z_left_hand_2', 'z_left_hand_3', 'z_left_hand_4',\n    'z_left_hand_5', 'z_left_hand_6', 'z_left_hand_7', 'z_left_hand_8', 'z_left_hand_9',\n    'z_left_hand_10', 'z_left_hand_11', 'z_left_hand_12', 'z_left_hand_13', 'z_left_hand_14',\n    'z_left_hand_15', 'z_left_hand_16', 'z_left_hand_17', 'z_left_hand_18', 'z_left_hand_19',\n    'z_left_hand_20'\n]\n\ndata = []\n\ni = 1\ntot_len = 0\ndirectory = '/kaggle/input/asl-fingerspelling/train_landmarks'\nfor file in os.listdir(directory):\n    print(i, end='\\r')\n    path = os.path.join(directory, file)\n    df = pd.read_parquet(path)[col].dropna(how='all').fillna(0).values\n    tot_len += df.shape[0]\n    \n    # Extend the data list with the current dataframe\n    data.extend(df)\n    \n    i += 1\n\n# Convert the data list to a Numpy matrix\ndata_matrix = np.array(data)\nnp.save('data_matrix.npy', data_matrix)\ndata_matrix.shape\n\n\n\n\n\nmeta = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')[['path','sequence_id','phrase']]\nlength = len(meta['sequence_id'].unique())\ni = 1\nfor paths in meta['path'].unique():\n    path =os.join('/kaggle/input/asl-fingerspelling',paths)\n    file = pd.read_parquet(path)[col].reset_index()\n    display(file)\n    filt_meta  = meta[meta['path']==paths]\n    for ids in filt_meta['sequence_id']:\n        matrix = file[file['sequence_id']==ids].values\n        \n          \n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-15T09:52:06.280574Z","iopub.execute_input":"2023-08-15T09:52:06.280959Z","iopub.status.idle":"2023-08-15T09:52:06.350081Z","shell.execute_reply.started":"2023-08-15T09:52:06.28093Z","shell.execute_reply":"2023-08-15T09:52:06.349027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DBSCAN Clustering","metadata":{}},{"cell_type":"code","source":"pip install sompy","metadata":{"execution":{"iopub.status.busy":"2023-08-16T18:18:16.473256Z","iopub.execute_input":"2023-08-16T18:18:16.473628Z","iopub.status.idle":"2023-08-16T18:18:34.214161Z","shell.execute_reply.started":"2023-08-16T18:18:16.4736Z","shell.execute_reply":"2023-08-16T18:18:34.212673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-08-22T02:50:29.268749Z","iopub.execute_input":"2023-08-22T02:50:29.269267Z","iopub.status.idle":"2023-08-22T02:50:29.317797Z","shell.execute_reply.started":"2023-08-22T02:50:29.269229Z","shell.execute_reply":"2023-08-22T02:50:29.316396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom sompy import SOMFactory\nfrom sklearn.cluster import DBSCAN\n\n\n\n# Define the parameters for the MiniSom\ninput_size = len(col)\nmap_size = (10, 10)  # Adjust based on your preferences\nlearning_rate = 0.5\nsigma = 1.0\n\nsom = SOMFactory.build(input_size, map_size, initialization='pca', component_names=col)\n\ni=1\ntot_len=0\ndirectory = '/kaggle/input/asl-fingerspelling/train_landmarks'\nfor file in os.listdir(directory):\n    print(i,end='\\r')\n    path = os.path.join(directory,file)\n    df = pd.read_parquet(path)[col].dropna(how='all').fillna(0).values\n    tot_len+=df.shape[0]\n    som.train(data=df)\n    i += 1\n            \nbmu_indices = np.array([som.winner(input_vector) for input_vector in df])\ndb = DBSCAN(eps=0.5, min_samples=5).fit(bmu_indices)\ncluster_labels = db.labels_\nprint(len(cluster_labels),tot_len)\n    ","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-08-21T05:58:00.425541Z","iopub.execute_input":"2023-08-21T05:58:00.42597Z","iopub.status.idle":"2023-08-21T05:58:00.479633Z","shell.execute_reply.started":"2023-08-21T05:58:00.425939Z","shell.execute_reply":"2023-08-21T05:58:00.477899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom minisom import MiniSom\nfrom sklearn.cluster import DBSCAN\nfrom tqdm import tqdm  \n\ncol = [\n    'x_right_hand_0', 'x_right_hand_1', 'x_right_hand_2', 'x_right_hand_3', 'x_right_hand_4',\n    'x_right_hand_5', 'x_right_hand_6', 'x_right_hand_7', 'x_right_hand_8', 'x_right_hand_9',\n    'x_right_hand_10', 'x_right_hand_11', 'x_right_hand_12', 'x_right_hand_13', 'x_right_hand_14',\n    'x_right_hand_15', 'x_right_hand_16', 'x_right_hand_17', 'x_right_hand_18', 'x_right_hand_19',\n    'x_right_hand_20',\n    'x_left_hand_0', 'x_left_hand_1', 'x_left_hand_2', 'x_left_hand_3', 'x_left_hand_4',\n    'x_left_hand_5', 'x_left_hand_6', 'x_left_hand_7', 'x_left_hand_8', 'x_left_hand_9',\n    'x_left_hand_10', 'x_left_hand_11', 'x_left_hand_12', 'x_left_hand_13', 'x_left_hand_14',\n    'x_left_hand_15', 'x_left_hand_16', 'x_left_hand_17', 'x_left_hand_18', 'x_left_hand_19',\n    'x_left_hand_20',\n    'y_right_hand_0', 'y_right_hand_1', 'y_right_hand_2', 'y_right_hand_3', 'y_right_hand_4',\n    'y_right_hand_5', 'y_right_hand_6', 'y_right_hand_7', 'y_right_hand_8', 'y_right_hand_9',\n    'y_right_hand_10', 'y_right_hand_11', 'y_right_hand_12', 'y_right_hand_13', 'y_right_hand_14',\n    'y_right_hand_15', 'y_right_hand_16', 'y_right_hand_17', 'y_right_hand_18', 'y_right_hand_19',\n    'y_right_hand_20',\n    'y_left_hand_0', 'y_left_hand_1', 'y_left_hand_2', 'y_left_hand_3', 'y_left_hand_4',\n    'y_left_hand_5', 'y_left_hand_6', 'y_left_hand_7', 'y_left_hand_8', 'y_left_hand_9',\n    'y_left_hand_10', 'y_left_hand_11', 'y_left_hand_12', 'y_left_hand_13', 'y_left_hand_14',\n    'y_left_hand_15', 'y_left_hand_16', 'y_left_hand_17', 'y_left_hand_18', 'y_left_hand_19',\n    'y_left_hand_20',\n    'z_right_hand_0', 'z_right_hand_1', 'z_right_hand_2', 'z_right_hand_3', 'z_right_hand_4',\n    'z_right_hand_5', 'z_right_hand_6', 'z_right_hand_7', 'z_right_hand_8', 'z_right_hand_9',\n    'z_right_hand_10', 'z_right_hand_11', 'z_right_hand_12', 'z_right_hand_13', 'z_right_hand_14',\n    'z_right_hand_15', 'z_right_hand_16', 'z_right_hand_17', 'z_right_hand_18', 'z_right_hand_19',\n    'z_right_hand_20',\n    'z_left_hand_0', 'z_left_hand_1', 'z_left_hand_2', 'z_left_hand_3', 'z_left_hand_4',\n    'z_left_hand_5', 'z_left_hand_6', 'z_left_hand_7', 'z_left_hand_8', 'z_left_hand_9',\n    'z_left_hand_10', 'z_left_hand_11', 'z_left_hand_12', 'z_left_hand_13', 'z_left_hand_14',\n    'z_left_hand_15', 'z_left_hand_16', 'z_left_hand_17', 'z_left_hand_18', 'z_left_hand_19',\n    'z_left_hand_20'\n]\n\n\ndata = []\n\ni = 1\ntot_len = 0\ndirectory = '/kaggle/input/asl-fingerspelling/train_landmarks'\nfor file in os.listdir(directory):\n    print(i, end='\\r')\n    path = os.path.join(directory, file)\n    df = pd.read_parquet(path)[col].dropna(how='all').fillna(0).values\n    tot_len += df.shape[0]\n    \n    data.extend(df)\n    \n    i += 1\n\ndata_matrix = np.array(data)\nnp.save('data_matrix.npy', data_matrix)\nprint(data_matrix.shape)\n\ninput_size = len(col)\nmap_size = (100, 100) \nlearning_rate = 0.5\nsigma = 1.0\nprint('initialisation complete')\n\nsom = MiniSom(map_size[0], map_size[1], input_size, sigma=sigma, learning_rate=learning_rate)\n\nprint('Training started')\nnum_epochs = 100  \nwith tqdm(total=num_epochs) as pbar:  \n    for epoch in range(num_epochs):\n        som.train(data_matrix, 1)  \n        pbar.update(1) \n        \n\n            ","metadata":{"execution":{"iopub.status.busy":"2023-08-22T11:17:07.078081Z","iopub.execute_input":"2023-08-22T11:17:07.078533Z","iopub.status.idle":"2023-08-22T11:38:04.65231Z","shell.execute_reply.started":"2023-08-22T11:17:07.078499Z","shell.execute_reply":"2023-08-22T11:38:04.649958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install minisom","metadata":{"execution":{"iopub.status.busy":"2023-08-22T11:16:45.573056Z","iopub.execute_input":"2023-08-22T11:16:45.573501Z","iopub.status.idle":"2023-08-22T11:17:03.1142Z","shell.execute_reply.started":"2023-08-22T11:16:45.573451Z","shell.execute_reply":"2023-08-22T11:17:03.113001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from minisom import MiniSom\nfrom sklearn.cluster import DBSCAN\nfrom tqdm import tqdm  # Import tqdm\nprint('libraries imported')\n# Define the parameters for the MiniSom\ninput_size = len(col)\nmap_size = (100, 100)  # Adjust based on your preferences\nlearning_rate = 0.5\nsigma = 10\nprint('initialisation complete')\n\n# Create the SOM instance\nsom = MiniSom(map_size[0], map_size[1], input_size, sigma=sigma, learning_rate=learning_rate)\n\nprint('Training started')\nnum_epochs = 100  # Number of training epochs\nwith tqdm(total=num_epochs) as pbar:  # Initialize tqdm\n    for epoch in range(num_epochs):\n        # Your training code here\n        som.train(data_matrix, 1)  # Train for one iteration\n        pbar.update(1)  # Update tqdm progress bar\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T03:12:18.438119Z","iopub.execute_input":"2023-08-22T03:12:18.438668Z","iopub.status.idle":"2023-08-22T03:12:22.045468Z","shell.execute_reply.started":"2023-08-22T03:12:18.438614Z","shell.execute_reply":"2023-08-22T03:12:22.044351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport multiprocessing\n\n# Function to calculate BMU indices for a given input vector\ndef calculate_bmu(input_vector):\n    return som.winner(input_vector)\n\n# Define the number of parallel processes\nnum_processes = multiprocessing.cpu_count()\n\n# Create a ThreadPool for parallel execution\npool = multiprocessing.Pool(num_processes)\n\nindices = []\nwith tqdm(total=len(data_matrix)) as pbar:\n    for bmu_index in pool.imap(calculate_bmu, data_matrix):\n        indices.append(bmu_index)\n        pbar.update(1)\n\n# Close the pool of processes\npool.close()\npool.join()\n\nbmu_indices = np.array(indices)\nprint(bmu_indices.shape)\nnp.save('bmu_indices.npy', bmu_indices)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-22T11:47:30.901119Z","iopub.execute_input":"2023-08-22T11:47:30.9034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nindx = []\nfor vec in data_matrix:\n    indx.append(som.winner(vec))\n    ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Clustering started')\nwith tqdm(total=len(data_matrix)) as pbar:  # Initialize tqdm\n    db = DBSCAN(eps=0.5, min_samples=5).fit(bmu_indices)\n    cluster_labels = db.labels_\n    pbar.update(len(data_matrix))  # Update tqdm progress bar to indicate completion\n\nprint('Clustering complete')\n\nprint(len(cluster_labels), tot_len)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from minisom import MiniSom\nfrom sklearn.cluster import DBSCAN\n\n# Define the parameters for the MiniSom\ninput_size = len(col)\nmap_size = (100, 100)  # Adjust based on your preferences\nlearning_rate = 0.5\nsigma = 1.0\nprint('initialisation complete')\n# Create the SOM instance\nsom = MiniSom(map_size[0], map_size[1], input_size, sigma=sigma, learning_rate=learning_rate)\n\nprint('training started')\nsom.train(data_matrix, 100)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-20T14:01:39.284099Z","iopub.execute_input":"2023-08-20T14:01:39.284749Z","iopub.status.idle":"2023-08-20T14:01:39.363217Z","shell.execute_reply.started":"2023-08-20T14:01:39.284674Z","shell.execute_reply":"2023-08-20T14:01:39.360943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')[['path','sequence_id','phrase']]\nlength = len(meta['sequence_id'].unique())\ni = 1\nfor paths in meta['path'].unique():\n    path =os.join('/kaggle/input/asl-fingerspelling',paths)\n    file = pd.read_parquet(path)[col].reset_index()\n    display(file)\n    filt_meta  = meta[meta['path']==paths]\n    for ids in filt_meta['sequence_id']:\n        matrix = file[file['sequence_id']==ids].values\n        indices = []\n        for vec in matrix:\n             indices.append(som.winner(vec))","metadata":{},"execution_count":null,"outputs":[]}]}