{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52950,"databundleVersionId":5973250,"sourceType":"competition"},{"sourceId":6349328,"sourceType":"datasetVersion","datasetId":3656378},{"sourceId":6349450,"sourceType":"datasetVersion","datasetId":3656454}],"dockerImageVersionId":30527,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.cluster import DBSCAN\nimport numpy as np\nimport json\nimport numpy as np\nimport tensorflow as tf\nfrom minisom import MiniSom\nfrom tqdm import tqdm  \n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nmeta_data\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\npd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet').reset_index().groupby('sequence_id').mean()","metadata":{"execution":{"iopub.status.busy":"2023-08-23T15:21:09.062003Z","iopub.execute_input":"2023-08-23T15:21:09.062469Z","iopub.status.idle":"2023-08-23T15:21:15.219105Z","shell.execute_reply.started":"2023-08-23T15:21:09.062433Z","shell.execute_reply":"2023-08-23T15:21:15.217701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\n\ndirectory = '/kaggle/input/asl-fingerspelling/train_landmarks'\nfor file in os.listdir(directory):\n    path = os.path.join(directory,file)\n    df = pd.read_parquet(path)\n    for uniq in df['sequence_id'].unique():\n        data = df[df['sequence_id']==uniq].values\n        indices =[]\n        for vec in data:\n            indices.append(som.winner(vec))\n        bmu_indices=np.array(indices)\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def adjust_dbscan_parameters(data, epsilon, min_samples, desired_clusters, num_iterations):\n    for iteration in range(num_iterations):\n        dbscan = DBSCAN(eps=epsilon, min_samples=min_samples)\n        cluster_labels = dbscan.fit_predict(data)\n        unique_clusters = len(np.unique(cluster_labels))\n\n        if unique_clusters == desired_clusters:\n            break\n        \n        epsilon_factor = (unique_clusters / desired_clusters) ** 0.5\n        min_samples_factor = (desired_clusters / unique_clusters) ** 0.5\n        \n        epsilon *= epsilon_factor\n        min_samples = max(2, int(min_samples * min_samples_factor))\n        \n    return epsilon, min_samples\n\nmeta_data = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nwith open('/kaggle/input/asl-fingerspelling/character_to_prediction_index.json', \"r\") as json_file:\n    data = json.load(json_file)\nchars = [i for i in data.keys()]\n\nmeta_data['num_clusters'] = [len([j for j in i if j in chars]) for i in  meta_data['phrase']]\nmeta_data= meta_data[['sequence_id','phrase','num_clusters']]\n\nnum_iterations = 10\nepsilon = 1\nmin_samples =5\ndesired_clusters = 58\n\nbmu = \nepsilon, min_samples = adjust_dbscan_parameters(matrix_2d, epsilon, min_samples, desired_clusters, num_iterations)\n\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bmu=[]\ni=1\nfor vec in data_matrix:\n    print(i,end='\\r')\n    i+=1\n    bmu.append(som.winner(vec))\nnp.save('bmu.npy',np.array(bmu))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Install and Import Libraries","metadata":{}},{"cell_type":"code","source":"pip install minisom","metadata":{"execution":{"iopub.status.busy":"2023-08-24T06:44:18.383662Z","iopub.execute_input":"2023-08-24T06:44:18.384027Z","iopub.status.idle":"2023-08-24T06:44:33.127985Z","shell.execute_reply.started":"2023-08-24T06:44:18.383999Z","shell.execute_reply":"2023-08-24T06:44:33.126792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pafmd\nfrom sklearn.cluster import DBSCAN\nimport numpy as np\nimport json\nimport pandas as pd\nimport tensorflow as tf\nfrom minisom import MiniSom\nfrom tqdm import tqdm  \nimport os\nimport matplotlib.pyplot as plt\n","metadata":{"execution":{"iopub.status.busy":"2023-08-24T06:44:33.130666Z","iopub.execute_input":"2023-08-24T06:44:33.131088Z","iopub.status.idle":"2023-08-24T06:44:41.929692Z","shell.execute_reply.started":"2023-08-24T06:44:33.131054Z","shell.execute_reply":"2023-08-24T06:44:41.928607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get Data Matrix","metadata":{}},{"cell_type":"code","source":"\ncol = [\n    'x_right_hand_0', 'x_right_hand_1', 'x_right_hand_2', 'x_right_hand_3', 'x_right_hand_4',\n    'x_right_hand_5', 'x_right_hand_6', 'x_right_hand_7', 'x_right_hand_8', 'x_right_hand_9',\n    'x_right_hand_10', 'x_right_hand_11', 'x_right_hand_12', 'x_right_hand_13', 'x_right_hand_14',\n    'x_right_hand_15', 'x_right_hand_16', 'x_right_hand_17', 'x_right_hand_18', 'x_right_hand_19',\n    'x_right_hand_20',\n    'x_left_hand_0', 'x_left_hand_1', 'x_left_hand_2', 'x_left_hand_3', 'x_left_hand_4',\n    'x_left_hand_5', 'x_left_hand_6', 'x_left_hand_7', 'x_left_hand_8', 'x_left_hand_9',\n    'x_left_hand_10', 'x_left_hand_11', 'x_left_hand_12', 'x_left_hand_13', 'x_left_hand_14',\n    'x_left_hand_15', 'x_left_hand_16', 'x_left_hand_17', 'x_left_hand_18', 'x_left_hand_19',\n    'x_left_hand_20',\n    'y_right_hand_0', 'y_right_hand_1', 'y_right_hand_2', 'y_right_hand_3', 'y_right_hand_4',\n    'y_right_hand_5', 'y_right_hand_6', 'y_right_hand_7', 'y_right_hand_8', 'y_right_hand_9',\n    'y_right_hand_10', 'y_right_hand_11', 'y_right_hand_12', 'y_right_hand_13', 'y_right_hand_14',\n    'y_right_hand_15', 'y_right_hand_16', 'y_right_hand_17', 'y_right_hand_18', 'y_right_hand_19',\n    'y_right_hand_20',\n    'y_left_hand_0', 'y_left_hand_1', 'y_left_hand_2', 'y_left_hand_3', 'y_left_hand_4',\n    'y_left_hand_5', 'y_left_hand_6', 'y_left_hand_7', 'y_left_hand_8', 'y_left_hand_9',\n    'y_left_hand_10', 'y_left_hand_11', 'y_left_hand_12', 'y_left_hand_13', 'y_left_hand_14',\n    'y_left_hand_15', 'y_left_hand_16', 'y_left_hand_17', 'y_left_hand_18', 'y_left_hand_19',\n    'y_left_hand_20',\n    'z_right_hand_0', 'z_right_hand_1', 'z_right_hand_2', 'z_right_hand_3', 'z_right_hand_4',\n    'z_right_hand_5', 'z_right_hand_6', 'z_right_hand_7', 'z_right_hand_8', 'z_right_hand_9',\n    'z_right_hand_10', 'z_right_hand_11', 'z_right_hand_12', 'z_right_hand_13', 'z_right_hand_14',\n    'z_right_hand_15', 'z_right_hand_16', 'z_right_hand_17', 'z_right_hand_18', 'z_right_hand_19',\n    'z_right_hand_20',\n    'z_left_hand_0', 'z_left_hand_1', 'z_left_hand_2', 'z_left_hand_3', 'z_left_hand_4',\n    'z_left_hand_5', 'z_left_hand_6', 'z_left_hand_7', 'z_left_hand_8', 'z_left_hand_9',\n    'z_left_hand_10', 'z_left_hand_11', 'z_left_hand_12', 'z_left_hand_13', 'z_left_hand_14',\n    'z_left_hand_15', 'z_left_hand_16', 'z_left_hand_17', 'z_left_hand_18', 'z_left_hand_19',\n    'z_left_hand_20'\n]\ndata_matrix =[]\ni =1\ndirectory = '/kaggle/input/asl-fingerspelling/train_landmarks'\nfor file in os.listdir(directory):\n    print(i,end='\\r')\n    i+=1\n    \n    path = os.path.join(directory, file)\n    data = pd.read_parquet(path)[col].dropna(how='all').reset_index()\n    lookup = data.groupby('sequence_id').mean()\n    \n    j=1\n    for seq_id in data['sequence_id'].unique():\n        \n        for column in data.drop('sequence_id', axis=1).columns:\n            fill_value = lookup.loc[seq_id, column]\n            data.loc[data['sequence_id'] == seq_id, column] = data.loc[data['sequence_id'] == seq_id, column].fillna(fill_value)\n    data_matrix.append(data.drop('sequence_id', axis=1).values)\n    \n\ndata_matrix = np.vstack(data_matrix)\nprint(data_matrix.shape)\nprint(dbscan_atrix.shape)\n\nnp.save('data_matrix.npy',data_matrix)\nnp.save('dbscan_matrix.npy',dbscan_matrix)","metadata":{"execution":{"iopub.status.busy":"2023-08-24T06:44:41.931372Z","iopub.execute_input":"2023-08-24T06:44:41.932134Z","iopub.status.idle":"2023-08-24T10:11:13.748149Z","shell.execute_reply.started":"2023-08-24T06:44:41.932099Z","shell.execute_reply":"2023-08-24T10:11:13.746661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data_matrix.shape)\n\nnp.save('data_matrix.npy',data_matrix)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-24T10:17:09.843457Z","iopub.execute_input":"2023-08-24T10:17:09.843839Z","iopub.status.idle":"2023-08-24T10:17:19.329123Z","shell.execute_reply.started":"2023-08-24T10:17:09.843803Z","shell.execute_reply":"2023-08-24T10:17:19.32787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train SOM","metadata":{}},{"cell_type":"code","source":"data_matrix[:10]","metadata":{"execution":{"iopub.status.busy":"2023-08-24T10:38:34.436387Z","iopub.execute_input":"2023-08-24T10:38:34.436751Z","iopub.status.idle":"2023-08-24T10:38:34.448039Z","shell.execute_reply.started":"2023-08-24T10:38:34.436721Z","shell.execute_reply":"2023-08-24T10:38:34.447006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ninput_size = len(col)\nmap_size = (150, 150)  \nlearning_rate = 0.3\nsigma = 7\nprint('initialisation complete')\n\nsom = MiniSom(map_size[0], map_size[1], input_size, sigma=sigma, learning_rate=learning_rate)\ndata_matrix = np.nan_to_num(data_matrix)\nprint('Training started')\nnum_epochs = 100  \nwith tqdm(total=num_epochs) as pbar:  \n    for epoch in range(num_epochs):\n        som.train(data_matrix, 100) \n        pbar.update(1)  \n\nbmu=[]\ni=1\nfor vec in data_matrix[:10000]:\n    print(i,end='\\r')\n    i+=1\n    bmu.append(som.winner(vec))\n    \n\nplt.figure(figsize=(10, 10))\nplt.pcolor(som.distance_map().T, cmap='bone_r')\n\n# Mark the BMUs with red circles\nbmu_y, bmu_x = zip(*bmu)  # Assuming 'bmu' is a list of (y, x) BMU indices\nplt.scatter(bmu_x, bmu_y, marker='o', s=50, c='red', label='BMU')\n\nplt.colorbar()\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-24T10:44:37.60054Z","iopub.execute_input":"2023-08-24T10:44:37.601007Z","iopub.status.idle":"2023-08-24T10:44:38.051329Z","shell.execute_reply.started":"2023-08-24T10:44:37.600967Z","shell.execute_reply":"2023-08-24T10:44:38.04984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find BMU","metadata":{}},{"cell_type":"code","source":"\ndata_ds = tf.data.Dataset.from_tensor_slices(data_matrix[:100])\ntpu_strategy = tf.distribute.TPUStrategy(tf.distribute.cluster_resolver.TPUClusterResolver())\n\n@tf.function\ndef find_bmu(input_vector, som):\n    return som.winner(input_vector)\n\nbmu_indices = []\nwith tpu_strategy.scope():\n    for input_vector in tqdm(data_ds, total=len(data_matrix)):  # Use tqdm to show progress\n        bmu_indices.append(find_bmu(input_vector, som))\n\nbmu_indices = np.array(bmu_indices)\n\nprint(\"BMU indices shape:\", bmu_indices.shape)\nnp.save('bmu.npy',np.array(bmu))","metadata":{"execution":{"iopub.status.busy":"2023-08-24T10:29:26.645787Z","iopub.execute_input":"2023-08-24T10:29:26.646803Z","iopub.status.idle":"2023-08-24T10:29:30.240152Z","shell.execute_reply.started":"2023-08-24T10:29:26.646746Z","shell.execute_reply":"2023-08-24T10:29:30.23893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize the SOM","metadata":{}},{"cell_type":"code","source":"\nplt.figure(figsize=(10, 10))\nplt.pcolor(som.distance_map().T, cmap='bone_r')\n\n# Mark the BMUs with red circles\nbmu_y, bmu_x = zip(*bmu)  # Assuming 'bmu' is a list of (y, x) BMU indices\nplt.scatter(bmu_x, bmu_y, marker='o', s=50, c='red', label='BMU')\n\nplt.colorbar()\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-23T14:03:35.645965Z","iopub.execute_input":"2023-08-23T14:03:35.646648Z","iopub.status.idle":"2023-08-23T14:03:37.390472Z","shell.execute_reply.started":"2023-08-23T14:03:35.646598Z","shell.execute_reply":"2023-08-23T14:03:37.389407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# applying Birch","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.cluster import Birch\n\n# Load the best matching units (BMUs) from the numpy file\nbmu = np.load('/kaggle/input/bmu-idx/bmu_indices.npy')\n\n# Reshape the BMUs to match the expected input shape for Birch\nbmu_reshaped = bmu.reshape(-1, 1)\n\n# Initialize the BIRCH clustering algorithm\nbirch = Birch(n_clusters=58)\n\n# Fit the BIRCH algorithm to the reshaped BMUs\nbirch.fit(bmu_reshaped)\n\n# Get the predicted cluster labels for each BMU\ncluster_labels = birch.predict(bmu_reshaped)\n\n# Print the number of data points in each cluster\nunique_clusters, counts = np.unique(cluster_labels, return_counts=True)\nfor cluster, count in zip(unique_clusters, counts):\n    print(f\"Cluster {cluster}: {count} data points\",end='\\r')\n","metadata":{"execution":{"iopub.status.busy":"2023-08-23T13:50:08.675579Z","iopub.execute_input":"2023-08-23T13:50:08.676041Z","iopub.status.idle":"2023-08-23T13:50:10.61184Z","shell.execute_reply.started":"2023-08-23T13:50:08.676003Z","shell.execute_reply":"2023-08-23T13:50:10.609686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making submision.zip file","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}