{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52950,"databundleVersionId":5973250,"sourceType":"competition"},{"sourceId":9084251,"sourceType":"datasetVersion","datasetId":5481080}],"dockerImageVersionId":30512,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install mediapipe\n","metadata":{"_uuid":"bd1cedae-437a-4757-bb22-2be7ec274fae","_cell_guid":"ac02726f-11e0-4c20-a96d-283404ad0a43","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-01T18:11:32.265557Z","iopub.execute_input":"2024-08-01T18:11:32.266489Z","iopub.status.idle":"2024-08-01T18:11:50.908795Z","shell.execute_reply.started":"2024-08-01T18:11:32.266441Z","shell.execute_reply":"2024-08-01T18:11:50.907715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install numpy pandas pyarrow tensorflow mediapipe matplotlib scikit-image tqdm\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:06:55.796623Z","iopub.execute_input":"2024-08-01T19:06:55.797065Z","iopub.status.idle":"2024-08-01T19:07:18.931352Z","shell.execute_reply.started":"2024-08-01T19:06:55.797032Z","shell.execute_reply":"2024-08-01T19:07:18.930047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport pyarrow.parquet as pq\nimport tensorflow as tf\nimport json\nimport mediapipe\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport random\n\nfrom skimage.transform import resize\nfrom mediapipe.framework.formats import landmark_pb2\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tqdm.notebook import tqdm\nfrom matplotlib import animation, rc","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:07:29.622481Z","iopub.execute_input":"2024-08-01T19:07:29.622896Z","iopub.status.idle":"2024-08-01T19:07:29.630709Z","shell.execute_reply.started":"2024-08-01T19:07:29.622861Z","shell.execute_reply":"2024-08-01T19:07:29.629387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2024-08-01T18:12:29.498296Z","iopub.execute_input":"2024-08-01T18:12:29.499157Z","iopub.status.idle":"2024-08-01T18:12:29.714591Z","shell.execute_reply.started":"2024-08-01T18:12:29.49912Z","shell.execute_reply":"2024-08-01T18:12:29.713417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-01T18:12:32.691174Z","iopub.execute_input":"2024-08-01T18:12:32.692005Z","iopub.status.idle":"2024-08-01T18:12:32.711612Z","shell.execute_reply.started":"2024-08-01T18:12:32.691953Z","shell.execute_reply":"2024-08-01T18:12:32.710417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Đường dẫn tới thư mục đầu vào của dataset\ninput_path = '/kaggle/input/asl-fingerspelling'\n\n# Hiển thị danh sách các tập tin trong thư mục\nfor root, dirs, files in os.walk(input_path):\n    for file in files:\n        print(os.path.join(root, file))\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:00:29.11814Z","iopub.execute_input":"2024-08-01T19:00:29.118561Z","iopub.status.idle":"2024-08-01T19:00:29.16795Z","shell.execute_reply.started":"2024-08-01T19:00:29.118533Z","shell.execute_reply":"2024-08-01T19:00:29.166958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/project-1 /kaggle/working/\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T18:16:35.174835Z","iopub.execute_input":"2024-08-01T18:16:35.175602Z","iopub.status.idle":"2024-08-01T18:16:36.80202Z","shell.execute_reply.started":"2024-08-01T18:16:35.175571Z","shell.execute_reply":"2024-08-01T18:16:36.800551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install neptune==1.0.0\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T18:18:32.328102Z","iopub.execute_input":"2024-08-01T18:18:32.328559Z","iopub.status.idle":"2024-08-01T18:18:54.591747Z","shell.execute_reply.started":"2024-08-01T18:18:32.328522Z","shell.execute_reply":"2024-08-01T18:18:54.590564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ndef list_files_and_dirs(base_path):\n    for root, dirs, files in os.walk(base_path):\n        print(f\"Current directory: {root}\")\n        if dirs:\n            print(\"Directories:\")\n            for dir_name in dirs:\n                print(f\"  - {dir_name}\")\n        else:\n            print(\"No directories found\")\n        if files:\n            print(\"Files:\")\n            for file_name in files:\n                print(f\"  - {file_name}\")\n        else:\n            print(\"No files found\")\n        print(\"-\" * 40)\n\n# Đặt đường dẫn cơ sở của dự án\nbase_path = '/kaggle/working/project-1'\n\n# Gọi hàm để liệt kê các tệp và thư mục\nlist_files_and_dirs(base_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:11:50.658998Z","iopub.execute_input":"2024-08-01T19:11:50.659943Z","iopub.status.idle":"2024-08-01T19:11:50.670886Z","shell.execute_reply.started":"2024-08-01T19:11:50.65991Z","shell.execute_reply":"2024-08-01T19:11:50.669818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working/huyenho712002/Project_asl\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T18:46:18.136281Z","iopub.execute_input":"2024-08-01T18:46:18.13701Z","iopub.status.idle":"2024-08-01T18:46:18.14274Z","shell.execute_reply.started":"2024-08-01T18:46:18.136963Z","shell.execute_reply":"2024-08-01T18:46:18.14174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nimport os\n\n# Xóa toàn bộ thư mục và các tệp con trong thư mục `datamount`\nshutil.rmtree('/kaggle/working/project-1/datamount')\n\n# Tạo lại thư mục `datamount`\nos.makedirs('/kaggle/working/project-1/datamount')\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:02:30.87761Z","iopub.execute_input":"2024-08-01T19:02:30.878061Z","iopub.status.idle":"2024-08-01T19:02:30.888603Z","shell.execute_reply.started":"2024-08-01T19:02:30.878029Z","shell.execute_reply":"2024-08-01T19:02:30.887566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\n# Đường dẫn đến thư mục chứa dataset mới\ninput_dataset_path = '/kaggle/input/asl-fingerspelling/'\n\n# Các tệp cần sao chép\nfiles_to_copy = [\n    'supplemental_metadata.csv',\n    'character_to_prediction_index.json',\n    'train.csv',\n    'supplemental_landmarks/371169664.parquet',\n    # ... (thêm các tệp còn lại vào danh sách)\n]\n\n# Sao chép các tệp từ dataset mới vào thư mục `datamount`\nfor file_name in files_to_copy:\n    src = os.path.join(input_dataset_path, file_name)\n    dst = os.path.join('/kaggle/working/project-1/datamount', file_name)\n    # Tạo thư mục đích nếu chưa tồn tại\n    os.makedirs(os.path.dirname(dst), exist_ok=True)\n    shutil.copy(src, dst)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:02:40.738407Z","iopub.execute_input":"2024-08-01T19:02:40.739277Z","iopub.status.idle":"2024-08-01T19:03:06.850699Z","shell.execute_reply.started":"2024-08-01T19:02:40.739243Z","shell.execute_reply":"2024-08-01T19:03:06.849747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade tensorflow-io\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:08:42.570945Z","iopub.execute_input":"2024-08-01T19:08:42.572117Z","iopub.status.idle":"2024-08-01T19:08:58.233124Z","shell.execute_reply.started":"2024-08-01T19:08:42.572072Z","shell.execute_reply":"2024-08-01T19:08:58.231895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Định nghĩa đường dẫn của tệp\nsource_path = 'datamount/train.csv'\ndestination_path = 'datamount/train_folded.csv'\n\n# Kiểm tra nếu tệp nguồn tồn tại\nif os.path.isfile(source_path):\n    # Đổi tên tệp\n    os.rename(source_path, destination_path)\n    print(f\"Tệp đã được đổi tên thành: {destination_path}\")\nelse:\n    print(f\"Tệp nguồn không tồn tại: {source_path}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:14:39.804946Z","iopub.execute_input":"2024-08-01T19:14:39.80564Z","iopub.status.idle":"2024-08-01T19:14:39.811776Z","shell.execute_reply.started":"2024-08-01T19:14:39.805608Z","shell.execute_reply":"2024-08-01T19:14:39.810821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Đọc tệp CSV\ntrain_csv = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')\nsupplemental_metadata_csv = pd.read_csv('/kaggle/input/asl-fingerspelling/supplemental_metadata.csv')\n\n# Đọc tệp Parquet\nparquet_files = [\n    '/kaggle/input/asl-fingerspelling/supplemental_landmarks/371169664.parquet',\n    '/kaggle/input/asl-fingerspelling/supplemental_landmarks/369584223.parquet',\n    # Thêm các tệp Parquet khác ở đây\n]\nparquet_data = [pd.read_parquet(file) for file in parquet_files]\n\n# Gộp các tệp Parquet lại thành một DataFrame duy nhất\nsupplemental_landmarks = pd.concat(parquet_data, ignore_index=True)\n\n# Hiển thị thông tin của các DataFrame\nprint(train_csv.head())\nprint(supplemental_metadata_csv.head())\nprint(supplemental_landmarks.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:39:54.5584Z","iopub.execute_input":"2024-08-01T19:39:54.559203Z","iopub.status.idle":"2024-08-01T19:40:19.957003Z","shell.execute_reply.started":"2024-08-01T19:39:54.559168Z","shell.execute_reply":"2024-08-01T19:40:19.955889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Đọc dữ liệu từ file CSV\ndf = pd.read_csv(cfg.train_df)\n\n# Thêm cột 'fold' nếu nó không tồn tại\nif 'fold' not in df.columns:\n    # Thêm cột 'fold' với giá trị mặc định là 0 hoặc giá trị khác mà bạn chọn\n    df['fold'] = 0\n    print(\"Cột 'fold' không tồn tại trong dữ liệu. Đã thêm cột 'fold' với giá trị mặc định là 0.\")\n\n# Chia dữ liệu dựa trên cột 'fold'\ntrain_df = df[df[\"fold\"] != cfg.fold].copy()\n\n# Tiếp tục với quá trình huấn luyện\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:26:54.676426Z","iopub.execute_input":"2024-08-01T19:26:54.677252Z","iopub.status.idle":"2024-08-01T19:26:54.78573Z","shell.execute_reply.started":"2024-08-01T19:26:54.677219Z","shell.execute_reply":"2024-08-01T19:26:54.784756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.system('python train.py --config cfg_1 --gpu_id 0 --epochs 1')\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:27:10.368797Z","iopub.execute_input":"2024-08-01T19:27:10.36918Z","iopub.status.idle":"2024-08-01T19:27:23.928647Z","shell.execute_reply.started":"2024-08-01T19:27:10.369149Z","shell.execute_reply":"2024-08-01T19:27:23.927513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nimport os\nimport importlib\nimport copy\n\n# Thay đổi thư mục làm việc hiện tại\nos.chdir('/kaggle/working/project-1')\n\n# Thêm thư mục configs vào sys.path\nsys.path.append('./configs')\n\n# In ra đường dẫn để kiểm tra\nprint(\"Current sys.path:\", sys.path)\n\n# Kiểm tra các tệp trong thư mục configs\nprint(\"Files in configs directory:\")\nfor root, dirs, files in os.walk('./configs'):\n    for file in files:\n        print(os.path.join(root, file))\n\n# Kiểm tra import\ntry:\n    # Nhập mô-đun cấu hình\n    cfg_module = importlib.import_module('cfg_1')\n    cfg = copy.copy(cfg_module.cfg)\n    print(\"Module imported successfully\")\nexcept ModuleNotFoundError as e:\n    print(f\"Module not found: {e}\")\nexcept ImportError as e:\n    print(f\"Import error: {e}\")\n\n# Chạy script với tham số cấu hình\nos.system('python train.py --config cfg_1 --gpu_id 0 --epochs 1')\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T19:18:34.398031Z","iopub.execute_input":"2024-08-01T19:18:34.398406Z","iopub.status.idle":"2024-08-01T19:18:47.837467Z","shell.execute_reply.started":"2024-08-01T19:18:34.398379Z","shell.execute_reply":"2024-08-01T19:18:47.836439Z"},"trusted":true},"execution_count":null,"outputs":[]}]}