{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":107710,"databundleVersionId":13245580,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ICCV BinEgo-360 baseline classifier\n\nThis is the basic starter code for the competition. The code can be run directly on kaggle, remember to open GPU (default P100) when you start the session. Despite simply running this code may **not** reach a good performance on the leaderboard, it may give you a quick go through of how to get the training data, and how the inference works.","metadata":{}},{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import os, json, numpy as np\nfrom pathlib import Path\nfrom typing import List, Dict, Any\n\nimport torch, torch.nn as nn, torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader, random_split\nfrom torchvision.io import read_video\nfrom torchvision.transforms.functional import resize\nfrom torchvision.transforms._transforms_video import NormalizeVideo, RandomHorizontalFlipVideo\n\nfrom tqdm import tqdm\nfrom huggingface_hub import snapshot_download\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-03T02:43:55.162273Z","iopub.status.idle":"2025-08-03T02:43:55.162556Z","shell.execute_reply.started":"2025-08-03T02:43:55.16241Z","shell.execute_reply":"2025-08-03T02:43:55.162423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --upgrade decord\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T04:23:27.859599Z","iopub.execute_input":"2025-08-01T04:23:27.85995Z","iopub.status.idle":"2025-08-01T04:23:32.83227Z","shell.execute_reply.started":"2025-08-01T04:23:27.859924Z","shell.execute_reply":"2025-08-01T04:23:32.83133Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prepare the training set\n\nNote that the training set need to be accessed elsewhere for this dataset. \nThe X360 training set is from https://huggingface.co/datasets/quchenyuan/360x_dataset_LR in this starter code.\nYou may use other available variants:\n - https://huggingface.co/datasets/quchenyuan/360x_dataset_HR\n - https://huggingface.co/datasets/quchenyuan/360x_dataset_features\n\n**Remember to change your HF token before running!**","metadata":{}},{"cell_type":"code","source":"\n# ⬇️ Download dataset if not cached\nroot = Path(\"./360x_dataset\")\nif not root.exists():\n    # --------------------------------------------------- #\n    # set API TOKEN to yours, \n    # check https://huggingface.co/settings/tokens for instruction.\n    HF_TOKEN = None\n    # --------------------------------------------------- #\n    assert HF_TOKEN, \"Please export HF_TOKEN env variable with your Hugging Face token\"\n    snapshot_download(repo_id=\"quchenyuan/360x_dataset\",\n                      repo_type=\"dataset\",\n                      token=HF_TOKEN,\n                      local_dir=str(root),\n                      local_dir_use_symlinks=False)\n\nindex_path = root / \"index.json\"\nassert index_path.exists(), \"index.json missing — dataset download might have failed\"\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-01T04:23:32.83341Z","iopub.execute_input":"2025-08-01T04:23:32.833705Z","iopub.status.idle":"2025-08-01T04:27:24.400302Z","shell.execute_reply.started":"2025-08-01T04:23:32.833672Z","shell.execute_reply":"2025-08-01T04:27:24.399587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# 🔖 Load the official class mapping\nmapping_path = Path(\"/kaggle/input/bin-ego-360-challenge-classification-ext/cls_mapping.npy\")\ncls_map = np.load(mapping_path, allow_pickle=True).item()\nnum_classes = cls_map[\"counter\"]\nlabel2idx = {k: v for k, v in cls_map.items() if isinstance(k, str)}\nidx2label = {v: k for k, v in label2idx.items()}\nprint(f\"{num_classes} classes loaded.\")\nprint(label2idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-03T02:44:11.047127Z","iopub.execute_input":"2025-08-03T02:44:11.047843Z","iopub.status.idle":"2025-08-03T02:44:11.057005Z","shell.execute_reply.started":"2025-08-03T02:44:11.047816Z","shell.execute_reply":"2025-08-03T02:44:11.056208Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataloader\nCurrently, we only load third_view video frames. You may add more to achieve better performances.\n\nTo do this, simply adjust the code below to make it takes a list of video_types, with items corresponds to names of the subdirectories under ```/kaggle/input/bin-ego-360-challenge-classification-ext/kaggle_ready/```.\n","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nfrom glob import glob\nimport json\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset\nfrom torchvision.transforms.functional import resize as resize_frames\nfrom torchvision.transforms._transforms_video import (\n    RandomHorizontalFlipVideo,\n    NormalizeVideo,\n)\nfrom decord import VideoReader, cpu\nimport torch.nn.functional as F \n\n\nimport torch, numpy as np, torch.nn.functional as F\n    \nclass VideoDataset(Dataset):\n    \"\"\"\n    TRAIN mode – uses index.json annotations\n    TEST  mode – globs *.mp4 files (no labels)\n\n    The tensor layout is kept as **(C T H W)** during preprocessing so that\n    `NormalizeVideo`, which normalises along dim 0, sees channels first.\n    If your model expects **(T C H W)**, a final `permute` restores that order.\n    --------------------\n    +++\n        This is just a basic dataloader, \n        you need to customize it to support \n        multi-modalitie loading.\n    +++\n    --------------------\n    \"\"\"\n\n    def __init__(\n        self,\n        root: Path,\n        index_json: Path | None,\n        video_type: str = \"third_person\",\n        num_frames: int = 16,\n        frame_size: int = 224,\n        train: bool = True,\n    ):\n        self.root = Path(root)\n        self.train = train\n        self.video_type = video_type\n        self.num_frames = num_frames\n        self.H = self.W = frame_size\n\n        if self.train:\n            # ─── TRAIN: load records & build label map ───\n            with open(index_json, encoding=\"utf-8\") as f:\n                meta: list[dict] = json.load(f)\n\n            categories = sorted({m[\"category\"] for m in meta})\n            self.label2idx = {c: i for i, c in enumerate(categories)}\n\n            self.records = [\n                {\n                    \"path\": self.root / video_type / f\"{m['uuid']}.mp4\",\n                    \"label\": self.label2idx[m[\"category\"]],\n                    \"id\" : str(m['uuid'])\n                }\n                for m in meta\n            ]\n        else:\n            # ─── TEST: scan /<type>/*.mp4 ───\n            pattern = str(self.root / video_type / \"*.mp4\")\n            self.records = [{\"path\": Path(p)} for p in sorted(glob(pattern))]\n            if not self.records:\n                raise RuntimeError(f\"No test videos found at {pattern}\")\n\n        # transforms\n        self.flip = RandomHorizontalFlipVideo(p=0.5) if self.train else None\n        self.norm = NormalizeVideo(\n            mean=[0.45, 0.45, 0.45], std=[0.225, 0.225, 0.225]\n        )\n\n    # ------------------------------------------------------------------ #\n    def __len__(self) -> int:\n        return len(self.records)\n\n    # ------------------------------------------------------------------ #\n    def _uniform_sample(self, total: int) -> np.ndarray:\n        \"\"\"Return `self.num_frames` indices uniformly spanning the clip.\"\"\"\n        return np.linspace(0, max(total - 1, 0), self.num_frames, dtype=\"int64\")\n\n    # ------------------------------------------------------------------ #\n    def __getitem__(self, idx: int):\n        rec = self.records[idx]\n        vpath: Path = rec[\"path\"]\n\n        # ─── Video read (Decord) ─────────────────────────────────────── #\n        if not os.path.isfile(str(vpath)):\n            print(f\"{vpath} not found\")\n        vr = VideoReader(str(vpath), ctx=cpu(0))\n        frame_ids = self._uniform_sample(len(vr))\n        frames = vr.get_batch(frame_ids).asnumpy()  # (T H W C) uint8\n\n        video = torch.from_numpy(frames).permute(3, 0, 1, 2).float() / 255.0\n\n        video = resize_frames(video, [self.H, self.W])  # (C T H W)\n\n        if self.train and self.flip is not None:\n            video = self.flip(video)  # still (C T H W)\n\n        video = self.norm(video)  # (C T H W)\n\n        # video = video.permute(1, 0, 2, 3)  # (T C H W)\n        if self.train:\n        # print(rec[\"label\"])\n            return video, rec[\"label\"]        # (video, int)\n        else:\n            return video, str(vpath.stem)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-03T02:53:41.32495Z","iopub.execute_input":"2025-08-03T02:53:41.325274Z","iopub.status.idle":"2025-08-03T02:53:41.33986Z","shell.execute_reply.started":"2025-08-03T02:53:41.325248Z","shell.execute_reply":"2025-08-03T02:53:41.339099Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hyperparameters","metadata":{}},{"cell_type":"code","source":"# ─────────────── TRAIN / VAL loaders ───────────────\nBATCH_SIZE  = 4\nVAL_SPLIT   = 0.1\nNUM_FRAMES  = 16\n\nEPOCHS = 50\nACCUM = 1\nfull_ds  = VideoDataset(root, index_path,\n                           num_frames=NUM_FRAMES, train=True)\nval_len  = int(len(full_ds) * VAL_SPLIT)\ntrain_ds, val_ds = random_split(full_ds, [len(full_ds) - val_len, val_len])\n\ntrain_loader = DataLoader(train_ds, batch_size=BATCH_SIZE, shuffle=True,\n                          num_workers=4)\nval_loader   = DataLoader(val_ds,   batch_size=BATCH_SIZE, shuffle=False,\n                          num_workers=4)\n\nprint(f\"Training videos: {len(train_ds)}, Validation videos: {len(val_ds)}\")\n\n# ─────────────── TEST-set loader ───────────────\n# test videos live in /kaggle/working/360x_dataset/{video_type}/*.mp4\ntest_root   = Path(\"/kaggle/input/bin-ego-360-challenge-classification-ext/kaggle_ready/\")\ntest_ds     = VideoDataset(test_root, index_json=None,\n                              num_frames=NUM_FRAMES, train=False)\ntest_loader = DataLoader(test_ds, batch_size=1, shuffle=False,\n                         num_workers=4)\nprint(f\"Test videos: {len(test_ds)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-03T02:53:41.341234Z","iopub.execute_input":"2025-08-03T02:53:41.341436Z","iopub.status.idle":"2025-08-03T02:53:41.36607Z","shell.execute_reply.started":"2025-08-03T02:53:41.341421Z","shell.execute_reply":"2025-08-03T02:53:41.365426Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pick your favourite model here!\nAny model under https://docs.pytorch.org/vision/main/models.html#video-classification should be a good start point.","metadata":{}},{"cell_type":"code","source":"from torchvision.models.video import r2plus1d_18\nfrom torch.optim.lr_scheduler import SequentialLR, LinearLR, CosineAnnealingLR\n# pick your favorite model\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = r2plus1d_18(weights=\"KINETICS400_V1\")\nmodel.fc = nn.Linear(model.fc.in_features, num_classes)\nmodel.to(device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.AdamW(model.parameters(), lr=1e-4, weight_decay=1e-4)\nscaler = torch.cuda.amp.GradScaler()\nWARMUP_EPOCHS = 1  # 1–3 is typical for fine-tuning\n\n# after you create the optimizer\nwarmup = LinearLR(optimizer, start_factor=0.1, total_iters=WARMUP_EPOCHS)\ncosine = CosineAnnealingLR(optimizer, T_max=EPOCHS - WARMUP_EPOCHS, eta_min=5e-6)\nscheduler = SequentialLR(optimizer, schedulers=[warmup, cosine], milestones=[WARMUP_EPOCHS])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-03T02:53:41.366843Z","iopub.execute_input":"2025-08-03T02:53:41.367133Z","iopub.status.idle":"2025-08-03T02:53:41.92979Z","shell.execute_reply.started":"2025-08-03T02:53:41.367117Z","shell.execute_reply":"2025-08-03T02:53:41.928877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n@torch.no_grad()\ndef evaluate(loader):\n    model.eval()\n    top1 = 0\n    total = 0\n    for vids, labels in tqdm(loader):\n        vids, labels = vids.to(device), labels.to(device)\n        logits = model(vids)\n        preds = logits.argmax(1)\n        top1 += (preds == labels).sum().item()\n        total += labels.size(0)\n    return top1 / total\n\ndef train_one_epoch(loader, accumulation=1):\n    model.train()\n    running_loss = 0.0\n    optimizer.zero_grad(set_to_none=True)\n    for step, (vids, labels) in enumerate(tqdm(loader)):\n        vids, labels = vids.to(device), labels.to(device)\n        with torch.cuda.amp.autocast():\n            logits = model(vids)\n            loss = criterion(logits, labels) / accumulation\n        scaler.scale(loss).backward()\n        if (step + 1) % accumulation == 0:\n            scaler.step(optimizer)\n            scaler.update()\n            optimizer.zero_grad(set_to_none=True)\n        running_loss += loss.item() * accumulation\n    return running_loss / len(loader)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-03T02:53:41.930627Z","iopub.execute_input":"2025-08-03T02:53:41.930977Z","iopub.status.idle":"2025-08-03T02:53:41.937672Z","shell.execute_reply.started":"2025-08-03T02:53:41.930938Z","shell.execute_reply":"2025-08-03T02:53:41.936947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nACCUM = 1\nbest_acc = 0.0\nckpt_dir = Path(\"./checkpoints\")\nckpt_dir.mkdir(exist_ok=True)\n\nfor epoch in range(1, EPOCHS + 1):\n    loss = train_one_epoch(train_loader, ACCUM)\n    acc = evaluate(val_loader)\n    print(f\"Epoch {epoch:02d}/{EPOCHS} | loss {loss:.4f} | val@1 {acc*100:.2f}%\")\n    scheduler.step()\n\n    if acc > best_acc:\n        best_acc = acc\n        torch.save({\n            \"epoch\": epoch,\n            \"model_state\": model.state_dict(),\n            \"optimizer_state\": optimizer.state_dict(),\n            \"scheduler_state\": scheduler.state_dict(),  # <-- save this too\n            \"scaler_state\": scaler.state_dict(),        # <-- optional but recommended\n            \"label2idx\": label2idx\n        }, ckpt_dir / \"best_third_person.pt\")\n        print(f\"✔️  New best model saved to {ckpt_dir}\")\nprint(f\"Training complete. Best validation accuracy: {best_acc*100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-03T02:53:41.939302Z","iopub.execute_input":"2025-08-03T02:53:41.939532Z","iopub.status.idle":"2025-08-03T02:56:56.142641Z","shell.execute_reply.started":"2025-08-03T02:53:41.939516Z","shell.execute_reply":"2025-08-03T02:56:56.140613Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Get your results\nDownload the my_submission.csv and upload it to the competition after you finish!","metadata":{}},{"cell_type":"code","source":"# # ─────────────── TEST inference + CSV writer ───────────────\n\n# ─────────────── TEST inference + CSV writer (one-hot) ───────────────\n@torch.no_grad()\ndef infer_and_save(model, loader, num_classes: int, outfile: str = \"submission.csv\"):\n    \"\"\"\n    Run *model* on every sample in *loader* (test mode) and write predictions\n    to `outfile` in the competition’s one-hot CSV format:\n\n        id,0,1,2,…,27\n        f2c7f948-…,0,0,1,…,0\n        …\n\n    Each row has exactly one “1” at the predicted class index.\n    \"\"\"\n    import pandas as pd\n\n    model.eval()\n    rows = []\n\n    for vids, names in tqdm(loader, desc=\"Infer\"):\n        vids = vids.to(device)            # (B C T H W)\n        logits = model(vids)\n        preds  = logits.argmax(1).cpu().tolist()     # list[int]\n\n        for name, cls_idx in zip(names, preds):\n            one_hot          = [0] * num_classes\n            one_hot[cls_idx] = 1\n            # print(name)\n            rows.append({\"id\": name, **{str(i): one_hot[i] for i in range(num_classes)}})\n\n    df = pd.DataFrame(rows)\n    df = df[[\"id\"] + [str(i) for i in range(num_classes)]]  # enforce column order\n    df.to_csv(outfile, index=False)\n    print(f\"✅  Submission saved to {outfile}  ({len(df)} rows)\")\n\n\ninfer_and_save(model, test_loader, num_classes, \"my_submission.csv\")\n\n# ─────────────── TEST inference + CSV writer (probabilities) ───────────────\n# @torch.no_grad()\n# def infer_and_save(model, loader, num_classes: int, outfile: str = \"submission.csv\"):\n#     \"\"\"\n#     Run *model* on every sample in *loader* (test mode) and write probabilistic\n#     predictions to `outfile` with columns:\n#         id,0,1,...,C-1\n#     where each row contains softmax probabilities for all classes.\n#     \"\"\"\n#     import pandas as pd\n#     import torch.nn.functional as F\n\n#     model.eval()\n#     rows = []\n\n#     for vids, names in tqdm(loader, desc=\"Infer\"):\n#         vids = vids.to(device)                    # (B, C, T, H, W)\n#         logits = model(vids)                      # (B, num_classes)\n#         probs  = F.softmax(logits, dim=1).cpu().numpy()  # (B, num_classes)\n\n#         # Make 'names' robust to different collate behaviours\n#         if torch.is_tensor(names):\n#             names = [str(x) for x in names.tolist()]\n#         elif isinstance(names, (Path, str)):\n#             names = [str(names)]\n#         else:\n#             names = [str(n) for n in list(names)]\n\n#         for name, p in zip(names, probs):\n#             row = {\"id\": name}\n#             row.update({str(i): float(p[i]) for i in range(num_classes)})\n#             rows.append(row)\n\n#     df = pd.DataFrame(rows)\n#     df = df[[\"id\"] + [str(i) for i in range(num_classes)]]  # enforce column order\n#     df.to_csv(outfile, index=False, float_format=\"%.8f\")\n#     print(f\"✅  Saved probability submission with {len(df)} rows → {outfile}\")\n\n\n# # call the function (everything else remains unchanged)\n# infer_and_save(model, test_loader, num_classes, \"my_submission.csv\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-03T02:56:59.126988Z","iopub.execute_input":"2025-08-03T02:56:59.127302Z","iopub.status.idle":"2025-08-03T02:59:02.996716Z","shell.execute_reply.started":"2025-08-03T02:56:59.127276Z","shell.execute_reply":"2025-08-03T02:59:02.995877Z"}},"outputs":[],"execution_count":null}]}