{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  Stanford RNA 3D Folding Challenge\nThis notebook loads the competition dataset, generates predictions, and creates a valid submission file.\n\nWe will:\n1. Load train, validation, and test datasets  \n2. Visualize sequences with simple plots (white, black, red, orange)  \n3. Generate placeholder predictions (replace with model outputs later)  \n4. Save submission CSV\n\n","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Paths\nDATA_DIR = '/kaggle/input/stanford-rna-3d-folding-2'\nMSA_DIR = os.path.join(DATA_DIR, 'MSA')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T19:58:39.075499Z","iopub.execute_input":"2026-01-16T19:58:39.075846Z","iopub.status.idle":"2026-01-16T19:58:39.081892Z","shell.execute_reply.started":"2026-01-16T19:58:39.075814Z","shell.execute_reply":"2026-01-16T19:58:39.080953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load datasets\ntrain_seq = pd.read_csv(os.path.join(DATA_DIR, \"train_sequences.csv\"))\ntrain_labels = pd.read_csv(os.path.join(DATA_DIR, \"train_labels.csv\"))\nval_seq = pd.read_csv(os.path.join(DATA_DIR, \"validation_sequences.csv\"))\nval_labels = pd.read_csv(os.path.join(DATA_DIR, \"validation_labels.csv\"))\ntest_seq = pd.read_csv(os.path.join(DATA_DIR, \"test_sequences.csv\"))\nsample_submission = pd.read_csv(os.path.join(DATA_DIR, \"sample_submission.csv\"))\n\n# Quick overview\nprint(\"Train columns:\", train_seq.columns)\nprint(\"Test columns:\", test_seq.columns)\nprint(\"Number of test rows:\", len(test_seq))\ntrain_labels = pd.read_csv(\n    os.path.join(DATA_DIR, \"train_labels.csv\"),\n    dtype=str\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T20:00:40.426464Z","iopub.execute_input":"2026-01-16T20:00:40.426837Z","iopub.status.idle":"2026-01-16T20:01:07.38168Z","shell.execute_reply.started":"2026-01-16T20:00:40.426802Z","shell.execute_reply":"2026-01-16T20:01:07.380364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sequence lengths\nseq_lengths = train_seq['sequence'].str.len()\n\n# Counts per length\nlength_counts = seq_lengths.value_counts().sort_index()\n\n# Color pattern for bars\ncolors = ['white', 'black', 'red', 'orange']\nbar_colors = [colors[i % 4] for i in range(len(length_counts))]\n\n# Plot\nplt.figure(figsize=(12,6))\nplt.bar(length_counts.index, length_counts.values, color=bar_colors, edgecolor='black', alpha=0.9)\nplt.title(\"Distribution of RNA Sequence Lengths\", fontsize=16)\nplt.xlabel(\"Sequence Length\", fontsize=12)\nplt.ylabel(\"Number of Sequences\", fontsize=12)\nplt.xticks(rotation=45)\nplt.grid(axis='y', linestyle='--', alpha=0.5)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T19:58:51.738581Z","iopub.execute_input":"2026-01-16T19:58:51.738875Z","iopub.status.idle":"2026-01-16T19:58:53.127075Z","shell.execute_reply.started":"2026-01-16T19:58:51.738849Z","shell.execute_reply":"2026-01-16T19:58:53.126151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Map nucleotides to colors\ncolors_map = {'A':'white','U':'black','G':'red','C':'orange'}\n\n# Visualize first 5 test sequences\nfor idx in range(min(5, len(test_seq))):\n    seq = test_seq.iloc[idx]['sequence'] if 'sequence' in test_seq.columns else \"\"\n    color_list = [colors_map.get(nuc,'gray') for nuc in seq]\n\n    plt.figure(figsize=(len(seq)/2,1))\n    plt.bar(range(len(seq)), [1]*len(seq), color=color_list, edgecolor='black')\n    plt.title(f\"Sequence {test_seq.iloc[idx]['target_id']}\")\n    plt.xticks([])\n    plt.yticks([])\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T19:58:53.12828Z","iopub.execute_input":"2026-01-16T19:58:53.128578Z","iopub.status.idle":"2026-01-16T19:59:00.310179Z","shell.execute_reply.started":"2026-01-16T19:58:53.128549Z","shell.execute_reply":"2026-01-16T19:59:00.308987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nDATA_DIR = \"/kaggle/input/stanford-rna-3d-folding-2\"\n\nsample_submission = pd.read_csv(\n    os.path.join(DATA_DIR, \"sample_submission.csv\")\n)\n\nsubmission = sample_submission.copy()\n\noutput_path = \"/kaggle/working/submission.csv\"\nsubmission.to_csv(output_path, index=False)\n\nassert os.path.exists(output_path)\nassert submission.shape == sample_submission.shape\nassert list(submission.columns) == list(sample_submission.columns)\n\nprint(\"submission.csv created successfully\")\nprint(\"rows:\", submission.shape[0])\nprint(\"columns:\", submission.shape[1])\nprint(\"files in /kaggle/working:\", os.listdir(\"/kaggle/working\"))\nprint(\"file size (bytes):\", os.path.getsize(output_path))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T19:59:00.312281Z","iopub.execute_input":"2026-01-16T19:59:00.312774Z","iopub.status.idle":"2026-01-16T19:59:00.381068Z","shell.execute_reply.started":"2026-01-16T19:59:00.312734Z","shell.execute_reply":"2026-01-16T19:59:00.37992Z"}},"outputs":[],"execution_count":null}]}