{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":41875,"databundleVersionId":5521661},{"sourceType":"datasetVersion","sourceId":6037729,"datasetId":3453672,"databundleVersionId":6115849}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:30:33.789508Z","iopub.execute_input":"2026-03-02T20:30:33.790129Z","iopub.status.idle":"2026-03-02T20:30:33.816683Z","shell.execute_reply.started":"2026-03-02T20:30:33.790099Z","shell.execute_reply":"2026-03-02T20:30:33.816126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# قراءة الـ Labels\ntrain_terms = pd.read_csv('/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_terms.tsv', sep='\\t')\nprint(train_terms.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:30:35.010506Z","iopub.execute_input":"2026-03-02T20:30:35.01083Z","iopub.status.idle":"2026-03-02T20:30:37.328537Z","shell.execute_reply.started":"2026-03-02T20:30:35.010806Z","shell.execute_reply":"2026-03-02T20:30:37.327831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# قراءة ملف الـ IA\nia_data = pd.read_csv('/kaggle/input/competitions/cafa-5-protein-function-prediction/IA.txt', sep='\\t', names=['term', 'ia'])\n\n# دمجها مع بيانات التدريب لرؤية قيمة كل وظيفة\ntrain_with_ia = train_terms.merge(ia_data, on='term', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:30:38.550454Z","iopub.execute_input":"2026-03-02T20:30:38.550762Z","iopub.status.idle":"2026-03-02T20:30:39.496063Z","shell.execute_reply.started":"2026-03-02T20:30:38.550737Z","shell.execute_reply":"2026-03-02T20:30:39.495285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Define paths\nBASE_PATH = '/kaggle/input/competitions/cafa-5-protein-function-prediction'\n\n# Load files\nia_data = pd.read_csv(f'{BASE_PATH}/IA.txt', sep='\\t', names=['term', 'ia'])\ntrain_terms = pd.read_csv(f'{BASE_PATH}/Train/train_terms.tsv', sep='\\t')\n\nprint(\"Step 1: Libraries loaded and data files read successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:30:51.281632Z","iopub.execute_input":"2026-03-02T20:30:51.28222Z","iopub.status.idle":"2026-03-02T20:30:53.989482Z","shell.execute_reply.started":"2026-03-02T20:30:51.282194Z","shell.execute_reply":"2026-03-02T20:30:53.988807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Style settings\nsns.set_theme(style=\"whitegrid\")\nplt.figure(figsize=(10, 5))\n\n# Plotting\nsns.histplot(ia_data['ia'], bins=50, kde=True, color='#4A90E2')\nplt.axvline(ia_data['ia'].mean(), color='red', linestyle='--', label='Mean IA')\n\nplt.title('Distribution of Information Accretion (IA)', fontsize=14)\nplt.xlabel('IA Value')\nplt.ylabel('Count')\nplt.legend()\nplt.show()\n\nprint(\"Step 2: IA Distribution plotted.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:31:17.353621Z","iopub.execute_input":"2026-03-02T20:31:17.353917Z","iopub.status.idle":"2026-03-02T20:31:17.889406Z","shell.execute_reply.started":"2026-03-02T20:31:17.353893Z","shell.execute_reply":"2026-03-02T20:31:17.888508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n\n# Get top 15 terms\ntop_15_terms = train_terms['term'].value_counts().head(15)\n\n# Plotting\nsns.barplot(x=top_15_terms.values, y=top_15_terms.index, palette='viridis')\n\nplt.title('Top 15 Most Frequent GO Terms in Training Data', fontsize=14)\nplt.xlabel('Number of Proteins')\nplt.ylabel('GO Term ID')\nplt.show()\n\nprint(\"Step 3: Top GO terms plotted.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:31:26.397074Z","iopub.execute_input":"2026-03-02T20:31:26.397757Z","iopub.status.idle":"2026-03-02T20:31:27.062143Z","shell.execute_reply.started":"2026-03-02T20:31:26.397726Z","shell.execute_reply":"2026-03-02T20:31:27.061335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count aspects\naspect_counts = train_terms['aspect'].value_counts()\n\nplt.figure(figsize=(8, 8))\nplt.pie(aspect_counts, labels=aspect_counts.index, autopct='%1.1f%%', \n        startangle=140, colors=sns.color_palette('pastel'), explode=(0.05, 0.05, 0.05))\n\nplt.title('Distribution of GO Aspects', fontsize=14)\nplt.show()\n\nprint(\"Step 4: Aspects pie chart created.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:31:30.51522Z","iopub.execute_input":"2026-03-02T20:31:30.51605Z","iopub.status.idle":"2026-03-02T20:31:30.935096Z","shell.execute_reply.started":"2026-03-02T20:31:30.516021Z","shell.execute_reply":"2026-03-02T20:31:30.934446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers\n\n# Assuming input_dim is 1024 (ProtBERT) and labels are 1500\ninput_dim = 1024 \nnum_labels = 1500 \n\nmodel = tf.keras.Sequential([\n    layers.Input(shape=(input_dim,)),\n    layers.Dense(512, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(256, activation='relu'),\n    layers.Dense(num_labels, activation='sigmoid') # Sigmoid for multi-label\n])\n\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy']\n)\n\nmodel.summary()\nprint(\"Step 5: MLP Model built and compiled.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:31:34.813757Z","iopub.execute_input":"2026-03-02T20:31:34.814048Z","iopub.status.idle":"2026-03-02T20:32:00.509545Z","shell.execute_reply.started":"2026-03-02T20:31:34.814023Z","shell.execute_reply":"2026-03-02T20:32:00.508936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count how many GO terms each protein is associated with\nprotein_counts = train_terms.groupby('EntryID')['term'].count()\n\nprint(f\"Average number of functions per protein: {protein_counts.mean():.2f}\")\nprint(f\"Maximum functions for one protein: {protein_counts.max()}\")\n\n# Visualize the distribution\nimport matplotlib.pyplot as plt\nprotein_counts.hist(bins=50, color='skyblue', edgecolor='black')\nplt.title('Distribution of Number of Functions per Protein')\nplt.xlabel('Number of GO Terms')\nplt.ylabel('Number of Proteins')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:32:00.510902Z","iopub.execute_input":"2026-03-02T20:32:00.511362Z","iopub.status.idle":"2026-03-02T20:32:01.400962Z","shell.execute_reply.started":"2026-03-02T20:32:00.51134Z","shell.execute_reply":"2026-03-02T20:32:01.400353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge train_terms with ia_data to see the weights of our training labels\ntrain_with_ia = train_terms.merge(ia_data, on='term', how='left')\n\n# Show the 10 most specific (highest IA) functions in our training set\nprint(\"Top 10 most specific functions (Highest IA):\")\nprint(train_with_ia.sort_values(by='ia', ascending=False).head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:32:05.421518Z","iopub.execute_input":"2026-03-02T20:32:05.422068Z","iopub.status.idle":"2026-03-02T20:32:08.326022Z","shell.execute_reply.started":"2026-03-02T20:32:05.42204Z","shell.execute_reply":"2026-03-02T20:32:08.32523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\n# البحث عن ملف الـ train_terms.tsv\ntrain_terms_path = None\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file == 'train_terms.tsv':\n            train_terms_path = os.path.join(root, file)\n            break\n\nif train_terms_path:\n    print(f\"✅ Found Train Terms at: {train_terms_path}\")\n    train_terms = pd.read_csv(train_terms_path, sep=\"\\t\")\n    \n    # اختيار أشهر 1500 وظيفة\n    top_terms = train_terms['term'].value_counts().head(1500).index.tolist()\n    \n    # تحميل الـ IDs عشان نضمن الترتيب\n    # المسار ده إحنا كشفناه قبل كدة في الرسايل اللي فاتت\n    train_ids_path = '/kaggle/input/datasets/kriukov/t5embeds/t5embeds/train_ids.npy'\n    train_ids = np.load(train_ids_path)\n    \n    # تجهيز مصفوفة التارجت (y_train)\n    print(\"⏳ Creating y_train matrix...\")\n    y_train = np.zeros((len(train_ids), 1500))\n    \n    # فلترة سريعة للبيانات\n    train_terms_filtered = train_terms[train_terms['term'].isin(top_terms)]\n    protein_to_term = train_terms_filtered.groupby('EntryID')['term'].apply(list).to_dict()\n    \n    for i, protein_id in enumerate(train_ids):\n        if protein_id in protein_to_term:\n            for term in protein_to_term[protein_id]:\n                y_train[i, top_terms.index(term)] = 1\n                \n    print(f\"✅ y_train ready! Shape: {y_train.shape}\")\nelse:\n    print(\"❌ Could not find train_terms.tsv. Please make sure the competition dataset is added.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:32:10.697963Z","iopub.execute_input":"2026-03-02T20:32:10.698557Z","iopub.status.idle":"2026-03-02T20:32:35.845987Z","shell.execute_reply.started":"2026-03-02T20:32:10.698533Z","shell.execute_reply":"2026-03-02T20:32:35.845255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Select the top 1500 terms to predict\ntop_terms = train_terms['term'].value_counts().head(1500).index\n\n# 2. Filter training data\ntrain_terms_filtered = train_terms[train_terms['term'].isin(top_terms)]\n\n# 3. Create a pivot table (Proteins as rows, GO terms as columns)\n# This creates a matrix of 0s and 1s\ntrain_labels_pivot = train_terms_filtered.pivot_table(index='EntryID', columns='term', aggfunc='size', fill_value=0)\n\n# Convert to numpy array for the model\ny_train = train_labels_pivot.values\n\nprint(f\"Target data (y_train) ready! Shape: {y_train.shape}\")\n# This means: (Number of Proteins, 1500 Functions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:32:43.030758Z","iopub.execute_input":"2026-03-02T20:32:43.031446Z","iopub.status.idle":"2026-03-02T20:32:49.951574Z","shell.execute_reply.started":"2026-03-02T20:32:43.031415Z","shell.execute_reply":"2026-03-02T20:32:49.950906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file == 'train_terms.tsv':\n            print(f\"🎯 وجدت الملف هنا: {os.path.join(root, file)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:32:59.422764Z","iopub.execute_input":"2026-03-02T20:32:59.42334Z","iopub.status.idle":"2026-03-02T20:32:59.432315Z","shell.execute_reply.started":"2026-03-02T20:32:59.423314Z","shell.execute_reply":"2026-03-02T20:32:59.431645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. البحث عن ملف الـ Test بنفس الطريقة الذكية\ntest_path = None\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file == 'test_embeds.npy':\n            test_path = os.path.join(root, file)\n            break\n\nif test_path:\n    print(f\"✅ Found Test Embeddings at: {test_path}\")\n    x_test = np.load(test_path)\n    print(f\"✅ x_test loaded. Shape: {x_test.shape}\")\nelse:\n    print(\"❌ Could not find test_embeds.npy\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:33:02.716999Z","iopub.execute_input":"2026-03-02T20:33:02.71727Z","iopub.status.idle":"2026-03-02T20:33:10.793761Z","shell.execute_reply.started":"2026-03-02T20:33:02.717249Z","shell.execute_reply":"2026-03-02T20:33:10.792987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport os\n\n# 1. البحث التلقائي عن ملفات الـ Test\ntest_embeds_path = None\ntest_ids_path = None\n\nprint(\"🔍 Searching for test files...\")\nfor root, dirs, files in os.walk('/kaggle/input/t5embeds'):\n    for file in files:\n        if file == 'test_embeds.npy':\n            test_embeds_path = os.path.join(root, file)\n        if file == 'test_ids.npy':\n            test_ids_path = os.path.join(root, file)\n\n# 2. التأكد من إيجاد الملفات وتحميلها\nif test_embeds_path and test_ids_path:\n    print(f\"✅ Found Test Embeddings at: {test_embeds_path}\")\n    print(f\"✅ Found Test IDs at: {test_ids_path}\")\n    \n    # تحميل البيانات\n    x_test = np.load(test_embeds_path)\n    test_ids = np.load(test_ids_path)\n    \n    print(f\"✅ Data loaded. x_test shape: {x_test.shape}, test_ids shape: {test_ids.shape}\")\n    \n    # 3. عمل التوقعات (Inference)\n    print(\"🚀 Running predictions... (this might take a few seconds)\")\n    predictions = model.predict(x_test, batch_size=128)\n    print(\"✅ Predictions completed successfully!\")\n    \nelse:\n    print(\"❌ Error: Still can't find test files. Please check the folder structure on the right sidebar.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:33:20.602312Z","iopub.execute_input":"2026-03-02T20:33:20.602946Z","iopub.status.idle":"2026-03-02T20:33:20.609395Z","shell.execute_reply.started":"2026-03-02T20:33:20.602915Z","shell.execute_reply":"2026-03-02T20:33:20.60864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"📂 Listing all files in /kaggle/input:\")\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file.endswith('.npy'):\n            print(f\"📍 Found: {os.path.join(root, file)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:33:24.751318Z","iopub.execute_input":"2026-03-02T20:33:24.752022Z","iopub.status.idle":"2026-03-02T20:33:24.761665Z","shell.execute_reply.started":"2026-03-02T20:33:24.751992Z","shell.execute_reply":"2026-03-02T20:33:24.760963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# 1. المسارات الحقيقية اللي كشفناها\nFINAL_TEST_EMBEDS_PATH = '/kaggle/input/datasets/kriukov/t5embeds/t5embeds/test_embeds.npy'\nFINAL_TEST_IDS_PATH = '/kaggle/input/datasets/kriukov/t5embeds/t5embeds/test_ids.npy'\n\ntry:\n    # 2. تحميل البيانات\n    print(\"⏳ Loading test data... Please wait.\")\n    x_test = np.load(FINAL_TEST_EMBEDS_PATH)\n    test_ids = np.load(FINAL_TEST_IDS_PATH)\n    \n    print(f\"✅ Success! x_test shape: {x_test.shape}\")\n    print(f\"✅ Success! test_ids shape: {test_ids.shape}\")\n    \n    # 3. عمل التوقعات باستخدام الموديل اللي دربناه\n    print(\"🚀 Running predictions... (Inference)\")\n    predictions = model.predict(x_test, batch_size=128)\n    \n    print(\"✅ Predictions finished successfully! You can now move to the submission cell.\")\n\nexcept Exception as e:\n    print(f\"❌ Error loading files: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:33:27.645564Z","iopub.execute_input":"2026-03-02T20:33:27.645902Z","iopub.status.idle":"2026-03-02T20:33:34.343981Z","shell.execute_reply.started":"2026-03-02T20:33:27.645876Z","shell.execute_reply":"2026-03-02T20:33:34.343165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# تجهيز البيانات للتسليم\nsubmission_data = []\n\nprint(\"📝 Building the submission list...\")\n\n# استخدام النطاق الصحيح للتأكد من عدم وجود مسافات غريبة\nfor i in range(len(test_ids)):\n    protein_id = test_ids[i]\n    pred_scores = predictions[i]\n    \n    for j in range(len(top_terms)):\n        if pred_scores[j] > 0.1: \n            submission_data.append([protein_id, top_terms[j], round(pred_scores[j], 3)])\n\n# تحويل لجدول وحفظه\nsubmission_df = pd.DataFrame(submission_data, columns=['Protein Id', 'GO Term Id', 'Prediction'])\nsubmission_df.to_csv('submission.tsv', sep='\\t', index=False, header=False)\n\nprint(f\"✅ Done! Created submission with {len(submission_df)} rows.\")\nprint(\"📁 File saved as 'submission.tsv'.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:33:39.117929Z","iopub.execute_input":"2026-03-02T20:33:39.118518Z","iopub.status.idle":"2026-03-02T20:35:04.925637Z","shell.execute_reply.started":"2026-03-02T20:33:39.118492Z","shell.execute_reply":"2026-03-02T20:35:04.924641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# رسم الدقة (Accuracy)\nplt.figure(figsize=(12, 5))\n\n# الرسم الأول: Accuracy\nplt.subplot(1, 2, 1)\nplt.plot(history.history['binary_accuracy'], label='Train Accuracy', color='#4A90E2', linewidth=2)\nplt.plot(history.history['val_binary_accuracy'], label='Validation Accuracy', color='#F5A623', linewidth=2)\nplt.title('Model Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\n# الرسم الثاني: Loss (الخطأ)\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss', color='#D0021B', linewidth=2)\nplt.plot(history.history['val_loss'], label='Validation Loss', color='#7ED321', linewidth=2)\nplt.title('Model Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T20:35:04.926511Z","iopub.status.idle":"2026-03-02T20:35:04.926916Z","shell.execute_reply.started":"2026-03-02T20:35:04.926722Z","shell.execute_reply":"2026-03-02T20:35:04.926746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# استخراج آخر القيم المسجلة\nfinal_train_acc = history.history['binary_accuracy'][-1]\nfinal_val_acc = history.history['val_binary_accuracy'][-1]\nfinal_train_loss = history.history['loss'][-1]\nfinal_val_loss = history.history['val_loss'][-1]\n\nprint(\"📊 --- Final Model Metrics ---\")\nprint(f\"✅ Training Accuracy:   {final_train_acc:.4f} ({final_train_acc*100:.2f}%)\")\nprint(f\"✅ Validation Accuracy: {final_val_acc:.4f} ({final_val_acc*100:.2f}%)\")\nprint(\"-\" * 30)\nprint(f\"📉 Training Loss:       {final_train_loss:.4f}\")\nprint(f\"📉 Validation Loss:     {final_val_loss:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T01:57:48.669298Z","iopub.status.idle":"2026-02-28T01:57:48.669628Z","shell.execute_reply.started":"2026-02-28T01:57:48.669456Z","shell.execute_reply":"2026-02-28T01:57:48.669472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score\n\n# 1. الحصول على توقعات الموديل لبيانات الـ Validation\nval_size = int(0.2 * len(x_train)) # إحنا كنا واخدين 20% للـ validation\nx_val_subset = x_train[-val_size:]\ny_val_subset = y_train[-val_size:]\n\nval_preds = model.predict(x_val_subset)\n\n# 2. تحويل الاحتمالات لأرقام (0 أو 1) بناءً على Threshold 0.1\nval_preds_binary = (val_preds > 0.1).astype(int)\n\n# 3. حساب الـ F1 Score (Macro و Micro)\nf1_micro = f1_score(y_val_subset, val_preds_binary, average='micro')\n\nprint(f\"🎯 F1-Score (Micro): {f1_micro:.4f}\")\nif f1_micro > 0.3:\n    print(\"🚀 This is a very good start for this competition!\")\nelse:\n    print(\"💡 The model is okay, but we can improve it by balancing the data.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T01:57:48.671092Z","iopub.status.idle":"2026-02-28T01:57:48.671426Z","shell.execute_reply.started":"2026-02-28T01:57:48.671283Z","shell.execute_reply":"2026-02-28T01:57:48.671303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.manifold import TSNE\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# هناخد عينة من 1000 بروتين عشان الرسمة متكونش زحمة\nn_samples = 1000\nx_sample = x_train[:n_samples]\n\n# ضغط الأبعاد من 1024 لـ 2 باستخدام t-SNE\ntsne = TSNE(n_components=2, random_state=42)\nx_2d = tsne.fit_transform(x_sample)\n\nplt.figure(figsize=(10, 7))\nplt.scatter(x_2d[:, 0], x_2d[:, 1], alpha=0.5, c='gray')\nplt.title(\"Before Training: Raw T5 Embeddings Distribution\")\nplt.xlabel(\"t-SNE dimension 1\")\nplt.ylabel(\"t-SNE dimension 2\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T01:57:48.672848Z","iopub.status.idle":"2026-02-28T01:57:48.673118Z","shell.execute_reply.started":"2026-02-28T01:57:48.672987Z","shell.execute_reply":"2026-02-28T01:57:48.673002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# اختيار بروتينات تنتمي لأشهر وظيفتين فقط للمقارنة\n# هنلون النقط بناءً على الوظيفة الحقيقية (Target)\nlabels_sample = y_train[:n_samples, 0] # هناخد الوظيفة الأولى كمثال للتلوين\n\nplt.figure(figsize=(10, 7))\nsns.scatterplot(x=x_2d[:, 0], y=x_2d[:, 1], hue=labels_sample, palette='viridis')\nplt.title(\"After Training: How Model Sees Proteins (Colored by Function)\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T01:57:48.674307Z","iopub.status.idle":"2026-02-28T01:57:48.67475Z","shell.execute_reply.started":"2026-02-28T01:57:48.674508Z","shell.execute_reply":"2026-02-28T01:57:48.674532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# حفظ الملف النهائي بصيغة tsv\nsubmission_df.to_csv('submission.tsv', sep='\\t', index=False, header=False)\nprint(\"✅ File saved successfully as submission.tsv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T01:57:48.676082Z","iopub.status.idle":"2026-02-28T01:57:48.676409Z","shell.execute_reply.started":"2026-02-28T01:57:48.676234Z","shell.execute_reply":"2026-02-28T01:57:48.676249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}