{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nimport os\nfrom glob import glob\nimport matplotlib.pyplot as plt\n\n# تأكيد GPU\nprint(f\"PyTorch version: {torch.__version__}\")\nprint(f\"GPU Available: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n    print(f\"GPU Count: {torch.cuda.device_count()}\")\n    !nvidia-smi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T21:54:45.516062Z","iopub.execute_input":"2026-08-06T21:54:45.516338Z","iopub.status.idle":"2026-08-06T21:54:51.754327Z","shell.execute_reply.started":"2026-08-06T21:54:45.51631Z","shell.execute_reply":"2026-08-06T21:54:51.753449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# المسارات الأساسية\nDATA_PATH = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\n# تحميل ملفات CSV\ntrain_df = pd.read_csv(f\"{DATA_PATH}/train.csv\")\ntest_df = pd.read_csv(f\"{DATA_PATH}/test.csv\")\nsample_sub = pd.read_csv(f\"{DATA_PATH}/sample_submission.csv\")\n\nprint(\"=== بيانات التدريب ===\")\nprint(f\"عدد الصفوف: {len(train_df)}\")\nprint(f\"الأعمدة: {train_df.columns.tolist()}\")\nprint(f\"\\n{train_df.head()}\")\n\nprint(\"\\n=== القيم المستهدفة ===\")\ntarget_cols = ['ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', \n               'Medial OA', 'Lateral OA', 'PF OA', 'Effusion', \n               'Synovitis', \"Baker's\", 'Contusion', 'Fracture']\nprint(target_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T21:56:25.902149Z","iopub.execute_input":"2026-08-06T21:56:25.902885Z","iopub.status.idle":"2026-08-06T21:56:25.917434Z","shell.execute_reply.started":"2026-08-06T21:56:25.902849Z","shell.execute_reply":"2026-08-06T21:56:25.916059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# استيراد جميع المكتبات\nimport torch\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nimport os\nfrom glob import glob\nimport matplotlib.pyplot as plt\n\n# تأكيد GPU\nprint(f\"PyTorch version: {torch.__version__}\")\nprint(f\"GPU Available: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n    print(f\"GPU Count: {torch.cuda.device_count()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T21:57:02.696471Z","iopub.execute_input":"2026-08-06T21:57:02.696952Z","iopub.status.idle":"2026-08-06T21:57:10.070785Z","shell.execute_reply.started":"2026-08-06T21:57:02.69692Z","shell.execute_reply":"2026-08-06T21:57:10.070044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# المسارات الأساسية\nDATA_PATH = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\n\n# تحميل ملفات CSV\ntrain_df = pd.read_csv(f\"{DATA_PATH}/train.csv\")\ntest_df = pd.read_csv(f\"{DATA_PATH}/test.csv\")\nsample_sub = pd.read_csv(f\"{DATA_PATH}/sample_submission.csv\")\n\nprint(\"=== بيانات التدريب ===\")\nprint(f\"عدد الصفوف: {len(train_df)}\")\nprint(f\"الأعمدة:\\n{train_df.columns.tolist()}\")\nprint(f\"\\nأول 5 صفوف:\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T21:57:43.727643Z","iopub.execute_input":"2026-08-06T21:57:43.728712Z","iopub.status.idle":"2026-08-06T21:57:43.944814Z","shell.execute_reply.started":"2026-08-06T21:57:43.728677Z","shell.execute_reply":"2026-08-06T21:57:43.944113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# البحث عن جميع ملفات CSV في مجلد البيانات\ncsv_files = glob(f\"{DATA_PATH}/*.csv\")\nprint(\"ملفات CSV المتاحة:\")\nfor f in csv_files:\n    print(f\"  - {os.path.basename(f)}\")\n\n# تحميل ملف التسميات الصحيح\nlabels_df = pd.read_csv(f\"{DATA_PATH}/train_labels.csv\")\nprint(f\"\\n=== ملف التسميات ===\")\nprint(f\"عدد الصفوف: {len(labels_df)}\")\nprint(f\"الأعمدة: {labels_df.columns.tolist()}\")\nprint(f\"\\nالقيم الفارغة:\")\nprint(labels_df.isnull().sum())\nprint(f\"\\nأول 5 صفوف:\")\nlabels_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T21:58:31.109811Z","iopub.execute_input":"2026-08-06T21:58:31.110385Z","iopub.status.idle":"2026-08-06T21:58:31.12862Z","shell.execute_reply.started":"2026-08-06T21:58:31.110353Z","shell.execute_reply":"2026-08-06T21:58:31.12752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# تحميل train.csv كامل\ntrain_df = pd.read_csv(f\"{DATA_PATH}/train.csv\")\n\nprint(f\"إجمالي الصفوف: {len(train_df)}\")\nprint(f\"\\nالقيم الفارغة في كل عمود:\")\nprint(train_df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T21:59:08.842891Z","iopub.execute_input":"2026-08-06T21:59:08.84376Z","iopub.status.idle":"2026-08-06T21:59:08.95533Z","shell.execute_reply.started":"2026-08-06T21:59:08.843728Z","shell.execute_reply":"2026-08-06T21:59:08.95457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# الأعمدة المستهدفة\ntarget_cols = ['ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', \n               'Medial OA', 'Lateral OA', 'PF OA', 'Effusion', \n               'Synovitis', \"Baker's\", 'Contusion', 'Fracture']\n\n# بيانات التدريب (الصفوف التي لها تسميات)\ntrain_labeled = train_df.dropna(subset=['ACL'])\nprint(f\"✅ بيانات التدريب: {len(train_labeled)} صف\")\n\n# بيانات الاختبار (بدون تسميات) - سنقدم توقعات لها\ntest_data = train_df[train_df['ACL'].isna()].copy()\nprint(f\"📤 بيانات الاختبار (للتقديم): {len(test_data)} صف\")\n\n# عرض عينة من بيانات التدريب\nprint(\"\\n=== عينة من بيانات التدريب ===\")\ntrain_labeled[['StudyInstanceUID'] + target_cols].head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T21:59:43.122635Z","iopub.execute_input":"2026-08-06T21:59:43.123442Z","iopub.status.idle":"2026-08-06T21:59:43.155462Z","shell.execute_reply.started":"2026-08-06T21:59:43.123412Z","shell.execute_reply":"2026-08-06T21:59:43.1545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# توزيع التسميات\nprint(\"=== توزيع الحالات (58 مريض) ===\\n\")\nprint(f\"{'الحالة':<20s} {'إيجابي':>8s} {'سلبي':>8s} {'نسبة %':>8s}\")\nprint(\"-\" * 48)\nfor col in target_cols:\n    vals = train_labeled[col].value_counts()\n    pos = int(vals.get(1, 0))\n    neg = int(vals.get(0, 0))\n    pct = pos / (pos + neg) * 100 if (pos + neg) > 0 else 0\n    print(f\"{col:<20s} {pos:>8d} {neg:>8d} {pct:>7.1f}%\")\n\nprint(f\"\\n✅ إجمالي الحالات الإيجابية: {train_labeled[target_cols].sum().sum():.0f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:00:20.291629Z","iopub.execute_input":"2026-08-06T22:00:20.291918Z","iopub.status.idle":"2026-08-06T22:00:20.316719Z","shell.execute_reply.started":"2026-08-06T22:00:20.291893Z","shell.execute_reply":"2026-08-06T22:00:20.315631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoTokenizer, AutoModel\nimport torch.nn as nn\n\n# تجهيز البيانات النصية\ntrain_texts = train_labeled['Report'].tolist()\ntrain_labels = train_labeled[target_cols].values.astype(np.float32)\n\nprint(f\"نصوص التدريب: {len(train_texts)}\")\nprint(f\"تسميات التدريب: {train_labels.shape}\")\n\n# تحميل BioBERT\ntokenizer = AutoTokenizer.from_pretrained(\"dmis-lab/biobert-v1.1\")\nbiobert = AutoModel.from_pretrained(\"dmis-lab/biobert-v1.1\")\nbiobert = biobert.to(device)\nbiobert.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:04:35.812679Z","iopub.execute_input":"2026-08-06T22:04:35.813171Z","iopub.status.idle":"2026-08-06T22:05:08.012108Z","shell.execute_reply.started":"2026-08-06T22:04:35.813139Z","shell.execute_reply":"2026-08-06T22:05:08.002026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoTokenizer, AutoModel\nimport torch.nn as nn\n\n# تعريف device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# تجهيز البيانات النصية\ntrain_texts = train_labeled['Report'].tolist()\ntrain_labels = train_labeled[target_cols].values.astype(np.float32)\n\nprint(f\"نصوص التدريب: {len(train_texts)}\")\nprint(f\"تسميات التدريب: {train_labels.shape}\")\n\n# تحميل BioBERT\ntokenizer = AutoTokenizer.from_pretrained(\"dmis-lab/biobert-v1.1\")\nbiobert = AutoModel.from_pretrained(\"dmis-lab/biobert-v1.1\")\nbiobert = biobert.to(device)\nbiobert.eval()\n\nprint(\"✅ BioBERT تم تحميله بنجاح!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:05:51.737117Z","iopub.execute_input":"2026-08-06T22:05:51.737578Z","iopub.status.idle":"2026-08-06T22:05:53.503103Z","shell.execute_reply.started":"2026-08-06T22:05:51.737547Z","shell.execute_reply":"2026-08-06T22:05:53.502156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# استخراج التضمينات من النصوص\ndef get_text_embeddings(texts, batch_size=4):\n    embeddings = []\n    for i in range(0, len(texts), batch_size):\n        batch = texts[i:i+batch_size]\n        inputs = tokenizer(batch, padding=True, truncation=True, \n                          max_length=512, return_tensors=\"pt\").to(device)\n        with torch.no_grad():\n            outputs = biobert(**inputs)\n            cls_emb = outputs.last_hidden_state[:, 0, :].cpu().numpy()\n            embeddings.append(cls_emb)\n    return np.vstack(embeddings)\n\nprint(\"⏳ استخراج التضمينات من 58 تقرير...\")\ntrain_embeddings = get_text_embeddings(train_texts)\nprint(f\"✅ حجم التضمينات: {train_embeddings.shape}\")  # (58, 768)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:06:37.742128Z","iopub.execute_input":"2026-08-06T22:06:37.742826Z","iopub.status.idle":"2026-08-06T22:06:39.770592Z","shell.execute_reply.started":"2026-08-06T22:06:37.742793Z","shell.execute_reply":"2026-08-06T22:06:39.76981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\n\n# تدريب مصنف لكل فئة\nmodels = {}\nauc_scores = []\n\nprint(\"=== نتائج التدريب (AUC) ===\\n\")\nfor i, col in enumerate(target_cols):\n    y = train_labels[:, i]\n    mask = ~np.isnan(y)\n    if mask.sum() > 0 and len(np.unique(y[mask])) > 1:\n        clf = LogisticRegression(max_iter=1000, random_state=42)\n        clf.fit(train_embeddings[mask], y[mask])\n        y_pred = clf.predict_proba(train_embeddings[mask])[:, 1]\n        auc = roc_auc_score(y[mask], y_pred)\n        models[col] = clf\n        auc_scores.append(auc)\n        print(f\"{col:20s}: AUC = {auc:.4f}\")\n    else:\n        models[col] = None\n        print(f\"{col:20s}: SKIP (لا توجد بيانات كافية)\")\n\nprint(f\"\\n{'='*40}\")\nprint(f\"✅ متوسط AUC: {np.mean(auc_scores):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:07:29.498049Z","iopub.execute_input":"2026-08-06T22:07:29.498705Z","iopub.status.idle":"2026-08-06T22:07:30.024984Z","shell.execute_reply.started":"2026-08-06T22:07:29.498673Z","shell.execute_reply":"2026-08-06T22:07:30.024259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score, StratifiedKFold\nfrom sklearn.linear_model import LogisticRegression\nimport numpy as np\n\nprint(\"=== AUC الحقيقي مع Cross-Validation ===\\n\")\nreal_auc_scores = []\n\nfor i, col in enumerate(target_cols):\n    y = train_labels[:, i]\n    mask = ~np.isnan(y)\n    \n    if mask.sum() > 0 and len(np.unique(y[mask])) > 1:\n        X = train_embeddings[mask]\n        y_clean = y[mask]\n        \n        # استخدام Cross-Validation\n        cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n        clf = LogisticRegression(max_iter=1000, random_state=42)\n        \n        try:\n            auc_cv = cross_val_score(clf, X, y_clean, cv=cv, scoring='roc_auc')\n            real_auc_scores.append(auc_cv.mean())\n            print(f\"{col:20s}: AUC = {auc_cv.mean():.4f} (±{auc_cv.std():.4f})\")\n        except:\n            print(f\"{col:20s}: SKIP - بيانات غير كافية\")\n            real_auc_scores.append(0.5)\n\nprint(f\"\\n{'='*50}\")\nprint(f\"✅ متوسط AUC الحقيقي: {np.mean(real_auc_scores):.4f}\")\nprint(f\"⚠️ هذا أقل بكثير من 0.9994 - الفرق هو overfitting!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:08:09.299846Z","iopub.execute_input":"2026-08-06T22:08:09.300565Z","iopub.status.idle":"2026-08-06T22:08:11.31745Z","shell.execute_reply.started":"2026-08-06T22:08:09.300531Z","shell.execute_reply":"2026-08-06T22:08:11.316584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# تحميل بيانات السلاسل\ntrain_series = pd.read_csv(f\"{DATA_PATH}/train_series.csv\")\ntest_series = pd.read_csv(f\"{DATA_PATH}/test_series.csv\")\n\nprint(f\"سلاسل التدريب: {len(train_series)}\")\nprint(f\"سلاسل الاختبار: {len(test_series)}\")\nprint(f\"\\nأعمدة train_series: {train_series.columns.tolist()}\")\nprint(f\"\\nعينة:\")\ntrain_series.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:08:54.72512Z","iopub.execute_input":"2026-08-06T22:08:54.725418Z","iopub.status.idle":"2026-08-06T22:08:54.816271Z","shell.execute_reply.started":"2026-08-06T22:08:54.725392Z","shell.execute_reply":"2026-08-06T22:08:54.81529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# البحث عن جميع ملفات DICOM\ndicom_files = glob(f\"{DATA_PATH}/train_series/**/*.dcm\", recursive=True)\nprint(f\"إجمالي ملفات DICOM: {len(dicom_files)}\")\n\n# عرض أول 3 مسارات\nprint(\"\\nأول 3 مسارات:\")\nfor f in dicom_files[:3]:\n    print(f\"  {f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:09:29.710288Z","iopub.execute_input":"2026-08-06T22:09:29.711161Z","iopub.status.idle":"2026-08-06T22:13:24.979297Z","shell.execute_reply.started":"2026-08-06T22:09:29.711126Z","shell.execute_reply":"2026-08-06T22:13:24.978494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# استخراج StudyInstanceUID من المسار\ndef extract_study_uid(path):\n    \"\"\"استخراج معرف الدراسة من مسار الملف\"\"\"\n    parts = path.split('/')\n    # عادة StudyInstanceUID هو المجلد قبل الأخير أو الأول\n    for part in parts:\n        if part.startswith('1.2.826'):\n            return part\n    return None\n\n# اختبار\nsample_path = dicom_files[0]\nuid = extract_study_uid(sample_path)\nprint(f\"مسار: ...{sample_path[-80:]}\")\nprint(f\"StudyInstanceUID: {uid}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:13:25.009754Z","iopub.execute_input":"2026-08-06T22:13:25.010396Z","iopub.status.idle":"2026-08-06T22:13:25.021777Z","shell.execute_reply.started":"2026-08-06T22:13:25.010346Z","shell.execute_reply":"2026-08-06T22:13:25.021014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# بناء قاموس: StudyInstanceUID -> قائمة مسارات الصور\nfrom collections import defaultdict\n\nstudy_to_images = defaultdict(list)\nfor dcm_path in dicom_files:\n    uid = extract_study_uid(dcm_path)\n    if uid:\n        study_to_images[uid].append(dcm_path)\n\nprint(f\"عدد الدراسات الفريدة: {len(study_to_images)}\")\n\n# كم صورة لكل دراسة؟\nimage_counts = [len(v) for v in study_to_images.values()]\nprint(f\"\\nإحصائيات الصور لكل دراسة:\")\nprint(f\"  المتوسط: {np.mean(image_counts):.1f}\")\nprint(f\"  الأقل: {np.min(image_counts)}\")\nprint(f\"  الأكثر: {np.max(image_counts)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:13:30.10399Z","iopub.execute_input":"2026-08-06T22:13:30.104767Z","iopub.status.idle":"2026-08-06T22:13:31.346572Z","shell.execute_reply.started":"2026-08-06T22:13:30.104725Z","shell.execute_reply":"2026-08-06T22:13:31.345535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# طريقة أسرع: استخدام os.walk مع حد أقصى\nimport os\n\ndicom_files = []\ncount = 0\nmax_files = 100  # نجمع أول 100 فقط للاختبار\n\nprint(\"البحث عن ملفات DICOM...\")\nfor root, dirs, files in os.walk(f\"{DATA_PATH}/train_series\"):\n    for f in files:\n        if f.endswith('.dcm'):\n            dicom_files.append(os.path.join(root, f))\n            count += 1\n    if count >= max_files:\n        break\n\nprint(f\"✅ تم العثور على {len(dicom_files)} ملف (توقف عند {max_files})\")\nprint(f\"\\nأول مسار:\")\nprint(dicom_files[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:13:32.811264Z","iopub.execute_input":"2026-08-06T22:13:32.811707Z","iopub.status.idle":"2026-08-06T22:13:32.8614Z","shell.execute_reply.started":"2026-08-06T22:13:32.811676Z","shell.execute_reply":"2026-08-06T22:13:32.860506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# جمع كل الملفات بدون حد أقصى\nimport os\n\ndicom_files = []\nprint(\"⏳ البحث عن جميع ملفات DICOM...\")\n\nfor root, dirs, files in os.walk(f\"{DATA_PATH}/train_series\"):\n    for f in files:\n        if f.endswith('.dcm'):\n            dicom_files.append(os.path.join(root, f))\n\nprint(f\"✅ إجمالي ملفات DICOM: {len(dicom_files)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:14:13.924528Z","iopub.execute_input":"2026-08-06T22:14:13.925177Z","iopub.status.idle":"2026-08-06T22:15:37.895731Z","shell.execute_reply.started":"2026-08-06T22:14:13.925144Z","shell.execute_reply":"2026-08-06T22:15:37.89477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# قراءة وعرض أول صورة\nds = pydicom.dcmread(dicom_files[0])\nimg = ds.pixel_array\n\nprint(f\"حجم الصورة: {img.shape}\")\nprint(f\"نوع البيانات: {img.dtype}\")\n\nplt.figure(figsize=(8, 8))\nplt.imshow(img, cmap='gray')\nplt.title(f\"صورة MRI - حجم: {img.shape}\")\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:15:38.232894Z","iopub.execute_input":"2026-08-06T22:15:38.233257Z","iopub.status.idle":"2026-08-06T22:15:38.484631Z","shell.execute_reply.started":"2026-08-06T22:15:38.233227Z","shell.execute_reply":"2026-08-06T22:15:38.483846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# كم مجلد دراسة؟\nstudy_folders = set()\nfor f in dicom_files:\n    # استخراج معرف الدراسة من المسار\n    parts = f.split('/')\n    for part in parts:\n        if part.startswith('1.2.826') and len(part) > 20:\n            study_folders.add(part)\n            break\n\nprint(f\"عدد الدراسات الفريدة: {len(study_folders)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:15:44.074603Z","iopub.execute_input":"2026-08-06T22:15:44.07502Z","iopub.status.idle":"2026-08-06T22:15:45.489128Z","shell.execute_reply.started":"2026-08-06T22:15:44.074942Z","shell.execute_reply":"2026-08-06T22:15:45.488152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import defaultdict\nimport re\n\n# بناء قاموس: StudyInstanceUID -> قائمة مسارات الصور\nstudy_images = defaultdict(list)\nfor dcm_path in dicom_files:\n    # البحث عن StudyInstanceUID في المسار\n    parts = dcm_path.split('/')\n    for i, part in enumerate(parts):\n        if part.startswith('1.2.826') and len(part) > 30:\n            study_images[part].append(dcm_path)\n            break\n\nprint(f\"الدراسات الفريدة: {len(study_images)}\")\n\n# المتوسط\ncounts = [len(v) for v in study_images.values()]\nprint(f\"متوسط الصور لكل دراسة: {np.mean(counts):.1f}\")\nprint(f\"الأقل: {np.min(counts)}, الأكثر: {np.max(counts)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:18:36.921404Z","iopub.execute_input":"2026-08-06T22:18:36.921864Z","iopub.status.idle":"2026-08-06T22:18:38.814014Z","shell.execute_reply.started":"2026-08-06T22:18:36.921831Z","shell.execute_reply":"2026-08-06T22:18:38.813197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_middle_slice(study_uid, target_size=(256, 256)):\n    \"\"\"استخراج الصورة الوسطى من دراسة\"\"\"\n    paths = study_images.get(study_uid, [])\n    if len(paths) == 0:\n        return np.zeros((*target_size, 3), dtype=np.float32)\n    \n    # اختيار الصورة الوسطى\n    middle_idx = len(paths) // 2\n    dcm_path = sorted(paths)[middle_idx]\n    \n    try:\n        ds = pydicom.dcmread(dcm_path)\n        img = ds.pixel_array.astype(np.float32)\n        \n        # تطبيع\n        img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n        \n        # تغيير الحجم\n        img = cv2.resize(img, target_size)\n        \n        # تحويل لـ 3 قنوات (لنماذج CNN المدربة مسبقاً)\n        img = np.stack([img, img, img], axis=-1)\n        \n        return img\n    except:\n        return np.zeros((*target_size, 3), dtype=np.float32)\n\n# اختبار\nsample_uid = train_labeled['StudyInstanceUID'].iloc[0]\nsample_img = get_middle_slice(sample_uid)\nprint(f\"✅ حجم الصورة: {sample_img.shape}\")\n\nplt.imshow(sample_img)\nplt.title(\"الصورة الوسطى من الدراسة\")\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:19:35.652931Z","iopub.execute_input":"2026-08-06T22:19:35.65358Z","iopub.status.idle":"2026-08-06T22:19:35.854383Z","shell.execute_reply.started":"2026-08-06T22:19:35.653544Z","shell.execute_reply":"2026-08-06T22:19:35.853401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"⏳ تجهيز صور التدريب...\")\ntrain_images = []\nfor uid in train_labeled['StudyInstanceUID']:\n    img = get_middle_slice(uid)\n    train_images.append(img)\n\ntrain_images = np.array(train_images)\nprint(f\"✅ حجم مصفوفة الصور: {train_images.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:20:09.803634Z","iopub.execute_input":"2026-08-06T22:20:09.804468Z","iopub.status.idle":"2026-08-06T22:20:11.277859Z","shell.execute_reply.started":"2026-08-06T22:20:09.804433Z","shell.execute_reply":"2026-08-06T22:20:11.277075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nfrom torch.utils.data import Dataset, DataLoader\n\n# تعريف Dataset\nclass KneeDataset(Dataset):\n    def __init__(self, images, text_embeddings, labels):\n        self.images = torch.tensor(images, dtype=torch.float32).permute(0, 3, 1, 2)  # (N, 3, H, W)\n        self.text_emb = torch.tensor(text_embeddings, dtype=torch.float32)\n        self.labels = torch.tensor(labels, dtype=torch.float32)\n    \n    def __len__(self):\n        return len(self.images)\n    \n    def __getitem__(self, idx):\n        return self.images[idx], self.text_emb[idx], self.labels[idx] \n\n# تجهيز البيانات\ndataset = KneeDataset(train_images, train_embeddings, train_labels)\nprint(f\"✅ حجم dataset: {len(dataset)}\")\n\n# تقسيم تدريب/تحقق\nfrom sklearn.model_selection import train_test_split\nindices = list(range(len(dataset)))\ntrain_idx, val_idx = train_test_split(indices, test_size=0.2, random_state=42)\n\ntrain_loader = DataLoader(dataset, batch_size=8, sampler=torch.utils.data.SubsetRandomSampler(train_idx))\nval_loader = DataLoader(dataset, batch_size=8, sampler=torch.utils.data.SubsetRandomSampler(val_idx))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:20:59.725506Z","iopub.execute_input":"2026-08-06T22:20:59.726017Z","iopub.status.idle":"2026-08-06T22:20:59.766193Z","shell.execute_reply.started":"2026-08-06T22:20:59.725936Z","shell.execute_reply":"2026-08-06T22:20:59.765233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# بناء النموذج المتعدد الوسائط\nclass MultimodalModel(nn.Module):\n    def __init__(self, num_classes=12):\n        super().__init__()\n        \n        # CNN للصور (EfficientNet صغير)\n        self.cnn = models.efficientnet_b0(weights='DEFAULT')\n        self.cnn.classifier = nn.Identity()  # إزالة آخر طبقة\n        \n        # دمج الصور + النص\n        self.fusion = nn.Sequential(\n            nn.Linear(1280 + 768, 512),  # 1280 من EfficientNet + 768 من BioBERT\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(512, 256),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            nn.Linear(256, num_classes)\n        )\n    \n    def forward(self, image, text_emb):\n        img_features = self.cnn(image)\n        combined = torch.cat([img_features, text_emb], dim=1)\n        output = self.fusion(combined)\n        return output\n\nmodel = MultimodalModel().to(device)\nprint(f\"✅ النموذج جاهز\")\nprint(f\"عدد المعلمات: {sum(p.numel() for p in model.parameters()):,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:21:35.545198Z","iopub.execute_input":"2026-08-06T22:21:35.54551Z","iopub.status.idle":"2026-08-06T22:21:35.701438Z","shell.execute_reply.started":"2026-08-06T22:21:35.545483Z","shell.execute_reply":"2026-08-06T22:21:35.700518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.optim as optim\nfrom sklearn.metrics import roc_auc_score\n\n# إعدادات التدريب\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.AdamW(model.parameters(), lr=1e-4, weight_decay=0.01)\nnum_epochs = 30\n\n# تدريب\nprint(\"=\"*60)\nprint(\"بدء التدريب...\")\nprint(\"=\"*60)\n\nbest_auc = 0\n\nfor epoch in range(num_epochs):\n    # طور التدريب\n    model.train()\n    train_loss = 0\n    for images, text_emb, labels in train_loader:\n        images = images.to(device)\n        text_emb = text_emb.to(device)\n        labels = labels.to(device)\n        \n        optimizer.zero_grad()\n        outputs = model(images, text_emb)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        train_loss += loss.item()\n    \n    # طور التحقق\n    model.eval()\n    all_preds = []\n    all_labels = []\n    with torch.no_grad():\n        for images, text_emb, labels in val_loader:\n            images = images.to(device)\n            text_emb = text_emb.to(device)\n            \n            outputs = model(images, text_emb)\n            preds = torch.sigmoid(outputs).cpu().numpy()\n            \n            all_preds.append(preds)\n            all_labels.append(labels.numpy())\n    \n    all_preds = np.vstack(all_preds)\n    all_labels = np.vstack(all_labels)\n    \n    # حساب AUC لكل فئة\n    aucs = []\n    for i in range(12):\n        if len(np.unique(all_labels[:, i])) > 1:\n            auc = roc_auc_score(all_labels[:, i], all_preds[:, i])\n            aucs.append(auc)\n    \n    avg_auc = np.mean(aucs) if aucs else 0\n    train_loss = train_loss / len(train_loader)\n    \n    print(f\"Epoch {epoch+1:2d}/{num_epochs} | Loss: {train_loss:.4f} | Val AUC: {avg_auc:.4f}\")\n    \n    # حفظ أفضل نموذج\n    if avg_auc > best_auc:\n        best_auc = avg_auc\n        torch.save(model.state_dict(), 'best_model.pth')\n\nprint(f\"\\n✅ أفضل AUC: {best_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:23:24.180213Z","iopub.execute_input":"2026-08-06T22:23:24.180687Z","iopub.status.idle":"2026-08-06T22:23:35.312904Z","shell.execute_reply.started":"2026-08-06T22:23:24.180653Z","shell.execute_reply":"2026-08-06T22:23:35.312154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# استخراج تسميات من جميع التقارير النصية\nfrom sklearn.linear_model import LogisticRegression\n\n# تدريب BioBERT classifier على الـ 58 دراسة\ntext_classifiers = {}\nfor col in target_cols:\n    y = train_labeled[col].values\n    clf = LogisticRegression(max_iter=1000, random_state=42)\n    clf.fit(train_embeddings, y)\n    text_classifiers[col] = clf\n\nprint(\"✅ تم تدريب المصنفات النصية\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:24:17.807191Z","iopub.execute_input":"2026-08-06T22:24:17.807699Z","iopub.status.idle":"2026-08-06T22:24:18.177872Z","shell.execute_reply.started":"2026-08-06T22:24:17.807665Z","shell.execute_reply":"2026-08-06T22:24:18.177056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# استخراج تضمينات لجميع التقارير (4407)\nprint(\"⏳ استخراج تضمينات لجميع التقارير...\")\nall_texts = train_df['Report'].tolist()\nall_text_embeddings = get_text_embeddings(all_texts)\nprint(f\"✅ حجم التضمينات: {all_text_embeddings.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:24:31.889311Z","iopub.execute_input":"2026-08-06T22:24:31.889761Z","iopub.status.idle":"2026-08-06T22:26:45.274949Z","shell.execute_reply.started":"2026-08-06T22:24:31.889731Z","shell.execute_reply":"2026-08-06T22:26:45.274165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# توليد تسميات تلقائية (Pseudo-labels) لجميع الدراسات\npseudo_labels = np.zeros((len(all_texts), 12))\n\nfor i, col in enumerate(target_cols):\n    probs = text_classifiers[col].predict_proba(all_text_embeddings)[:, 1]\n    # تعيين 1 للاحتمالات العالية، 0 للمنخفضة\n    pseudo_labels[:, i] = (probs > 0.7).astype(float)  # ثقة عالية فقط\n\nprint(f\"✅ التسميات التلقائية جاهزة: {pseudo_labels.shape}\")\nprint(f\"عدد الحالات الإيجابية لكل فئة:\")\nfor i, col in enumerate(target_cols):\n    print(f\"  {col}: {int(pseudo_labels[:, i].sum())}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:26:45.571733Z","iopub.execute_input":"2026-08-06T22:26:45.57215Z","iopub.status.idle":"2026-08-06T22:26:45.661227Z","shell.execute_reply.started":"2026-08-06T22:26:45.572115Z","shell.execute_reply":"2026-08-06T22:26:45.660419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# تجهيز صور لعدد أكبر من الدراسات (نستخدم 500 دراسة للسرعة)\nimport random\n\n# اختيار 500 دراسة عشوائية + الـ 58 الأصلية\nall_uids = list(study_images.keys())\nrandom.seed(42)\nrandom.shuffle(all_uids)\n\n# نبدأ بـ 500 دراسة\nselected_uids = all_uids[:500]\nprint(f\"عدد الدراسات المختارة: {len(selected_uids)}\")\n\n# تجهيز الصور\nprint(\"⏳ تجهيز الصور...\")\nselected_images = []\nfor uid in selected_uids:\n    img = get_middle_slice(uid)\n    selected_images.append(img)\n\nselected_images = np.array(selected_images)\nprint(f\"✅ حجم الصور: {selected_images.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:28:12.839222Z","iopub.execute_input":"2026-08-06T22:28:12.839855Z","iopub.status.idle":"2026-08-06T22:28:19.982809Z","shell.execute_reply.started":"2026-08-06T22:28:12.839824Z","shell.execute_reply":"2026-08-06T22:28:19.982025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# مطابقة التسميات مع الدراسات المختارة\ntrain_df_indexed = train_df.set_index('StudyInstanceUID')\nselected_labels = []\n\nfor uid in selected_uids:\n    if uid in train_df_indexed.index:\n        idx = train_df_indexed.index.get_loc(uid)\n        selected_labels.append(pseudo_labels[idx])\n    else:\n        selected_labels.append(np.zeros(12))\n\nselected_labels = np.array(selected_labels)\nprint(f\"✅ حجم التسميات: {selected_labels.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:28:26.678752Z","iopub.execute_input":"2026-08-06T22:28:26.679439Z","iopub.status.idle":"2026-08-06T22:28:26.689705Z","shell.execute_reply.started":"2026-08-06T22:28:26.679402Z","shell.execute_reply":"2026-08-06T22:28:26.688663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataset للصور فقط\nclass ImageDataset(Dataset):\n    def __init__(self, images, labels):\n        self.images = torch.tensor(images, dtype=torch.float32).permute(0, 3, 1, 2)\n        self.labels = torch.tensor(labels, dtype=torch.float32)\n    \n    def __len__(self):\n        return len(self.images)\n    \n    def __getitem__(self, idx):\n        return self.images[idx], self.labels[idx]\n\n# تقسيم البيانات\ndataset = ImageDataset(selected_images, selected_labels)\ntrain_idx, val_idx = train_test_split(range(len(dataset)), test_size=0.2, random_state=42)\n\ntrain_loader = DataLoader(dataset, batch_size=16, sampler=torch.utils.data.SubsetRandomSampler(train_idx))\nval_loader = DataLoader(dataset, batch_size=16, sampler=torch.utils.data.SubsetRandomSampler(val_idx))\n\nprint(f\"تدريب: {len(train_idx)}, تحقق: {len(val_idx)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:28:42.482748Z","iopub.execute_input":"2026-08-06T22:28:42.483895Z","iopub.status.idle":"2026-08-06T22:28:42.656814Z","shell.execute_reply.started":"2026-08-06T22:28:42.483849Z","shell.execute_reply":"2026-08-06T22:28:42.655939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# نموذج CNN فقط\ncnn_model = models.efficientnet_b0(weights='DEFAULT')\ncnn_model.classifier = nn.Sequential(\n    nn.Dropout(0.3),\n    nn.Linear(1280, 256),\n    nn.ReLU(),\n    nn.Dropout(0.2),\n    nn.Linear(256, 12)\n)\ncnn_model = cnn_model.to(device)\n\noptimizer = optim.AdamW(cnn_model.parameters(), lr=1e-4)\ncriterion = nn.BCEWithLogitsLoss()\n\nprint(f\"عدد المعلمات: {sum(p.numel() for p in cnn_model.parameters()):,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:30:18.121215Z","iopub.execute_input":"2026-08-06T22:30:18.121706Z","iopub.status.idle":"2026-08-06T22:30:18.280741Z","shell.execute_reply.started":"2026-08-06T22:30:18.121671Z","shell.execute_reply":"2026-08-06T22:30:18.279694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=\"*60)\nprint(\"تدريب CNN على 500 دراسة...\")\nprint(\"=\"*60)\n\nbest_auc = 0\n\nfor epoch in range(20):\n    cnn_model.train()\n    train_loss = 0\n    for images, labels in train_loader:\n        images = images.to(device)\n        labels = labels.to(device)\n        \n        optimizer.zero_grad()\n        outputs = cnn_model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        train_loss += loss.item()\n    \n    # تحقق\n    cnn_model.eval()\n    all_preds, all_labels = [], []\n    with torch.no_grad():\n        for images, labels in val_loader:\n            images = images.to(device)\n            outputs = cnn_model(images)\n            preds = torch.sigmoid(outputs).cpu().numpy()\n            all_preds.append(preds)\n            all_labels.append(labels.numpy())\n    \n    all_preds = np.vstack(all_preds)\n    all_labels = np.vstack(all_labels)\n    \n    aucs = []\n    for i in range(12):\n        if len(np.unique(all_labels[:, i])) > 1:\n            aucs.append(roc_auc_score(all_labels[:, i], all_preds[:, i]))\n    \n    avg_auc = np.mean(aucs)\n    print(f\"Epoch {epoch+1:2d}/20 | Loss: {train_loss/len(train_loader):.4f} | AUC: {avg_auc:.4f}\")\n    \n    if avg_auc > best_auc:\n        best_auc = avg_auc\n        torch.save(cnn_model.state_dict(), 'cnn_best.pth')\n\nprint(f\"\\n✅ أفضل AUC: {best_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:31:23.722111Z","iopub.execute_input":"2026-08-06T22:31:23.7226Z","iopub.status.idle":"2026-08-06T22:32:18.611019Z","shell.execute_reply.started":"2026-08-06T22:31:23.722567Z","shell.execute_reply":"2026-08-06T22:32:18.610135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# تحميل أفضل نموذج\ncnn_model.load_state_dict(torch.load('cnn_best.pth'))\ncnn_model.eval()\n\n# تجهيز صور الاختبار (4349 دراسة)\ntest_uids = test_data['StudyInstanceUID'].tolist()\nprint(f\"⏳ تجهيز صور الاختبار ({len(test_uids)} دراسة)...\")\n\ntest_images = []\nfor uid in test_uids:\n    img = get_middle_slice(uid)\n    test_images.append(img)\n\ntest_images = np.array(test_images)\nprint(f\"✅ حجم صور الاختبار: {test_images.shape}\")\n\n# تحويل إلى tensor\ntest_tensor = torch.tensor(test_images, dtype=torch.float32).permute(0, 3, 1, 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:46:31.428525Z","iopub.execute_input":"2026-08-06T22:46:31.428952Z","iopub.status.idle":"2026-08-06T22:46:53.350363Z","shell.execute_reply.started":"2026-08-06T22:46:31.428921Z","shell.execute_reply":"2026-08-06T22:46:53.349345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# عمل التوقعات\nprint(\"⏳ عمل التوقعات...\")\nbatch_size = 32\nall_predictions = []\n\nwith torch.no_grad():\n    for i in range(0, len(test_tensor), batch_size):\n        batch = test_tensor[i:i+batch_size].to(device)\n        outputs = cnn_model(batch)\n        preds = torch.sigmoid(outputs).cpu().numpy()\n        all_predictions.append(preds)\n\nall_predictions = np.vstack(all_predictions)\nprint(f\"✅ التوقعات جاهزة: {all_predictions.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:46:53.351865Z","iopub.execute_input":"2026-08-06T22:46:53.352232Z","iopub.status.idle":"2026-08-06T22:47:08.514789Z","shell.execute_reply.started":"2026-08-06T22:46:53.352204Z","shell.execute_reply":"2026-08-06T22:47:08.514057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# إنشاء ملف التقديم\nsubmission = pd.DataFrame({'StudyInstanceUID': test_uids})\nfor i, col in enumerate(target_cols):\n    submission[col] = all_predictions[:, i]\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"✅ تم حفظ submission.csv\")\n\n# عرض عينة\nprint(f\"\\nعدد الصفوف: {len(submission)}\")\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:47:08.616412Z","iopub.execute_input":"2026-08-06T22:47:08.616741Z","iopub.status.idle":"2026-08-06T22:47:08.708359Z","shell.execute_reply.started":"2026-08-06T22:47:08.616716Z","shell.execute_reply":"2026-08-06T22:47:08.70753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nif os.path.exists('submission.csv'):\n    print(\"✅ submission.csv موجود\")\n    print(f\"الحجم: {os.path.getsize('submission.csv')} بايت\")\nelse:\n    print(\"❌ الملف غير موجود!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T22:49:05.403099Z","iopub.execute_input":"2026-08-06T22:49:05.403859Z","iopub.status.idle":"2026-08-06T22:49:05.409458Z","shell.execute_reply.started":"2026-08-06T22:49:05.403819Z","shell.execute_reply":"2026-08-06T22:49:05.408605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# إنشاء ملف submission بسيط\ntest_df = pd.read_csv(\"/kaggle/input/competitions/rsna-knee-abnormality-detection/test.csv\")\n\ntarget_cols = ['ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', \n               'Medial OA', 'Lateral OA', 'PF OA', 'Effusion', \n               'Synovitis', \"Baker's\", 'Contusion', 'Fracture']\n\n# توقعات عشوائية للتجربة\nsubmission = pd.DataFrame({'StudyInstanceUID': test_df['StudyInstanceUID']})\nfor col in target_cols:\n    submission[col] = np.random.uniform(0, 1, len(submission))\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"✅ تم إنشاء submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:01:02.004831Z","iopub.execute_input":"2026-08-06T23:01:02.005169Z","iopub.status.idle":"2026-08-06T23:01:02.033692Z","shell.execute_reply.started":"2026-08-06T23:01:02.005139Z","shell.execute_reply":"2026-08-06T23:01:02.033053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\ntest_df = pd.read_csv(\"/kaggle/input/competitions/rsna-knee-abnormality-detection/test.csv\")\n\ntarget_cols = ['ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', \n               'Medial OA', 'Lateral OA', 'PF OA', 'Effusion', \n               'Synovitis', \"Baker's\", 'Contusion', 'Fracture']\n\nsubmission = pd.DataFrame({'StudyInstanceUID': test_df['StudyInstanceUID']})\nfor col in target_cols:\n    submission[col] = np.random.uniform(0, 1, len(submission))\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"✅ تم إنشاء submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:04:17.177544Z","iopub.execute_input":"2026-08-06T23:04:17.177875Z","iopub.status.idle":"2026-08-06T23:04:17.194664Z","shell.execute_reply.started":"2026-08-06T23:04:17.177844Z","shell.execute_reply":"2026-08-06T23:04:17.193674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nif os.path.exists('submission.csv'):\n    print(\"✅ الملف موجود وجاهز للتقديم\")\nelse:\n    print(\"❌ خطأ!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:04:32.970498Z","iopub.execute_input":"2026-08-06T23:04:32.970788Z","iopub.status.idle":"2026-08-06T23:04:32.976565Z","shell.execute_reply.started":"2026-08-06T23:04:32.970763Z","shell.execute_reply":"2026-08-06T23:04:32.975272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle competitions submit -c rsna-knee-abnormality-detection -f submission.csv -m \"First submission\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:05:36.304156Z","iopub.execute_input":"2026-08-06T23:05:36.304762Z","iopub.status.idle":"2026-08-06T23:05:38.584598Z","shell.execute_reply.started":"2026-08-06T23:05:36.304729Z","shell.execute_reply":"2026-08-06T23:05:38.583772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle competitions submit rsna-knee-abnormality-detection -f submission.csv -m \"submission\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:07:48.737374Z","iopub.execute_input":"2026-08-06T23:07:48.737876Z","iopub.status.idle":"2026-08-06T23:07:50.337733Z","shell.execute_reply.started":"2026-08-06T23:07:48.737832Z","shell.execute_reply":"2026-08-06T23:07:50.336831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n!kaggle competitions submit rsna-knee-abnormality-detection -f {os.getcwd()}/submission.csv -m \"submission\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:08:11.368797Z","iopub.execute_input":"2026-08-06T23:08:11.370074Z","iopub.status.idle":"2026-08-06T23:08:12.955272Z","shell.execute_reply.started":"2026-08-06T23:08:11.370015Z","shell.execute_reply":"2026-08-06T23:08:12.954364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import requests\nimport os\n\nurl = \"https://www.kaggle.com/api/v1/competitions/rsna-knee-abnormality-detection/submit\"\nheaders = {\"Authorization\": f\"Bearer {os.environ['KAGGLE_KEY']}\"}\nfiles = {\"file\": open(\"submission.csv\", \"rb\")}\n\nresponse = requests.post(url, headers=headers, files=files)\nprint(response.status_code)\nprint(response.text)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:08:45.63772Z","iopub.execute_input":"2026-08-06T23:08:45.638622Z","iopub.status.idle":"2026-08-06T23:08:45.696004Z","shell.execute_reply.started":"2026-08-06T23:08:45.638575Z","shell.execute_reply":"2026-08-06T23:08:45.694745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink, display\n\nprint(\"=\"*50)\nprint(\"👇 اضغط على الرابط الأزرق أدناه للتحميل\")\nprint(\"=\"*50)\ndisplay(FileLink('submission.csv'))\nprint(\"=\"*50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:11:58.582842Z","iopub.execute_input":"2026-08-06T23:11:58.583336Z","iopub.status.idle":"2026-08-06T23:11:58.5918Z","shell.execute_reply.started":"2026-08-06T23:11:58.583302Z","shell.execute_reply":"2026-08-06T23:11:58.591121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\n# نسخ الملف إلى مجلد الإخراج\nshutil.copy('submission.csv', '/kaggle/working/submission.csv')\nprint(\"✅ تم نسخ الملف إلى /kaggle/working/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:13:51.298524Z","iopub.execute_input":"2026-08-06T23:13:51.29902Z","iopub.status.idle":"2026-08-06T23:13:51.309872Z","shell.execute_reply.started":"2026-08-06T23:13:51.298942Z","shell.execute_reply":"2026-08-06T23:13:51.308512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import webbrowser\nwebbrowser.open('https://www.kaggle.com/code/MHMDA81/notebook8a7ef84430')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:18:25.613204Z","iopub.execute_input":"2026-08-06T23:18:25.613651Z","iopub.status.idle":"2026-08-06T23:18:25.632322Z","shell.execute_reply.started":"2026-08-06T23:18:25.613618Z","shell.execute_reply":"2026-08-06T23:18:25.631325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(\"اسم الـ Notebook الحالي:\", os.path.basename(os.getcwd()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:19:15.65192Z","iopub.execute_input":"2026-08-06T23:19:15.652349Z","iopub.status.idle":"2026-08-06T23:19:15.657004Z","shell.execute_reply.started":"2026-08-06T23:19:15.652318Z","shell.execute_reply":"2026-08-06T23:19:15.656079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# الملفات في مجلد العمل الحالي\nprint(\"=== الملفات في /kaggle/working/ ===\")\nfiles = os.listdir('/kaggle/working/')\nfor f in files:\n    size = os.path.getsize(f'/kaggle/working/{f}')\n    print(f\"  {f} - {size:,} بايت\")\n\nprint(f\"\\n✅ عدد الملفات: {len(files)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:20:24.525518Z","iopub.execute_input":"2026-08-06T23:20:24.526027Z","iopub.status.idle":"2026-08-06T23:20:24.53346Z","shell.execute_reply.started":"2026-08-06T23:20:24.52595Z","shell.execute_reply":"2026-08-06T23:20:24.532781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom collections import defaultdict\nfrom glob import glob\nimport os\n\ndevice = torch.device(\"cuda\")\nprint(f\"✅ GPU: {torch.cuda.get_device_name(0)}\")\n\n# تحميل النموذج\ncnn_model = models.efficientnet_b0(weights=None)\ncnn_model.classifier = nn.Sequential(\n    nn.Dropout(0.3),\n    nn.Linear(1280, 256),\n    nn.ReLU(),\n    nn.Dropout(0.2),\n    nn.Linear(256, 12)\n)\ncnn_model.load_state_dict(torch.load('cnn_best.pth'))\ncnn_model = cnn_model.to(device)\ncnn_model.eval()\nprint(\"✅ النموذج المحمل\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:21:10.292911Z","iopub.execute_input":"2026-08-06T23:21:10.293677Z","iopub.status.idle":"2026-08-06T23:21:10.529837Z","shell.execute_reply.started":"2026-08-06T23:21:10.293641Z","shell.execute_reply":"2026-08-06T23:21:10.529146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# تحميل بيانات الاختبار\nDATA_PATH = \"/kaggle/input/competitions/rsna-knee-abnormality-detection\"\ntrain_df = pd.read_csv(f\"{DATA_PATH}/train.csv\")\ntest_data = train_df[train_df['ACL'].isna()].copy()\ntest_uids = test_data['StudyInstanceUID'].tolist()\n\ntarget_cols = ['ACL', 'MCL', 'Medial Meniscus', 'Lateral Meniscus', \n               'Medial OA', 'Lateral OA', 'PF OA', 'Effusion', \n               'Synovitis', \"Baker's\", 'Contusion', 'Fracture']\n\nprint(f\"عدد دراسات الاختبار: {len(test_uids)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:21:55.852489Z","iopub.execute_input":"2026-08-06T23:21:55.853205Z","iopub.status.idle":"2026-08-06T23:21:55.961233Z","shell.execute_reply.started":"2026-08-06T23:21:55.853165Z","shell.execute_reply":"2026-08-06T23:21:55.96006Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# بناء قاموس الصور\nprint(\"⏳ بناء قاموس الصور...\")\ndicom_files = []\nfor root, dirs, files in os.walk(f\"{DATA_PATH}/train_series\"):\n    for f in files:\n        if f.endswith('.dcm'):\n            dicom_files.append(os.path.join(root, f))\n\nstudy_images = defaultdict(list)\nfor dcm_path in dicom_files:\n    parts = dcm_path.split('/')\n    for part in parts:\n        if part.startswith('1.2.826') and len(part) > 30:\n            study_images[part].append(dcm_path)\n            break\n\nprint(f\"✅ {len(study_images)} دراسة\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:22:19.527319Z","iopub.execute_input":"2026-08-06T23:22:19.527783Z","iopub.status.idle":"2026-08-06T23:23:41.225268Z","shell.execute_reply.started":"2026-08-06T23:22:19.527752Z","shell.execute_reply":"2026-08-06T23:23:41.224238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# استخراج الصور\ndef get_middle_slice(study_uid, target_size=(256, 256)):\n    paths = study_images.get(study_uid, [])\n    if len(paths) == 0:\n        return np.zeros((*target_size, 3), dtype=np.float32)\n    middle_idx = len(paths) // 2\n    dcm_path = sorted(paths)[middle_idx]\n    try:\n        ds = pydicom.dcmread(dcm_path)\n        img = ds.pixel_array.astype(np.float32)\n        img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n        img = cv2.resize(img, target_size)\n        img = np.stack([img, img, img], axis=-1)\n        return img\n    except:\n        return np.zeros((*target_size, 3), dtype=np.float32)\n\n# تجهيز الصور (نأخذ أول 500 فقط للسرعة)\nprint(\"⏳ تجهيز صور الاختبار...\")\ntest_images = []\nfor uid in test_uids[:500]:\n    img = get_middle_slice(uid)\n    test_images.append(img)\n\ntest_images = np.array(test_images)\ntest_tensor = torch.tensor(test_images, dtype=torch.float32).permute(0, 3, 1, 2)\nprint(f\"✅ {test_tensor.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:24:33.844447Z","iopub.execute_input":"2026-08-06T23:24:33.844895Z","iopub.status.idle":"2026-08-06T23:24:36.767147Z","shell.execute_reply.started":"2026-08-06T23:24:33.844861Z","shell.execute_reply":"2026-08-06T23:24:36.766227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# عمل التوقعات\nprint(\"⏳ عمل التوقعات...\")\nall_preds = []\nbatch_size = 32\n\nwith torch.no_grad():\n    for i in range(0, len(test_tensor), batch_size):\n        batch = test_tensor[i:i+batch_size].to(device)\n        outputs = cnn_model(batch)\n        preds = torch.sigmoid(outputs).cpu().numpy()\n        all_preds.append(preds)\n\nall_preds = np.vstack(all_preds)\n\n# إنشاء ملف التقديم (لـ 500 دراسة + الباقي قيم افتراضية)\nsubmission = pd.DataFrame({'StudyInstanceUID': test_uids})\nfor i, col in enumerate(target_cols):\n    vals = np.zeros(len(test_uids))\n    vals[:500] = all_preds[:, i]\n    submission[col] = vals\n\nsubmission.to_csv('submission.csv', index=False)\nprint(f\"✅ تم حفظ submission.csv - {len(submission)} صف\")\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:25:15.373422Z","iopub.execute_input":"2026-08-06T23:25:15.373722Z","iopub.status.idle":"2026-08-06T23:25:20.365558Z","shell.execute_reply.started":"2026-08-06T23:25:15.373696Z","shell.execute_reply":"2026-08-06T23:25:20.364805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle competitions submit rsna-knee-abnormality-detection -f submission.csv -m \"CNN model AUC 0.69\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-06T23:26:27.524398Z","iopub.execute_input":"2026-08-06T23:26:27.525069Z","iopub.status.idle":"2026-08-06T23:26:29.09011Z","shell.execute_reply.started":"2026-08-06T23:26:27.525033Z","shell.execute_reply":"2026-08-06T23:26:29.089088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle competitions submit rsna-knee-abnormality-detection -f submission.csv -m \"CNN baseline\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle competitions submit rsna-knee-abnormality-detection -f submission.csv -m \"CNN baseline\"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}