{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport h5py\nimport torch\nimport numpy as np\nfrom transformers import AutoModel\nfrom tqdm.auto import tqdm\n\n# 1. إعداد الـ GPU والسرعة\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ntorch.backends.cudnn.benchmark = True\nprint(f\"🚀 Running Role 3 Pipeline on: {device}\")\n\n# 2. تحديد مسار الموديل أوفلاين والـ Output Directory\nMODEL_DIR = '/kaggle/input/dinov2/pytorch/base/1'\nif not os.path.exists(MODEL_DIR):\n    model_bins = glob.glob('/kaggle/input/**/pytorch_model.bin', recursive=True)\n    if model_bins:\n        MODEL_DIR = os.path.dirname(model_bins[0])\n    else:\n        raise FileNotFoundError(\"❌ لم يتم العثور على أوزان موديل DINOv2 في مجلدات الـ Input!\")\n\n# الحفظ مباشرة في /kaggle/working لضمان تصدير الملفات في الـ Output\nOUTPUT_DIR = '/kaggle/working'\nBATCH_SIZE = 32\n\nprint(f\"Loading DINOv2 offline from: {MODEL_DIR}\")\nextractor = AutoModel.from_pretrained(MODEL_DIR).to(device).eval()\n\n\ndef extract_features_batched(model, imgs_tensor, batch_size=32):\n    num_samples = len(imgs_tensor)\n    features_list = []\n    \n    with torch.inference_mode(), torch.amp.autocast('cuda'):\n        for i in range(0, num_samples, batch_size):\n            batch = imgs_tensor[i:i + batch_size]\n            outputs = model(pixel_values=batch)\n            batch_feats = outputs.last_hidden_state[:, 0, :].cpu().float().numpy()\n            features_list.append(batch_feats)\n            \n    return np.concatenate(features_list, axis=0)\n\n\ndef process_h5_node_fast(in_node, out_node):\n    for key in in_node.keys():\n        item = in_node[key]\n        \n        if isinstance(item, h5py.Dataset):\n            imgs = item[:]\n            \n            if imgs.ndim == 3:\n                imgs = np.stack([imgs] * 3, axis=1)\n            elif imgs.ndim == 4 and imgs.shape[-1] == 3:\n                imgs = np.transpose(imgs, (0, 3, 1, 2))\n                \n            imgs_tensor = torch.tensor(imgs, dtype=torch.float32).to(device)\n            if imgs_tensor.max() > 1.0:\n                imgs_tensor = imgs_tensor / 255.0\n                \n            feats = extract_features_batched(extractor, imgs_tensor, batch_size=BATCH_SIZE)\n            out_node.create_dataset(key, data=feats)\n            \n        elif isinstance(item, h5py.Group):\n            sub_out_group = out_node.create_group(key)\n            process_h5_node_fast(item, sub_out_group)\n\n\ndef main():\n    # البحث الشامل عن أي ملف h5 مرفوع في مجلدات الـ Input\n    input_h5_files = glob.glob('/kaggle/input/**/*.h5', recursive=True)\n    \n    print(f\"📦 Found {len(input_h5_files)} HDF5 shard files to process.\")\n    \n    # حماية صريحة: توقيف الكود فوراً ورؤية الخطأ إذا لم يجد ملفات\n    if len(input_h5_files) == 0:\n        raise FileNotFoundError(\"❌ لم يتم العثور على أي ملفات .h5 في الـ Input! تأكد من إضافة Shards من القائمة على اليمين.\")\n\n    for input_h5_path in input_h5_files:\n        shard_name = os.path.basename(input_h5_path)\n        output_h5_path = os.path.join(OUTPUT_DIR, f\"embeddings_{shard_name}\")\n        \n        print(f\"\\n⚡ Processing Shard: {shard_name} ...\")\n        \n        with h5py.File(input_h5_path, 'r') as in_h5, h5py.File(output_h5_path, 'w') as out_h5:\n            for study_id in tqdm(in_h5.keys(), desc=f\"Extracting {shard_name}\"):\n                study_group = in_h5[study_id]\n                out_study_group = out_h5.create_group(study_id)\n                process_h5_node_fast(study_group, out_study_group)\n\n        print(f\"✅ Finished: {output_h5_path}\")\n\n    print(\"\\n🎉 ALL SHARDS PROCESSED SUCCESSFULLY! Ready for Role 4.\")\n\nif __name__ == '__main__':\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}