{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":16320058},{"sourceType":"datasetVersion","sourceId":15083000,"datasetId":9656804,"databundleVersionId":15966452},{"sourceType":"datasetVersion","sourceId":14766723,"datasetId":9438849,"databundleVersionId":15618444},{"sourceType":"datasetVersion","sourceId":15530710,"datasetId":9936281,"databundleVersionId":16458371},{"sourceType":"datasetVersion","sourceId":15547929,"datasetId":9947354,"databundleVersionId":16476993},{"sourceType":"datasetVersion","sourceId":15355758,"datasetId":9789499,"databundleVersionId":16265986},{"sourceType":"datasetVersion","sourceId":15531080,"datasetId":9936502,"databundleVersionId":16458773},{"sourceType":"datasetVersion","sourceId":10855324,"datasetId":6742586,"databundleVersionId":11219268},{"sourceType":"modelInstanceVersion","sourceId":311741,"databundleVersionId":11641144,"modelInstanceId":264400,"modelId":285488},{"sourceType":"kernelVersion","sourceId":292115284},{"sourceType":"kernelVersion","sourceId":296416998},{"sourceType":"kernelVersion","sourceId":303447972},{"sourceType":"kernelVersion","sourceId":303451135},{"sourceType":"kernelVersion","sourceId":305681400}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nimport shutil\nsys.version","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:17.734551Z","iopub.execute_input":"2026-04-05T19:58:17.734768Z","iopub.status.idle":"2026-04-05T19:58:17.743133Z","shell.execute_reply.started":"2026-04-05T19:58:17.734745Z","shell.execute_reply":"2026-04-05T19:58:17.742233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.environ['PYTORCH_CUDA_ALLOC_CONF'] = \"expandable_segments:True\"\nos.environ['MAX_SEQ_LEN']='448'\nos.environ['OVERLAP_LEN']='96'\nos.environ['SEED']='7557'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:17.744377Z","iopub.execute_input":"2026-04-05T19:58:17.744999Z","iopub.status.idle":"2026-04-05T19:58:17.755458Z","shell.execute_reply.started":"2026-04-05T19:58:17.744955Z","shell.execute_reply":"2026-04-05T19:58:17.754895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shutil.copy2('/kaggle/input/usalign/USalign', '/kaggle/working/USalign')\nos.chmod('/kaggle/working/USalign', 0o755)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:17.756248Z","iopub.execute_input":"2026-04-05T19:58:17.756504Z","iopub.status.idle":"2026-04-05T19:58:17.800263Z","shell.execute_reply.started":"2026-04-05T19:58:17.756475Z","shell.execute_reply":"2026-04-05T19:58:17.799496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir ~/python-packages\n!cp -r /kaggle/input/datasets/stefanstefanov/srna3d2-protenix ~/python-packages/srna3d2-protenix\n!cp -r /kaggle/input/datasets/stefanstefanov/srna3d2-rnapro-tbm ~/python-packages/srna3d2-rnapro-tbm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:17.801235Z","iopub.execute_input":"2026-04-05T19:58:17.801536Z","iopub.status.idle":"2026-04-05T19:58:20.959499Z","shell.execute_reply.started":"2026-04-05T19:58:17.801505Z","shell.execute_reply":"2026-04-05T19:58:20.958705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for local_package_name in ['srna3d2-protenix', 'srna3d2-rnapro-tbm']:\n    local_package_path = os.path.expanduser(f'~/python-packages/{local_package_name}')\n    os.environ['PYTHONPATH'] = os.environ.get('PYTHONPATH', '') + os.pathsep + local_package_path\n    sys.path.append(local_package_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:20.960709Z","iopub.execute_input":"2026-04-05T19:58:20.960934Z","iopub.status.idle":"2026-04-05T19:58:20.966027Z","shell.execute_reply.started":"2026-04-05T19:58:20.960906Z","shell.execute_reply":"2026-04-05T19:58:20.965323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IS_SUBMISSION = os.getenv('KAGGLE_IS_COMPETITION_RERUN') is not None\nprint(f'{IS_SUBMISSION=}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:20.967114Z","iopub.execute_input":"2026-04-05T19:58:20.967619Z","iopub.status.idle":"2026-04-05T19:58:20.989333Z","shell.execute_reply.started":"2026-04-05T19:58:20.967583Z","shell.execute_reply":"2026-04-05T19:58:20.988675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport subprocess\nimport time\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:20.991754Z","iopub.execute_input":"2026-04-05T19:58:20.992071Z","iopub.status.idle":"2026-04-05T19:58:22.148212Z","shell.execute_reply.started":"2026-04-05T19:58:20.992051Z","shell.execute_reply":"2026-04-05T19:58:22.147379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from srna3d.chains import fill_non_rna_sequences\nfrom srna3d.tm_score import tm_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:22.149213Z","iopub.execute_input":"2026-04-05T19:58:22.149678Z","iopub.status.idle":"2026-04-05T19:58:22.16294Z","shell.execute_reply.started":"2026-04-05T19:58:22.149653Z","shell.execute_reply":"2026-04-05T19:58:22.162293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_seqs_df = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-2/test_sequences.csv')\nif not IS_SUBMISSION:\n    # test_seqs_df = test_seqs_df\n    test_seqs_df = test_seqs_df.loc[test_seqs_df['target_id'].isin({'9RVP','8ZNQ'})]\n    \ntest_seqs_df.to_csv('/kaggle/working/test_sequences.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2026-04-05T19:58:22.163895Z","iopub.execute_input":"2026-04-05T19:58:22.164289Z","iopub.status.idle":"2026-04-05T19:58:22.235276Z","shell.execute_reply.started":"2026-04-05T19:58:22.164267Z","shell.execute_reply":"2026-04-05T19:58:22.234785Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_seqs = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-2/train_sequences.csv')\ntrain_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-2/train_labels.csv', low_memory=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:22.236142Z","iopub.execute_input":"2026-04-05T19:58:22.236334Z","iopub.status.idle":"2026-04-05T19:58:35.407889Z","shell.execute_reply.started":"2026-04-05T19:58:22.236315Z","shell.execute_reply":"2026-04-05T19:58:35.407313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"validation_seqs = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-2/validation_sequences.csv')\nvalidation_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-2/validation_labels.csv', low_memory=False)\n\nvalidation_labels = validation_labels.loc[:, train_labels.columns]\n\ntrain_seqs = pd.concat([train_seqs, validation_seqs],    ignore_index=True)\ntrain_labels = pd.concat([train_labels, validation_labels],  ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2026-04-05T19:58:35.408766Z","iopub.execute_input":"2026-04-05T19:58:35.40903Z","iopub.status.idle":"2026-04-05T19:58:36.185996Z","shell.execute_reply.started":"2026-04-05T19:58:35.409006Z","shell.execute_reply":"2026-04-05T19:58:36.185444Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fill_non_rna_sequences(test_seqs_df)\nfill_non_rna_sequences(train_seqs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:36.186833Z","iopub.execute_input":"2026-04-05T19:58:36.187106Z","iopub.status.idle":"2026-04-05T19:58:36.833968Z","shell.execute_reply.started":"2026-04-05T19:58:36.187083Z","shell.execute_reply":"2026-04-05T19:58:36.833404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'{test_seqs_df.shape=}')\ntest_seqs_df.to_csv('/kaggle/working/test_sequences.csv', index=False)\n\nprint(f'{train_seqs.shape=} {train_labels.shape=}')\ntrain_seqs.to_csv('/kaggle/working/train_sequences.csv', index=False)\ntrain_labels.to_csv('/kaggle/working/train_labels.csv', index=False)\n\ndel test_seqs_df, train_seqs, train_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:58:36.834789Z","iopub.execute_input":"2026-04-05T19:58:36.835061Z","iopub.status.idle":"2026-04-05T19:59:05.688707Z","shell.execute_reply.started":"2026-04-05T19:58:36.835033Z","shell.execute_reply":"2026-04-05T19:59:05.688063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python -m srna3d.scripts.create_text_embeddings \\\n    --sequences_csv_path /kaggle/working/test_sequences.csv \\\n    --output_embeddings_path /kaggle/working/test_text_embeddings.safetensors \\\n    --model_name_or_path /kaggle/input/datasets/stefanstefanov/neuml-pubmedbert-base-model \\\n    --batch_size 32 \\\n    --max_length 512","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T19:59:05.68964Z","iopub.execute_input":"2026-04-05T19:59:05.689942Z","iopub.status.idle":"2026-04-05T20:00:05.294349Z","shell.execute_reply.started":"2026-04-05T19:59:05.689901Z","shell.execute_reply":"2026-04-05T20:00:05.293514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python -m srna3d.scripts.create_tbm_submission \\\n    --submission_path submission_tbm.csv \\\n    --sequences_csv_path /kaggle/working/test_sequences.csv \\\n    --train_sequences_csv_path /kaggle/working/train_sequences.csv \\\n    --train_labels_csv_path /kaggle/working/train_labels.csv \\\n    --n_predictions 5 \\\n    --len_diff_threshold 0.5 \\\n    --protein_similarity_weight 0.3 \\\n    --dna_similarity_weight 0.3 \\\n    --smiles_similarity_weight 0.1 \\\n    --text_embeddings_similarity_weight 0.9 \\\n    --train_embeddings_path /kaggle/input/datasets/stefanstefanov/srna3d2-train-text-embeddings/train_text_embeddings.safetensors \\\n    --test_embeddings_path /kaggle/working/test_text_embeddings.safetensors \\\n    --model_path /kaggle/input/datasets/stefanstefanov/srna3d2-tmscore-lightgbm/model.txt \\\n    --seed \"${SEED}\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:00:05.295802Z","iopub.execute_input":"2026-04-05T20:00:05.296123Z","iopub.status.idle":"2026-04-05T20:00:51.58703Z","shell.execute_reply.started":"2026-04-05T20:00:05.296088Z","shell.execute_reply":"2026-04-05T20:00:51.58623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python -m srna3d.scripts.split_into_chains \\\n    --input_csv /kaggle/working/test_sequences.csv \\\n    --max_seq_len ${MAX_SEQ_LEN} \\\n    --overlap_len ${OVERLAP_LEN} \\\n    --output_csv /kaggle/working/test_chain_sequences.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:00:51.588353Z","iopub.execute_input":"2026-04-05T20:00:51.588932Z","iopub.status.idle":"2026-04-05T20:00:52.300803Z","shell.execute_reply.started":"2026-04-05T20:00:51.588899Z","shell.execute_reply":"2026-04-05T20:00:52.300049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python -m srna3d.scripts.split_sequences_into_chunks \\\n    --input_csv /kaggle/working/test_chain_sequences.csv  \\\n    --max_seq_len ${MAX_SEQ_LEN} \\\n    --overlap_len ${OVERLAP_LEN} \\\n    --output_csv /kaggle/working/test_chunked_sequences.csv","metadata":{"execution":{"iopub.status.busy":"2026-04-05T20:00:52.302042Z","iopub.execute_input":"2026-04-05T20:00:52.30236Z","iopub.status.idle":"2026-04-05T20:00:53.025664Z","shell.execute_reply.started":"2026-04-05T20:00:52.302316Z","shell.execute_reply":"2026-04-05T20:00:53.024886Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !python -m preprocess.convert_templates_to_pt_files --input_csv /kaggle/input/stanford-rna-3d-folding-pt2-templates/templates_tbm.csv --output_path /kaggle/working/templates.pt\n!python -m preprocess.convert_templates_to_pt_files --input_csv /kaggle/working/submission_tbm.csv --output_path /kaggle/working/templates.pt","metadata":{"execution":{"iopub.status.busy":"2026-04-05T20:00:53.026896Z","iopub.execute_input":"2026-04-05T20:00:53.027216Z","iopub.status.idle":"2026-04-05T20:00:55.818075Z","shell.execute_reply.started":"2026-04-05T20:00:53.027167Z","shell.execute_reply":"2026-04-05T20:00:55.817294Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile rnapro_inference.sh\n\nDUMP_DIR=\"/kaggle/working/rnapro\"\nmkdir -p \"${DUMP_DIR}\"\ncd \"${DUMP_DIR}\"\npwd\n\nexport LAYERNORM_TYPE=fast_layernorm # fast_layernorm, torch\nexport RNAPRO_DATA_ROOT_DIR=/kaggle/input/rnapro-ccd-cache/ccd_cache/\nexport PYTORCH_CUDA_ALLOC_CONF=\"expandable_segments:True\"\nexport TORCH_CUDA_ARCH_LIST=\"7.5\"\n\n# Inference parameters (RNAPro)\nN_SAMPLE=1\nN_STEP=100\nN_CYCLE=10\n\n# Paths\n# Set a valid checkpoint file path below\nCHECKPOINT_PATH=\"/kaggle/input/rnapro-model-download/rnapro-private-best-500m.ckpt\"\n\n# Template/MSA settings\nTEMPLATE_DATA=\"/kaggle/working/templates.pt\"\n# Note: template_idx supports 5 choices and maps to top-k:\n# 0->top1, 1->top2, 2->top3, 3->top4, 4->top5\nTEMPLATE_IDX=4\nRNA_MSA_DIR=\"/kaggle/input/stanford-rna-3d-folding-2/MSA\"\n\nSEQUENCES_CSV=\"/kaggle/working/test_chunked_sequences.csv\"\n\n# RibonanzaNet2 path (keep as-is per request)\nRIBONANZA_PATH=\"/kaggle/input/ribonanzanet2/pytorch/alpha/1/\"\n\n# Model selection: keep to an existing key to align defaults (N_step=200, N_cycle=10)\nMODEL_NAME=\"rnapro_base\"\n\npython -m rnapro.runner.inference \\\n    --model_name \"${MODEL_NAME}\" \\\n    --dtype fp32 \\\n    --seeds 3333 \\\n    --dump_dir \"${DUMP_DIR}\" \\\n    --load_checkpoint_path \"${CHECKPOINT_PATH}\" \\\n    --use_msa true \\\n    --use_template \"ca_precomputed\" \\\n    --model.use_template \"ca_precomputed\" \\\n    --model.use_RibonanzaNet2 true \\\n    --model.template_embedder.n_blocks 2 \\\n    --model.ribonanza_net_path \"${RIBONANZA_PATH}\" \\\n    --template_data \"${TEMPLATE_DATA}\" \\\n    --template_idx ${TEMPLATE_IDX} \\\n    --rna_msa_dir \"${RNA_MSA_DIR}\" \\\n    --model.N_cycle ${N_CYCLE} \\\n    --sample_diffusion.N_sample ${N_SAMPLE} \\\n    --sample_diffusion.N_step ${N_STEP} \\\n    --load_strict true \\\n    --num_workers 0 \\\n    --triangle_attention \"triattention\" \\\n    --triangle_multiplicative \"torch\" \\\n    --skip_amp.confidence_head true \\\n    --skip_amp.sample_diffusion true \\\n    --sequences_csv \"${SEQUENCES_CSV}\" \\\n    --max_len ${MAX_SEQ_LEN}\n\nmv \"${DUMP_DIR}/submission_rnapro.csv\" \"${DUMP_DIR}/chunked_submission.csv\"\n\npython -m srna3d.scripts.combine_chunked_predictions \\\n    --chunked-sequences \"${SEQUENCES_CSV}\" \\\n    --chunked-predictions \"${DUMP_DIR}/chunked_submission.csv\" \\\n    --output \"${DUMP_DIR}/combined_chunks_submission.csv\"\n\npython -m srna3d.scripts.combine_chain_predictions \\\n    --chain-predictions \"${DUMP_DIR}/combined_chunks_submission.csv\" \\\n    --chain-sequences /kaggle/working/test_chain_sequences.csv \\\n    --source-sequences /kaggle/working/test_sequences.csv \\\n    --output /kaggle/working/submission_rnapro.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:00:55.819409Z","iopub.execute_input":"2026-04-05T20:00:55.820357Z","iopub.status.idle":"2026-04-05T20:00:55.82722Z","shell.execute_reply.started":"2026-04-05T20:00:55.820325Z","shell.execute_reply":"2026-04-05T20:00:55.82657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile protenix_inference.sh\n\nDUMP_DIR=\"/kaggle/working/protenix\"\nmkdir -p \"${DUMP_DIR}\"\ncd \"${DUMP_DIR}\"\npwd\n\nexport LAYERNORM_TYPE=fast_layernorm # fast_layernorm, torch\nexport PROTENIX_ROOT_DIR=/kaggle/input/notebooks/stefanstefanov/protenix-data-download/\nexport PYTORCH_CUDA_ALLOC_CONF=\"expandable_segments:True\"\nexport TORCH_CUDA_ARCH_LIST=\"7.5\"\n\n# Inference parameters\nN_SAMPLE=2\nN_STEP=150\nN_CYCLE=10\n\n# Paths\nCHECKPOINT_DIR=\"/kaggle/input/notebooks/stefanstefanov/protenix-model-download/\"\n\nRNA_MSA_DIR=\"/kaggle/input/stanford-rna-3d-folding-2/MSA\"\n\nSEQUENCES_CSV=\"/kaggle/working/test_chunked_sequences.csv\"\n\nMODEL_NAME=\"protenix_base_20250630_v1.0.0\"\n\npython -m runner.inference \\\n    --model_name \"${MODEL_NAME}\" \\\n    --dtype fp32 \\\n    --seeds 9999 \\\n    --dump_dir \"${DUMP_DIR}\" \\\n    --load_checkpoint_dir \"${CHECKPOINT_DIR}\" \\\n    --use_template false \\\n    --use_msa true \\\n    --use_rna_msa true \\\n    --rna_msa_dir \"${RNA_MSA_DIR}\" \\\n    --model.N_cycle ${N_CYCLE} \\\n    --sample_diffusion.N_sample ${N_SAMPLE} \\\n    --sample_diffusion.N_step ${N_STEP} \\\n    --load_strict true \\\n    --num_workers 0 \\\n    --triangle_attention \"triattention\" \\\n    --triangle_multiplicative \"torch\" \\\n    --skip_amp.confidence_head true \\\n    --skip_amp.sample_diffusion true \\\n    --sequences_csv \"${SEQUENCES_CSV}\" \\\n    --max_len_non_rna 128 \\\n    --max_len ${MAX_SEQ_LEN}\n\nmv \"${DUMP_DIR}/submission_protenix.csv\" \"${DUMP_DIR}/chunked_submission.csv\"\n\npython -m srna3d.scripts.combine_chunked_predictions \\\n    --chunked-sequences \"${SEQUENCES_CSV}\" \\\n    --chunked-predictions \"${DUMP_DIR}/chunked_submission.csv\" \\\n    --output \"${DUMP_DIR}/combined_chunks_submission.csv\"\n\npython -m srna3d.scripts.combine_chain_predictions \\\n    --chain-predictions \"${DUMP_DIR}/combined_chunks_submission.csv\" \\\n    --chain-sequences /kaggle/working/test_chain_sequences.csv \\\n    --source-sequences /kaggle/working/test_sequences.csv \\\n    --output /kaggle/working/submission_protenix.csv","metadata":{"execution":{"iopub.status.busy":"2026-04-05T20:00:55.828267Z","iopub.execute_input":"2026-04-05T20:00:55.828526Z","iopub.status.idle":"2026-04-05T20:00:55.847007Z","shell.execute_reply.started":"2026-04-05T20:00:55.828501Z","shell.execute_reply":"2026-04-05T20:00:55.846478Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rnapro_env =  os.environ.copy()\nrnapro_env['CUDA_VISIBLE_DEVICES'] = '0'\nrnapro_popen = subprocess.Popen(\n    ['bash', './rnapro_inference.sh'],\n    stdout=subprocess.PIPE, \n    stderr=subprocess.PIPE,\n    env=rnapro_env\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:00:55.84792Z","iopub.execute_input":"2026-04-05T20:00:55.848146Z","iopub.status.idle":"2026-04-05T20:00:55.866136Z","shell.execute_reply.started":"2026-04-05T20:00:55.848127Z","shell.execute_reply":"2026-04-05T20:00:55.865284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()\ntime.sleep(6*60)\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:00:55.87044Z","iopub.execute_input":"2026-04-05T20:00:55.870768Z","iopub.status.idle":"2026-04-05T20:06:56.027087Z","shell.execute_reply.started":"2026-04-05T20:00:55.870745Z","shell.execute_reply":"2026-04-05T20:06:56.026338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"protenix_env =  os.environ.copy()\nprotenix_env['CUDA_VISIBLE_DEVICES'] = '1'\nprotenix_popen = subprocess.Popen(\n    ['bash', './protenix_inference.sh'],\n    stdout=subprocess.PIPE, \n    stderr=subprocess.PIPE,\n    env=protenix_env\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:06:56.028164Z","iopub.execute_input":"2026-04-05T20:06:56.028678Z","iopub.status.idle":"2026-04-05T20:06:56.038563Z","shell.execute_reply.started":"2026-04-05T20:06:56.028642Z","shell.execute_reply":"2026-04-05T20:06:56.037792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"protenix_stdout, protenix_stderr = protenix_popen.communicate()\nprotenix_stdout = protenix_stdout.decode('utf-8')\nprotenix_stderr = protenix_stderr.decode('utf-8')\nprint(f'{protenix_popen.returncode=}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:06:56.04048Z","iopub.execute_input":"2026-04-05T20:06:56.040701Z","iopub.status.idle":"2026-04-05T20:13:01.25334Z","shell.execute_reply.started":"2026-04-05T20:06:56.04068Z","shell.execute_reply":"2026-04-05T20:13:01.25263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(protenix_stdout)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:01.254356Z","iopub.execute_input":"2026-04-05T20:13:01.254878Z","iopub.status.idle":"2026-04-05T20:13:01.258808Z","shell.execute_reply.started":"2026-04-05T20:13:01.254854Z","shell.execute_reply":"2026-04-05T20:13:01.258234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(protenix_stderr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:01.259662Z","iopub.execute_input":"2026-04-05T20:13:01.259941Z","iopub.status.idle":"2026-04-05T20:13:01.275133Z","shell.execute_reply.started":"2026-04-05T20:13:01.259912Z","shell.execute_reply":"2026-04-05T20:13:01.274603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del protenix_env, protenix_popen, protenix_stdout, protenix_stderr\ngc.collect()\ntime.sleep(10)\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:01.275972Z","iopub.execute_input":"2026-04-05T20:13:01.276244Z","iopub.status.idle":"2026-04-05T20:13:11.399822Z","shell.execute_reply.started":"2026-04-05T20:13:01.276222Z","shell.execute_reply":"2026-04-05T20:13:11.399224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rnapro_stdout, rnapro_stderr = rnapro_popen.communicate()\nrnapro_stdout = rnapro_stdout.decode('utf-8')\nrnapro_stderr = rnapro_stderr.decode('utf-8')\nprint(f'{rnapro_popen.returncode=}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:11.400757Z","iopub.execute_input":"2026-04-05T20:13:11.401043Z","iopub.status.idle":"2026-04-05T20:13:11.406024Z","shell.execute_reply.started":"2026-04-05T20:13:11.40102Z","shell.execute_reply":"2026-04-05T20:13:11.405439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(rnapro_stdout)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:11.406953Z","iopub.execute_input":"2026-04-05T20:13:11.407389Z","iopub.status.idle":"2026-04-05T20:13:11.420817Z","shell.execute_reply.started":"2026-04-05T20:13:11.407365Z","shell.execute_reply":"2026-04-05T20:13:11.420271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(rnapro_stderr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:11.421652Z","iopub.execute_input":"2026-04-05T20:13:11.421895Z","iopub.status.idle":"2026-04-05T20:13:11.435334Z","shell.execute_reply.started":"2026-04-05T20:13:11.42187Z","shell.execute_reply":"2026-04-05T20:13:11.434793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del rnapro_env, rnapro_popen,rnapro_stdout, rnapro_stderr\ngc.collect()\ntime.sleep(10)\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:11.436067Z","iopub.execute_input":"2026-04-05T20:13:11.436312Z","iopub.status.idle":"2026-04-05T20:13:21.563518Z","shell.execute_reply.started":"2026-04-05T20:13:11.436284Z","shell.execute_reply":"2026-04-05T20:13:21.562871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_tbm = pd.read_csv(\"/kaggle/working/submission_tbm.csv\")\ndf_protenix = pd.read_csv(\"/kaggle/working/submission_protenix.csv\")\ndf_rnapro = pd.read_csv(\"/kaggle/working/submission_rnapro.csv\")\n\ndf_final = pd.merge(\n    df_tbm[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1', 'x_2', 'y_2', 'z_2']],\n    df_protenix[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1', 'x_2', 'y_2', 'z_2']].rename(\n        columns={\n            'x_1': 'x_3', 'y_1': 'y_3', 'z_1': 'z_3',\n            'x_2': 'x_4', 'y_2': 'y_4', 'z_2': 'z_4',\n        }\n    ),\n    on=['ID', 'resname', 'resid'],\n    how='inner'\n)\n\n\ndf_final = pd.merge(\n    df_final[['ID', 'resname', 'resid'] + [f'{axis}_{i}' for i in range(1, 5) for axis in ['x', 'y', 'z']]],\n    df_rnapro[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']].rename(\n        columns={\n            'x_1': 'x_5', 'y_1': 'y_5', 'z_1': 'z_5',\n        }\n    ),\n    on=['ID', 'resname', 'resid'],\n    how='inner'\n)\n\n\ndf_final = df_final[['ID', 'resname', 'resid'] + [f'{axis}_{i}' for i in range(1, 6) for axis in ['x', 'y', 'z']]]\n\nprint('Saving merged submission.csv')\ndf_final.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2026-04-05T20:13:21.564469Z","iopub.execute_input":"2026-04-05T20:13:21.564698Z","iopub.status.idle":"2026-04-05T20:13:21.595324Z","shell.execute_reply.started":"2026-04-05T20:13:21.564677Z","shell.execute_reply":"2026-04-05T20:13:21.594685Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!head submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:21.596377Z","iopub.execute_input":"2026-04-05T20:13:21.596815Z","iopub.status.idle":"2026-04-05T20:13:21.724303Z","shell.execute_reply.started":"2026-04-05T20:13:21.596782Z","shell.execute_reply":"2026-04-05T20:13:21.723493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/working/submission.csv\")\nsub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:21.725726Z","iopub.execute_input":"2026-04-05T20:13:21.726045Z","iopub.status.idle":"2026-04-05T20:13:21.760688Z","shell.execute_reply.started":"2026-04-05T20:13:21.726013Z","shell.execute_reply":"2026-04-05T20:13:21.759968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not IS_SUBMISSION:\n    pdb_output_dir = '/kaggle/working/eval'\n    os.makedirs(pdb_output_dir, exist_ok=True)\n    sub = pd.read_csv('/kaggle/working/submission.csv')\n    sol = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-2/validation_labels.csv')\n\n    sub['target_id'] = sub['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\n    sol['target_id'] = sol['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\n    \n    # Get unique targets from submission\n    sub_targets = sub['target_id'].unique()\n    \n    results = []\n    for target_id in sub_targets:\n        group_native = sol[sol['target_id'] == target_id]\n        group_predicted = sub[sub['target_id'] == target_id]\n        result = tm_score(\n            group_native, group_predicted, 'ID',\n            pdb_output_dir=pdb_output_dir\n        )\n        print(f\"{target_id}: {result:.4f}\")\n        results.append(result)\n    \n    print(f\"\\nMean score: {sum(results)/len(results):.4f} (n={len(results)})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-05T20:13:21.761632Z","iopub.execute_input":"2026-04-05T20:13:21.761888Z","iopub.status.idle":"2026-04-05T20:13:22.160979Z","shell.execute_reply.started":"2026-04-05T20:13:21.761856Z","shell.execute_reply":"2026-04-05T20:13:22.160225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}