{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"},{"sourceId":10855324,"sourceType":"datasetVersion","datasetId":6742586},{"sourceId":14745885,"sourceType":"datasetVersion","datasetId":9424250},{"sourceId":290004465,"sourceType":"kernelVersion"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport random\nimport joblib\npd.options.mode.chained_assignment = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:15:11.136032Z","iopub.execute_input":"2026-02-05T19:15:11.136942Z","iopub.status.idle":"2026-02-05T19:15:11.176452Z","shell.execute_reply.started":"2026-02-05T19:15:11.13688Z","shell.execute_reply":"2026-02-05T19:15:11.17539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# install the metric and defined the score() function\n!ls /kaggle/usr/lib/\nimport runpy\nmodule_globals = runpy.run_path(\"/kaggle/usr/lib/tm-score-permutechains/metric.py\")\nscore = module_globals['score']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:13:10.752018Z","iopub.execute_input":"2026-02-05T19:13:10.752386Z","iopub.status.idle":"2026-02-05T19:13:10.901361Z","shell.execute_reply.started":"2026-02-05T19:13:10.752356Z","shell.execute_reply":"2026-02-05T19:13:10.900135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.listdir('/kaggle/input/1anr-new-coords-pkl')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:15:58.229862Z","iopub.execute_input":"2026-02-05T19:15:58.230781Z","iopub.status.idle":"2026-02-05T19:15:58.239545Z","shell.execute_reply.started":"2026-02-05T19:15:58.230747Z","shell.execute_reply":"2026-02-05T19:15:58.238468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_PATH = '/kaggle/input/stanford-rna-3d-folding-2/'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:16:20.639893Z","iopub.execute_input":"2026-02-05T19:16:20.640344Z","iopub.status.idle":"2026-02-05T19:16:20.645169Z","shell.execute_reply.started":"2026-02-05T19:16:20.6403Z","shell.execute_reply":"2026-02-05T19:16:20.644236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tid = '1ANR'\n\ndf = pd.read_csv(DATA_PATH + 'train_labels.csv')\ndf['target_id'] = df['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\no_coords = np.array(df.loc[df.target_id == tid,['x_1', 'y_1', 'z_1']])\nnew_coords = joblib.load('/kaggle/input/1anr-new-coords-pkl/1ANR_new_coords.pkl')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:16:23.277122Z","iopub.execute_input":"2026-02-05T19:16:23.27804Z","iopub.status.idle":"2026-02-05T19:16:38.559479Z","shell.execute_reply.started":"2026-02-05T19:16:23.278007Z","shell.execute_reply":"2026-02-05T19:16:38.558507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pairwise_distances(x):\n    return np.linalg.norm(x[:, np.newaxis, :] - x[np.newaxis, :, :], axis=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:17:32.012751Z","iopub.execute_input":"2026-02-05T19:17:32.013645Z","iopub.status.idle":"2026-02-05T19:17:32.01801Z","shell.execute_reply.started":"2026-02-05T19:17:32.013611Z","shell.execute_reply":"2026-02-05T19:17:32.016975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x = pairwise_distances(new_coords)\ny = pairwise_distances(o_coords)\n\nnp.linalg.norm(x - y)\n# np.float64(2.313941023308353e-11) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:18:00.760276Z","iopub.execute_input":"2026-02-05T19:18:00.760702Z","iopub.status.idle":"2026-02-05T19:18:00.767964Z","shell.execute_reply.started":"2026-02-05T19:18:00.760668Z","shell.execute_reply":"2026-02-05T19:18:00.767011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# now let us compute TM-score\nresname = df.loc[df.target_id == tid,'resname'].to_list()\n\n# formatting\nsol_df = pd.DataFrame(o_coords, columns=['x_1','y_1','z_1'])\nsol_df['ID'] = [tid + '_' + str(i+1) for i in range(len(sol_df))]\nsol_df['resname'] = resname\nsol_df['chain'] = 'A'\nsol_df['copy'] = 1\nsol_df['resid'] = np.arange(1, len(sol_df)+1)\n\nfor el in range(2,41):\n    sol_df['x_'+ str(el)] = -1*(10**18)\n    sol_df['y_'+ str(el)] = -1*(10**18)\n    sol_df['z_'+ str(el)] = -1*(10**18)\n    sol_df['x_'+ str(el)] = sol_df['x_'+ str(el)].astype(np.float64)\n    sol_df['y_'+ str(el)] = sol_df['y_'+ str(el)].astype(np.float64)\n    sol_df['z_'+ str(el)] = sol_df['z_'+ str(el)].astype(np.float64)\nsol_df['Usage'] = 'Public'\n\n\nsub_df = pd.DataFrame(new_coords, columns=['x_1','y_1','z_1'])\nsub_df['resname'] = resname\nsub_df['chain'] = 'A'\nsub_df['copy'] = 1\nsub_df['resid'] = np.arange(1, len(sub_df)+1)\nsub_df['ID'] = [tid + '_' + str(i+1) for i in range(len(sub_df))]\n\nfor el in range(2,6):\n    sub_df['x_'+ str(el)] = sub_df.x_1.astype(np.float64)\n    sub_df['y_'+ str(el)] = sub_df.y_1.astype(np.float64)\n    sub_df['z_'+ str(el)] = sub_df.z_1.astype(np.float64)\n\nsol = sol_df.copy(deep=True)\nsub = sub_df.copy(deep=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:22:59.276709Z","iopub.execute_input":"2026-02-05T19:22:59.27751Z","iopub.status.idle":"2026-02-05T19:22:59.994506Z","shell.execute_reply.started":"2026-02-05T19:22:59.277474Z","shell.execute_reply":"2026-02-05T19:22:59.993517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run the eval code on each target and get the score, if this is a notebook run using the public test set (whose size matches the solution file, `validation_labels.csv`]\nsol['target_id'] = sol['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\nsub['target_id'] = sub['ID'].apply(lambda x: '_'.join(str(x).split('_')[:-1]))\n\nif len(sol)==len(sub): # This tests if we're looking at public val\n    results = []\n    for target_id, group_native in sol.groupby('target_id'):\n        group_predicted = sub[sub['target_id'] == target_id]\n        result = score(group_native,group_predicted,'ID')\n        print(target_id,result)\n        results.append( result )\n    print( 'Mean score:',  \n          float(sum(results) / len(results)) if len(results)>0 else 0.0, \n          f'(n={len(results)})' )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-05T19:23:12.206869Z","iopub.execute_input":"2026-02-05T19:23:12.207266Z","iopub.status.idle":"2026-02-05T19:23:12.355625Z","shell.execute_reply.started":"2026-02-05T19:23:12.207234Z","shell.execute_reply":"2026-02-05T19:23:12.35452Z"}},"outputs":[],"execution_count":null}]}