{
  "id": 442872,
  "title": "ValueError : patient_df = test_df.query(\"patient_id == @patient_id\")",
  "url": "/competitions/rsna-2023-abdominal-trauma-detection/discussion/442872",
  "author_name": "Ahmed Raza",
  "post_date": "2023-09-24T14:38:39.530000",
  "votes": 0,
  "comment_count": 1,
  "views": 0,
  "content": "<p>I am getting ValueError while using the string \"patient_id == <a href=\"https://www.kaggle.com/patient\" target=\"_blank\">@patient</a>_id\" in query() function</p>\n<p>Inference<br>\n`# Getting unique patient IDs from test dataset<br>\npatient_ids = test_df['patient_id'].unique()</p>\n<h1>Initializing array to store predictions</h1>\n<p>patient_preds = np.zeros(shape=(len(patient_ids), 2<em>2 + 3</em>3), dtype='float32')</p>\n<h1>Iterating over each patient</h1>\n<p>for pidx, patient_id in tqdm(enumerate(patient_ids), total=len(patient_ids), desc=\"Patients \"):<br>\n    # Query the dataframe for a particular patient<br>\n    patient_df = test_df.query(\"patient_id == <a href=\"https://www.kaggle.com/patient\" target=\"_blank\">@patient</a>_id\") #ERROR I AM GETTING<br>\n    # Initializing model predictions array<br>\n    model_preds = np.zeros(shape=(1, 11), dtype=np.float32)<br>\n    print(\"=\"<em>25)\n    print(f\"   Patient ID: {patient_id}\")\n    print(\"=\"</em>25)<br>\n    # Iterating over each model<br>\n    for midx, (img_size, fold_paths) in enumerate(MODEL_CONFIGS):     <br>\n        # Getting image paths for a patient<br>\n        patient_paths = patient_df.image_path.tolist()<br>\n        # Setting batch size based on number of patient paths and dimension of image<br>\n        dim = np.prod(img_size)<strong>0.5\n        if dim &gt;= 1024:\n            CFG.batch_size = REPLICAS * int(4 * 2)\n        elif dim &gt;= 768:\n            CFG.batch_size = REPLICAS * int(16 * 2)\n        elif dim &gt;= 640:\n            CFG.batch_size = REPLICAS * int(28 * 2)\n        else:\n            CFG.batch_size = REPLICAS * int(32 * 2)   \n        # Clip batch_sizs to min\n        min_bs = 2</strong>np.floor(np.log2(len(patient_paths)))<br>\n        CFG.batch_size = min(min_bs, CFG.batch_size)<br>\n        # Building dataset for prediction<br>\n        dtest = build_dataset(<br>\n            patient_paths, <br>\n            batch_size=CFG.batch_size, repeat=True, <br>\n            shuffle=False, augment=CFG.tta &gt; 1, cache=False,<br>\n            decode_fn=build_decoder(with_labels=False, target_size=img_size),<br>\n            augment_fn=build_augmenter(with_labels=False, dim=img_size)<br>\n        ) <br>\n        # Iterating over each fold<br>\n        for fold_path in fold_paths:<br>\n            with strategy.scope():<br>\n                # Loading a model from a fold path<br>\n                model = tf.keras.models.load_model(fold_path, compile=False) <br>\n            # Predicting with the model<br>\n            pred = model.predict(dtest, steps = CFG.tta * len(patient_paths) / CFG.batch_size, verbose=1)<br>\n            pred = np.concatenate(pred, axis=-1).astype('float32') # reducing memory footprint<br>\n            pred = pred[:CFG.tta * len(patient_paths), :]<br>\n            pred = np.mean(pred.reshape(CFG.tta, len(patient_paths), 11), axis=0)<br>\n            pred = np.max(pred, axis=0) # taking max prediction of all ct scans for a patient <br>\n            # Store model's prediction<br>\n            model_preds += pred / (len(fold_paths)*len(MODEL_CONFIGS)) <br>\n            # Deleting variables to free up memory<br>\n            del model, pred; gc.collect() <br>\n            print('\\n')<br>\n        del dtest, patient_paths; gc.collect()<br>\n    # Adding processed predictions to patient_preds<br>\n    patient_preds[pidx, :] += post_proc_v2(model_preds)[0]<br>\n    del model_preds; gc.collect()<br>\nprint(\"Prediction Done!\")<br>\n`</p>",
  "messages": [
    {
      "id": 2454580,
      "postDate": "2023-09-24T23:18:49.773Z",
      "content": "<p>Try this ..</p>\n<p><code>patient_df = test_df[test_df['patient_id'] == patient_id]</code></p>",
      "rawMarkdown": "Try this ..\n\n`patient_df = test_df[test_df['patient_id'] == patient_id]`\n"
    },
    {
      "id": 2454055,
      "postDate": "2023-09-24T14:38:39.530Z",
      "content": "<p>I am getting ValueError while using the string \"patient_id == <a href=\"https://www.kaggle.com/patient\" target=\"_blank\">@patient</a>_id\" in query() function</p>\n<p>Inference<br>\n`# Getting unique patient IDs from test dataset<br>\npatient_ids = test_df['patient_id'].unique()</p>\n<h1>Initializing array to store predictions</h1>\n<p>patient_preds = np.zeros(shape=(len(patient_ids), 2<em>2 + 3</em>3), dtype='float32')</p>\n<h1>Iterating over each patient</h1>\n<p>for pidx, patient_id in tqdm(enumerate(patient_ids), total=len(patient_ids), desc=\"Patients \"):<br>\n    # Query the dataframe for a particular patient<br>\n    patient_df = test_df.query(\"patient_id == <a href=\"https://www.kaggle.com/patient\" target=\"_blank\">@patient</a>_id\") #ERROR I AM GETTING<br>\n    # Initializing model predictions array<br>\n    model_preds = np.zeros(shape=(1, 11), dtype=np.float32)<br>\n    print(\"=\"<em>25)\n    print(f\"   Patient ID: {patient_id}\")\n    print(\"=\"</em>25)<br>\n    # Iterating over each model<br>\n    for midx, (img_size, fold_paths) in enumerate(MODEL_CONFIGS):     <br>\n        # Getting image paths for a patient<br>\n        patient_paths = patient_df.image_path.tolist()<br>\n        # Setting batch size based on number of patient paths and dimension of image<br>\n        dim = np.prod(img_size)<strong>0.5\n        if dim &gt;= 1024:\n            CFG.batch_size = REPLICAS * int(4 * 2)\n        elif dim &gt;= 768:\n            CFG.batch_size = REPLICAS * int(16 * 2)\n        elif dim &gt;= 640:\n            CFG.batch_size = REPLICAS * int(28 * 2)\n        else:\n            CFG.batch_size = REPLICAS * int(32 * 2)   \n        # Clip batch_sizs to min\n        min_bs = 2</strong>np.floor(np.log2(len(patient_paths)))<br>\n        CFG.batch_size = min(min_bs, CFG.batch_size)<br>\n        # Building dataset for prediction<br>\n        dtest = build_dataset(<br>\n            patient_paths, <br>\n            batch_size=CFG.batch_size, repeat=True, <br>\n            shuffle=False, augment=CFG.tta &gt; 1, cache=False,<br>\n            decode_fn=build_decoder(with_labels=False, target_size=img_size),<br>\n            augment_fn=build_augmenter(with_labels=False, dim=img_size)<br>\n        ) <br>\n        # Iterating over each fold<br>\n        for fold_path in fold_paths:<br>\n            with strategy.scope():<br>\n                # Loading a model from a fold path<br>\n                model = tf.keras.models.load_model(fold_path, compile=False) <br>\n            # Predicting with the model<br>\n            pred = model.predict(dtest, steps = CFG.tta * len(patient_paths) / CFG.batch_size, verbose=1)<br>\n            pred = np.concatenate(pred, axis=-1).astype('float32') # reducing memory footprint<br>\n            pred = pred[:CFG.tta * len(patient_paths), :]<br>\n            pred = np.mean(pred.reshape(CFG.tta, len(patient_paths), 11), axis=0)<br>\n            pred = np.max(pred, axis=0) # taking max prediction of all ct scans for a patient <br>\n            # Store model's prediction<br>\n            model_preds += pred / (len(fold_paths)*len(MODEL_CONFIGS)) <br>\n            # Deleting variables to free up memory<br>\n            del model, pred; gc.collect() <br>\n            print('\\n')<br>\n        del dtest, patient_paths; gc.collect()<br>\n    # Adding processed predictions to patient_preds<br>\n    patient_preds[pidx, :] += post_proc_v2(model_preds)[0]<br>\n    del model_preds; gc.collect()<br>\nprint(\"Prediction Done!\")<br>\n`</p>",
      "rawMarkdown": "I am getting ValueError while using the string \"patient_id == @patient_id\" in query() function\n\nInference\n`# Getting unique patient IDs from test dataset\npatient_ids = test_df['patient_id'].unique()\n# Initializing array to store predictions\npatient_preds = np.zeros(shape=(len(patient_ids), 2*2 + 3*3), dtype='float32')\n# Iterating over each patient\nfor pidx, patient_id in tqdm(enumerate(patient_ids), total=len(patient_ids), desc=\"Patients \"):\n    # Query the dataframe for a particular patient\n    patient_df = test_df.query(\"patient_id == @patient_id\") #ERROR I AM GETTING\n    # Initializing model predictions array\n    model_preds = np.zeros(shape=(1, 11), dtype=np.float32)\n    print(\"=\"*25)\n    print(f\"   Patient ID: {patient_id}\")\n    print(\"=\"*25)\n    # Iterating over each model\n    for midx, (img_size, fold_paths) in enumerate(MODEL_CONFIGS):     \n        # Getting image paths for a patient\n        patient_paths = patient_df.image_path.tolist()\n        # Setting batch size based on number of patient paths and dimension of image\n        dim = np.prod(img_size)**0.5\n        if dim >= 1024:\n            CFG.batch_size = REPLICAS * int(4 * 2)\n        elif dim >= 768:\n            CFG.batch_size = REPLICAS * int(16 * 2)\n        elif dim >= 640:\n            CFG.batch_size = REPLICAS * int(28 * 2)\n        else:\n            CFG.batch_size = REPLICAS * int(32 * 2)   \n        # Clip batch_sizs to min\n        min_bs = 2**np.floor(np.log2(len(patient_paths)))\n        CFG.batch_size = min(min_bs, CFG.batch_size)\n        # Building dataset for prediction\n        dtest = build_dataset(\n            patient_paths, \n            batch_size=CFG.batch_size, repeat=True, \n            shuffle=False, augment=CFG.tta > 1, cache=False,\n            decode_fn=build_decoder(with_labels=False, target_size=img_size),\n            augment_fn=build_augmenter(with_labels=False, dim=img_size)\n        ) \n        # Iterating over each fold\n        for fold_path in fold_paths:\n            with strategy.scope():\n                # Loading a model from a fold path\n                model = tf.keras.models.load_model(fold_path, compile=False) \n            # Predicting with the model\n            pred = model.predict(dtest, steps = CFG.tta * len(patient_paths) / CFG.batch_size, verbose=1)\n            pred = np.concatenate(pred, axis=-1).astype('float32') # reducing memory footprint\n            pred = pred[:CFG.tta * len(patient_paths), :]\n            pred = np.mean(pred.reshape(CFG.tta, len(patient_paths), 11), axis=0)\n            pred = np.max(pred, axis=0) # taking max prediction of all ct scans for a patient \n            # Store model's prediction\n            model_preds += pred / (len(fold_paths)*len(MODEL_CONFIGS)) \n            # Deleting variables to free up memory\n            del model, pred; gc.collect() \n            print('\\n')\n        del dtest, patient_paths; gc.collect()\n    # Adding processed predictions to patient_preds\n    patient_preds[pidx, :] += post_proc_v2(model_preds)[0]\n    del model_preds; gc.collect()\nprint(\"Prediction Done!\")\n`"
    }
  ],
  "comments": [
    {
      "id": 2454580,
      "author_name": "David Roberts",
      "author_url": "",
      "post_date": "2023-09-24T23:18:49.773000",
      "content": "<p>Try this ..</p>\n<p><code>patient_df = test_df[test_df['patient_id'] == patient_id]</code></p>",
      "votes": 0,
      "replies": []
    }
  ],
  "raw_markdown_by_id": {
    "2454580": "Try this ..\n\n`patient_df = test_df[test_df['patient_id'] == patient_id]`\n",
    "2454055": "I am getting ValueError while using the string \"patient_id == @patient_id\" in query() function\n\nInference\n`# Getting unique patient IDs from test dataset\npatient_ids = test_df['patient_id'].unique()\n# Initializing array to store predictions\npatient_preds = np.zeros(shape=(len(patient_ids), 2*2 + 3*3), dtype='float32')\n# Iterating over each patient\nfor pidx, patient_id in tqdm(enumerate(patient_ids), total=len(patient_ids), desc=\"Patients \"):\n    # Query the dataframe for a particular patient\n    patient_df = test_df.query(\"patient_id == @patient_id\") #ERROR I AM GETTING\n    # Initializing model predictions array\n    model_preds = np.zeros(shape=(1, 11), dtype=np.float32)\n    print(\"=\"*25)\n    print(f\"   Patient ID: {patient_id}\")\n    print(\"=\"*25)\n    # Iterating over each model\n    for midx, (img_size, fold_paths) in enumerate(MODEL_CONFIGS):     \n        # Getting image paths for a patient\n        patient_paths = patient_df.image_path.tolist()\n        # Setting batch size based on number of patient paths and dimension of image\n        dim = np.prod(img_size)**0.5\n        if dim >= 1024:\n            CFG.batch_size = REPLICAS * int(4 * 2)\n        elif dim >= 768:\n            CFG.batch_size = REPLICAS * int(16 * 2)\n        elif dim >= 640:\n            CFG.batch_size = REPLICAS * int(28 * 2)\n        else:\n            CFG.batch_size = REPLICAS * int(32 * 2)   \n        # Clip batch_sizs to min\n        min_bs = 2**np.floor(np.log2(len(patient_paths)))\n        CFG.batch_size = min(min_bs, CFG.batch_size)\n        # Building dataset for prediction\n        dtest = build_dataset(\n            patient_paths, \n            batch_size=CFG.batch_size, repeat=True, \n            shuffle=False, augment=CFG.tta > 1, cache=False,\n            decode_fn=build_decoder(with_labels=False, target_size=img_size),\n            augment_fn=build_augmenter(with_labels=False, dim=img_size)\n        ) \n        # Iterating over each fold\n        for fold_path in fold_paths:\n            with strategy.scope():\n                # Loading a model from a fold path\n                model = tf.keras.models.load_model(fold_path, compile=False) \n            # Predicting with the model\n            pred = model.predict(dtest, steps = CFG.tta * len(patient_paths) / CFG.batch_size, verbose=1)\n            pred = np.concatenate(pred, axis=-1).astype('float32') # reducing memory footprint\n            pred = pred[:CFG.tta * len(patient_paths), :]\n            pred = np.mean(pred.reshape(CFG.tta, len(patient_paths), 11), axis=0)\n            pred = np.max(pred, axis=0) # taking max prediction of all ct scans for a patient \n            # Store model's prediction\n            model_preds += pred / (len(fold_paths)*len(MODEL_CONFIGS)) \n            # Deleting variables to free up memory\n            del model, pred; gc.collect() \n            print('\\n')\n        del dtest, patient_paths; gc.collect()\n    # Adding processed predictions to patient_preds\n    patient_preds[pidx, :] += post_proc_v2(model_preds)[0]\n    del model_preds; gc.collect()\nprint(\"Prediction Done!\")\n`"
  }
}