{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport sklearn.datasets\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\") # Hide the warnings from calculating pfbeta with 0 denominators\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import classification_report, confusion_matrix","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-31T10:12:09.490333Z","iopub.execute_input":"2022-12-31T10:12:09.490609Z","iopub.status.idle":"2022-12-31T10:12:10.69747Z","shell.execute_reply.started":"2022-12-31T10:12:09.490583Z","shell.execute_reply":"2022-12-31T10:12:10.696574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PF1 metric","metadata":{}},{"cell_type":"code","source":"def pfbeta(labels, preds, beta=1):\n    preds = preds.clip(0, 1)\n    y_true_count = labels.sum()\n    ctp = preds[labels == 1].sum()\n    cfp = preds[labels == 0].sum()\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0.0\n    \ndef metrics(labels, preds, silent=False):\n    pfbeta_cont = pfbeta(labels, preds)\n    thresholds = np.linspace(0.05, 0.95, 19)\n    pfbeta_thresholds = [pfbeta(labels, preds > t) for t in thresholds]\n    best_idx = np.argmax(pfbeta_thresholds)\n    \n    thresh = pfbeta_thresholds[best_idx]\n    \n    if not silent:\n        print(f\"\\033[1;{31 if pfbeta_cont < thresh else 32};40m pf1_cont: {pfbeta_cont:.3f}\\t \\033[1;{32 if pfbeta_cont < thresh else 31};40m pf1_thresh: {pfbeta_thresholds[best_idx]:.3f} @ {thresholds[best_idx]:.2f}\")\n    return pfbeta_cont, pfbeta_thresholds[best_idx], ","metadata":{"execution":{"iopub.status.busy":"2022-12-31T10:12:09.452274Z","iopub.execute_input":"2022-12-31T10:12:09.453056Z","iopub.status.idle":"2022-12-31T10:12:09.488848Z","shell.execute_reply.started":"2022-12-31T10:12:09.452946Z","shell.execute_reply":"2022-12-31T10:12:09.487955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create run 190 simulations\n10 different random seeds\n19 different strengths of relationship between x and y_true","metadata":{}},{"cell_type":"code","source":"flip_proportions = np.linspace(0.05, 0.95, 19)\nweights = np.linspace(0.05, 0.95, 19)\nfor random_state in range(10):\n    for weight in weights:\n        for flip_proportion in flip_proportions:\n            x, y_true = sklearn.datasets.make_classification(n_samples=1000,n_classes=2,n_clusters_per_class=1,n_features=1,n_informative=1,n_redundant=0,n_repeated=0,flip_y=flip_proportion, random_state=random_state, weights=[weight, 1-weight])\n            model = LogisticRegression(solver='liblinear', random_state=0).fit(x, y_true)\n            y_pred = model.predict_proba(x)\n            y_pred = y_pred[:,1]  # Take only the probability for the positive class\n            pf1_cont, pf1_thresh = metrics(y_true, y_pred)\n            assert pf1_cont < pf1_thresh","metadata":{"execution":{"iopub.status.busy":"2022-12-31T10:22:52.545886Z","iopub.execute_input":"2022-12-31T10:22:52.54628Z","iopub.status.idle":"2022-12-31T10:23:03.857908Z","shell.execute_reply.started":"2022-12-31T10:22:52.546248Z","shell.execute_reply":"2022-12-31T10:23:03.856794Z"},"trusted":true},"execution_count":null,"outputs":[]}]}