Ver código fonte

some changes

zahra 1 ano atrás
pai
commit
a461dbd9bb

Diferenças do arquivo suprimidas por serem muito extensas
+ 49 - 0
python/Logisticregression Method_MLE.ipynb


Diferenças do arquivo suprimidas por serem muito extensas
+ 0 - 35
python/Testing real data with logit_1_Zahra.ipynb


+ 0 - 153
python/Testing real data with logit_2_Zahra.ipynb

@@ -1,153 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "8f50efc2",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import numpy as np\n",
-    "import matplotlib.pyplot as plt\n",
-    "import seaborn as sns\n",
-    "from scipy import io\n",
-    "from sklearn.linear_model import LogisticRegression\n",
-    "from sklearn.preprocessing import StandardScaler  # , PolynomialFeatures\n",
-    "from sklearn.model_selection import StratifiedKFold, cross_val_score\n",
-    "from sklearn.metrics import roc_curve, auc\n",
-    "\n",
-    "# ---------------------------\n",
-    "# Load the data from .mat files\n",
-    "# ---------------------------\n",
-    "data_path = \"../data/\"\n",
-    "suv_file = data_path + \"suv_percentilesSLOthenUWM.mat\"\n",
-    "flags_file = data_path + \"flags_combined.mat\"\n",
-    "normal_range_file = data_path + \"normal_range.mat\"\n",
-    "\n",
-    "# Load .mat files\n",
-    "suv_dict = io.loadmat(suv_file)\n",
-    "flags_dict = io.loadmat(flags_file)\n",
-    "normal_range_dict = io.loadmat(normal_range_file)\n",
-    "\n",
-    "# Extract relevant data\n",
-    "# Adjust the slicing as needed based on your data dimensions.\n",
-    "suv = suv_dict['lung_SUVperc_COMBINED'][0:58, :, :]\n",
-    "flags = flags_dict['flags'][0:58, 3]\n",
-    "\n",
-    "# ---------------------------\n",
-    "# Feature Engineering\n",
-    "# ---------------------------\n",
-    "# Choose the percentile of interest. For lung and bowel, p = 94 (0-indexed)\n",
-    "p = 94\n",
-    "\n",
-    "# Compute a set of summary statistics for the chosen percentile.\n",
-    "feature_max = np.nanmax(suv[:, :, p], axis=1)\n",
-    "feature_mean = np.nanmean(suv[:, :, p], axis=1)\n",
-    "feature_std = np.nanstd(suv[:, :, p], axis=1)\n",
-    "\n",
-    "# Stack the features so that each row corresponds to one sample and\n",
-    "# each column is a different feature.\n",
-    "X = np.column_stack((feature_max, feature_mean, feature_std))\n",
-    "\n",
-    "# Handle missing values by selecting only rows with complete data.\n",
-    "valid_indices = ~np.isnan(X).any(axis=1)\n",
-    "X = X[valid_indices]\n",
-    "flags = flags[valid_indices]\n",
-    "\n",
-    "# ---------------------------\n",
-    "# Preprocessing: Feature Scaling\n",
-    "# ---------------------------\n",
-    "scaler = StandardScaler()\n",
-    "X_scaled = scaler.fit_transform(X)\n",
-    "\n",
-    "# --------------\n",
-    "# (Optional) Experiment with Polynomial Features for non-linearity\n",
-    "# --------------\n",
-    "# Uncomment below to include non-linear interactions\n",
-    "# from sklearn.preprocessing import PolynomialFeatures\n",
-    "# poly = PolynomialFeatures(degree=2, include_bias=False)\n",
-    "# X_scaled = poly.fit_transform(X_scaled)\n",
-    "\n",
-    "# ---------------------------\n",
-    "# Model: Regularized Logistic Regression\n",
-    "# ---------------------------\n",
-    "# We use L2 regularization; you can change penalty to 'l1' to test Lasso.\n",
-    "clf = LogisticRegression(penalty='l2', solver='liblinear', random_state=42)\n",
-    "clf.fit(X_scaled, flags)\n",
-    "\n",
-    "# Display model coefficients (note: these are in the scaled feature space)\n",
-    "print(\"Intercept:\", clf.intercept_)\n",
-    "print(\"Coefficients:\", clf.coef_)\n",
-    "\n",
-    "# ---------------------------\n",
-    "# Model Evaluation: Cross-Validation Accuracy\n",
-    "# ---------------------------\n",
-    "cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n",
-    "accuracy_scores = cross_val_score(clf, X_scaled, flags, cv=cv, scoring='accuracy')\n",
-    "print(\"Cross-validated accuracy scores:\", accuracy_scores)\n",
-    "print(\"Mean cross-validated accuracy:\", np.mean(accuracy_scores))\n",
-    "\n",
-    "# ---------------------------\n",
-    "# Additional Evaluation: ROC Curve and AUC\n",
-    "# ---------------------------\n",
-    "# Get predicted probabilities for the positive class\n",
-    "y_scores = clf.predict_proba(X_scaled)[:, 1]\n",
-    "\n",
-    "# Compute ROC curve\n",
-    "fpr, tpr, thresholds = roc_curve(flags, y_scores)\n",
-    "roc_auc = auc(fpr, tpr)\n",
-    "print(\"AUC:\", roc_auc)\n",
-    "\n",
-    "# Plot ROC curve\n",
-    "plt.figure(figsize=(8, 6))\n",
-    "plt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (AUC = %0.2f)' % roc_auc)\n",
-    "plt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\n",
-    "plt.xlim([0.0, 1.0])\n",
-    "plt.ylim([0.0, 1.05])\n",
-    "plt.xlabel('False Positive Rate')\n",
-    "plt.ylabel('True Positive Rate')\n",
-    "plt.title('Receiver Operating Characteristic (ROC)')\n",
-    "plt.legend(loc=\"lower right\")\n",
-    "plt.show()\n",
-    "\n",
-    "# ---------------------------\n",
-    "# Plotting Logistic Curve for Primary Feature (maxSUV)\n",
-    "# ---------------------------\n",
-    "# For visualization, we pick the feature corresponding to maxSUV (first column).\n",
-    "feature_index = 0\n",
-    "x_range = np.linspace(X[:, feature_index].min(), X[:, feature_index].max(), 100)\n",
-    "\n",
-    "# For the additional features (mean and std), we set them to their average values.\n",
-    "avg_mean = np.mean(X[:, 1])\n",
-    "avg_std = np.mean(X[:, 2])\n",
-    "X_plot = np.column_stack((x_range, np.repeat(avg_mean, 100), np.repeat(avg_std, 100)))\n",
-    "\n",
-    "# Scale the plot data using the same scaler.\n",
-    "X_plot_scaled = scaler.transform(X_plot)\n",
-    "\n",
-    "# Compute predicted probabilities along the x_range.\n",
-    "pred_probs = clf.predict_proba(X_plot_scaled)[:, 1]\n",
-    "\n",
-    "plt.figure(figsize=(10, 6))\n",
-    "# Scatter plot of observed data along maxSUV (using flags to indicate classes).\n",
-    "plt.scatter(X[flags == 0, feature_index], flags[flags == 0], label=\"NC\", color='blue', alpha=0.6)\n",
-    "plt.scatter(X[flags == 1, feature_index], flags[flags == 1], label=\"AE\", color='red', alpha=0.6)\n",
-    "# Plot the logistic regression prediction curve.\n",
-    "plt.plot(x_range, pred_probs, color='green', lw=2, label=\"Regularized Logistic Regression\")\n",
-    "plt.title(\"Prediction Probability of AE Depending on maxSUV\")\n",
-    "plt.xlabel(\"maxSUV\")\n",
-    "plt.ylabel(\"P(AE)\")\n",
-    "plt.legend(loc='best')\n",
-    "plt.grid(True)\n",
-    "plt.show()\n"
-   ]
-  }
- ],
- "metadata": {
-  "language_info": {
-   "name": "python"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}

+ 0 - 86
python/Testing real data with logit_NEW.ipynb

@@ -1,86 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "id": "677c5e27",
-   "metadata": {},
-   "source": [
-    "Starting with MLE and logistic regression model\n"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "2520b517",
-   "metadata": {},
-   "source": [
-    "Prerequisites\n",
-    "Make sure you have the necessary Python libraries installed:\n",
-    "pip install numpy pandas scikit-learn scipy matplotlib statsmodels\n",
-    "\n",
-    "\n"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 1,
-   "id": "2e1f9bfa",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import numpy as np\n",
-    "import pandas as pd\n",
-    "from scipy.io import loadmat\n"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "37c6709b",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "#data_path\n",
-    "data_path = \"../data/\"\n",
-    "suv_filename = data_path + \"suv_percentilesSLOthenUWM.mat\"\n",
-    "flags_filename=  data_path + \"flags_combined.mat\"\n",
-    "normal_range_filename =  data_path + \"normal_range.mat\""
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "b103dce8",
-   "metadata": {},
-   "source": [
-    "Step 1: Data Preprocessing\n",
-    "Ensure that your data is in the correct format. \n",
-    "\n",
-    "We'll assume that:\n",
-    "\n",
-    "suv_percentiles is a matrix of SUV values.\n",
-    "\n",
-    "flags_combined contains the labels (1 for adverse effects and 0 for normal controls)."
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "base",
-   "language": "python",
-   "name": "python3"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.12.7"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}

Diferenças do arquivo suprimidas por serem muito extensas
+ 13 - 12
python/paper-logit_regression.ipynb


Alguns arquivos não foram mostrados porque muitos arquivos mudaram nesse diff