From 0ae09130be2da543c0ea0b2f571b86f878844358 Mon Sep 17 00:00:00 2001 From: Pritimay Sarkar Date: Wed, 8 Nov 2023 14:04:39 +0530 Subject: [PATCH] test model with sample data --- scripts/logistic2.py | 116 +++++++++++++++++++++++-------------------- 1 file changed, 61 insertions(+), 55 deletions(-) diff --git a/scripts/logistic2.py b/scripts/logistic2.py index ce664be..baa4ade 100644 --- a/scripts/logistic2.py +++ b/scripts/logistic2.py @@ -2,85 +2,91 @@ import pandas as pd from sklearn.model_selection import train_test_split from sklearn.linear_model import LogisticRegression from sklearn.preprocessing import LabelEncoder -from sklearn.metrics import accuracy_score, classification_report, confusion_matrix +from sklearn.metrics import accuracy_score, classification_report, confusion_matrix, roc_curve, roc_auc_score import pickle import statsmodels.api as sm import matplotlib.pyplot as plt import seaborn as sns import os -curdir = os.getcwd() -path_delim = '/' -data = pd.read_excel(curdir + path_delim + 'data/tests_24_10_2023_20_53_cleaned.xlsx', sheet_name="data") -data = data.dropna() -print(data) +# curdir = os.getcwd() +# path_delim = '/' +# data = pd.read_excel(curdir + path_delim + 'data/tests_24_10_2023_20_53_cleaned.xlsx', sheet_name="data") +# data = data.dropna() +# # print(data) -data.plot() +# data.plot() -label_encoder = LabelEncoder() -categorical_cols = ['deviceId', 'led1Buffer', 'led1Sample', 'led2Buffer', 'led2Sample'] -for col in categorical_cols: - data[col] = label_encoder.fit_transform(data[col]) +# label_encoder = LabelEncoder() +# categorical_cols = ['calculatedRatio', 'led1Buffer', 'led1Sample', 'led2Buffer', 'led2Sample'] +# # for col in categorical_cols: +# # data[col] = label_encoder.fit_transform(data[col]) -X = data[['deviceId', 'led1Buffer', 'led1Sample', 'led2Buffer', 'led2Sample']] -y = data['classificationResult'] +# X = data[['calculatedRatio', 'led1Buffer', 'led1Sample', 'led2Buffer', 'led2Sample']] +# y = data['classificationResult'] -corr = X.corr() -print(corr) -sm.graphics.plot_corr(corr, xnames=list(corr.columns)) -plt.show() +# corr = X.corr() +# print(corr) +# sm.graphics.plot_corr(corr, xnames=list(corr.columns)) +# # plt.show() -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42) +# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42) -model = LogisticRegression() -model.fit(X_train, y_train) +# model = LogisticRegression() +# model.fit(X_train, y_train) -y_pred = model.predict(X_test) +# y_pred = model.predict(X_test) -accuracy = accuracy_score(y_test, y_pred) -classification_report_result = classification_report(y_test, y_pred) -#sns.heatmap(pd.DataFrame(classification_report_result).iloc[:-1, :].T, annot=True) +# accuracy = accuracy_score(y_test, y_pred) +# classification_report_result = classification_report(y_test, y_pred) +# # sns.heatmap(pd.DataFrame(classification_report_result).iloc[:-1, :].T, annot=True) -# Calculate the confusion matrix -confusion = confusion_matrix(y_test, y_pred) +# # Calculate the confusion matrix +# confusion = confusion_matrix(y_test, y_pred) -# Plot the confusion matrix using Seaborn -plt.figure(figsize=(8, 6)) -sns.heatmap(confusion, annot=True, fmt='d', cmap='Blues', linewidths=0.5) -plt.xlabel('Predicted') -plt.ylabel('Actual') -plt.title('Confusion Matrix') -plt.show() +# # Plot the confusion matrix using Seaborn +# plt.figure(figsize=(8, 6)) +# sns.heatmap(confusion, annot=True, fmt='d', cmap='Blues', linewidths=0.5) +# plt.xlabel('Predicted') +# plt.ylabel('Actual') +# plt.title('Confusion Matrix') +# plt.show() -print(f"Accuracy: {accuracy}") -print("Classification Report:") -print(classification_report_result) +# print(f"Accuracy: {accuracy}") +# print("Classification Report:") +# print(classification_report_result) -# Accuracy: 0.6171938361719383 -# Classification Report: -# precision recall f1-score support +# # dataset: [6163 rows x 15 columns] +# # Accuracy: 0.6593673965936739 +# # Classification Report: +# # precision recall f1-score support -# Inconclusive. Repeat with test with lower volume of blood 0.00 0.00 0.00 6 -# Inconclusive. Very low Absorbance - Repeat test with Higher Blood Volume 0.17 0.05 0.07 21 -# Negative Borderline. Repeat Test 0.00 0.00 0.00 167 -# Normal 0.69 0.89 0.78 749 -# Positive for Sickle Cell. HPLC for Confirmation 0.00 0.00 0.00 41 -# Sickle Cell Disease 0.28 0.24 0.26 38 -# Sickle Cell Trait 0.37 0.39 0.38 211 - -# accuracy 0.62 1233 -# macro avg 0.22 0.22 0.21 1233 -# weighted avg 0.49 0.62 0.55 1233 +# # Inconclusive. Repeat with test with lower volume of blood 0.00 0.00 0.00 6 +# # Inconclusive. Very low Absorbance - Repeat test with Higher Blood Volume 0.88 0.71 0.79 21 +# # Negative Borderline. Repeat Test 0.11 0.01 0.01 167 +# # Normal 0.80 0.88 0.84 749 +# # Positive for Sickle Cell. HPLC for Confirmation 0.00 0.00 0.00 41 +# # Sickle Cell Disease 0.25 0.16 0.19 38 +# # Sickle Cell Trait 0.37 0.62 0.46 211 +# # accuracy 0.66 1233 +# # macro avg 0.34 0.34 0.33 1233 +# # weighted avg 0.58 0.66 0.61 1233 # with open('logistic_regression_model.pkl', 'wb') as model_file: # pickle.dump(model, model_file) -# with open('logistic_regression_model.pkl', 'rb') as model_file: -# loaded_model = pickle.load(model_file) +with open('logistic_regression_model.pkl', 'rb') as model_file: + loaded_model = pickle.load(model_file) -# new_data = pd.DataFrame({'Age': [30], 'Gender': ['MALE'], 'Caste': ['SC'], 'Category': [''], 'Marital Status': ['Single']}) -# predicted_result = loaded_model.predict(new_data) -# print(predicted_result) +# new_data = pd.DataFrame({'calculatedRatio': [0.126976079], 'led1Buffer': [24313.67], 'led1Sample': [20531], 'led2Buffer': [26565], 'led2Sample': [9975.33]}) #Normal +# new_data = pd.DataFrame({'calculatedRatio': [0.17500836], 'led1Buffer': [24843.33], 'led1Sample': [19678], 'led2Buffer': [26715.33], 'led2Sample': [13842.67]}) #SCT +# new_data = pd.DataFrame({'calculatedRatio': [0.251395102], 'led1Buffer': [24244], 'led1Sample': [17475], 'led2Buffer': [27059.67], 'led2Sample': [9258.33]}) #SCD +# new_data = pd.DataFrame({'calculatedRatio': [0.251061035], 'led1Buffer': [24172], 'led1Sample': [19345], 'led2Buffer': [26918], 'led2Sample': [12964.33]}) +# new_data = pd.DataFrame({'calculatedRatio': [0.242851779], 'led1Buffer': [25087.33], 'led1Sample': [20170.67], 'led2Buffer': [26578.33], 'led2Sample': [12724.67]}) +# new_data = pd.DataFrame({'calculatedRatio': [0.189778691], 'led1Buffer': [23209.33], 'led1Sample': [17672], 'led2Buffer': [19850.33], 'led2Sample': [10360.33]}) #SCT +new_data = pd.DataFrame({'calculatedRatio': [0.149174719], 'led1Buffer': [24122], 'led1Sample': [18368.33], 'led2Buffer': [21733.33], 'led2Sample': [9114.67]}) #Normal +predicted_result = loaded_model.predict(new_data) +print(predicted_result)