create dataset
This commit is contained in:
36
scripts/dataset_001.py
Normal file
36
scripts/dataset_001.py
Normal file
@@ -0,0 +1,36 @@
|
||||
import pandas as pd
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn.naive_bayes import GaussianNB
|
||||
from sklearn.preprocessing import LabelEncoder
|
||||
from sklearn.metrics import accuracy_score, classification_report, confusion_matrix
|
||||
import pickle
|
||||
import matplotlib.pyplot as plt
|
||||
import seaborn as sns
|
||||
import os
|
||||
|
||||
curdir = os.getcwd()
|
||||
path_delim = '/'
|
||||
df = pd.read_csv(curdir + path_delim + 'data/bquxjob_6ffc1cff_18c02a130c1.csv')
|
||||
df = df.dropna()
|
||||
print(df)
|
||||
|
||||
print(df["classificationResult"].unique())
|
||||
|
||||
df['testResult'] = df['classificationResult']
|
||||
df.loc[df['testResult'] == 'Normal', 'testResult'] = 'Normal'
|
||||
df.loc[df['testResult'] == 'Sickle Cell Trait', 'testResult'] = 'SCT'
|
||||
df.loc[df['testResult'] == 'Sickle Cell Disease', 'testResult'] = 'SCD'
|
||||
df.loc[df['testResult'] == 'Positive for Sickle Cell. HPLC for Confirmation', 'testResult'] = 'Inconclusive'
|
||||
df.loc[df['testResult'] == 'Inconclusive. Very low Absorbance - Repeat test with Higher Blood Volume', 'testResult'] = 'Inconclusive'
|
||||
df.loc[df['testResult'] == 'Negative Borderline. Repeat Test', 'testResult'] = 'Inconclusive'
|
||||
df.loc[df['testResult'] == 'Inconclusive. Very low Absorbance - Repeat test with Higher Blood Volume', 'testResult'] = 'Inconclusive'
|
||||
df.loc[df['testResult'] == 'Inconclusive. Repeat with test with lower volume of blood', 'testResult'] = 'Inconclusive'
|
||||
print(df.groupby(["testResult"]).describe())
|
||||
|
||||
df = df.drop(['classificationResult', "finalResult"], axis=1)
|
||||
|
||||
writer = pd.ExcelWriter(curdir + path_delim + "data/vertex_dataset3.xlsx", engine = 'openpyxl')
|
||||
df.to_excel(writer, sheet_name = 'op', index=False)
|
||||
df.to_csv(curdir + path_delim + "data/vertex_dataset3.csv", index=False)
|
||||
# df_count.to_excel(writer, sheet_name = "count")
|
||||
writer.close()
|
||||
Reference in New Issue
Block a user