Files
2024-05-21 12:10:53 +03:00

146 lines
5.7 KiB
Python

import json
import logging
import sys
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.ensemble import RandomForestClassifier
from sklearn.preprocessing import OneHotEncoder
import numpy as np
import re
import os
class DiseasePredictor:
def __init__(self, csv_path):
logging.info("\n\nInitializing DiseasePredictor")
self.model = RandomForestClassifier(random_state=42)
self.data = pd.read_csv(csv_path)
self.features = self.data.columns[0:-1] # Exclude the outcome variable
self.outcome = self.data.columns[0] # The outcome variable
self.categorical_features = self.data[self.features].select_dtypes(include=['object']).columns.tolist()
self.numerical_features = self.data[self.features].select_dtypes(exclude=['object']).columns.tolist()
self.one_hot_encoder = OneHotEncoder(handle_unknown='ignore') # Set handle_unknown to 'ignore'
self.train()
def train(self):
logging.info("Training model")
# Preprocess the data
X = self.data[self.features].copy()
y = self.data[self.outcome]
# One-hot encode the categorical variables
X_encoded = pd.get_dummies(X, columns=self.categorical_features)
# Use the DataFrame directly instead of converting to NumPy array
# This maintains the feature names
self.feature_names = X_encoded.columns.tolist()
self.one_hot_encoder.fit(X[self.categorical_features])
# Split the data
self.X_train, self.X_test, self.y_train, self.y_test = train_test_split(X_encoded, y, test_size=0.2,
random_state=42)
# Train the model
self.model.fit(self.X_train, self.y_train) # X_train is a DataFrame with feature names
logging.info(f"Training accuracy: {self.model.score(self.X_test, self.y_test):.2f}")
def extract_features(self, text):
logging.info("Extracting features from text")
feature_values = {}
for feature in self.features:
if feature != 'Age' and feature != 'Gender':
# These are the binary categorical features
feature_values[feature] = 'Yes' if feature.lower() in text.lower() else 'No'
# Extract age if mentioned
age_search = re.search(r'(\d+)\s*years?', text)
feature_values['Age'] = int(age_search.group(1)) if age_search else np.nan
# Extract gender if mentioned
if 'male' in text.lower():
feature_values['Gender'] = 'Male'
elif 'female' in text.lower():
feature_values['Gender'] = 'Female'
else:
feature_values['Gender'] = np.nan # Missing value
logging.info(f"Tokens extracted: {feature_values}")
# Convert the features into a DataFrame to be consistent with the model's input format
features_df = pd.DataFrame([feature_values])
# One-hot encode the extracted features
aligned_features_df = pd.DataFrame(columns=self.feature_names)
aligned_features_df = aligned_features_df._append(features_df, ignore_index=True)
# Fill missing columns with zeros
missing_cols = set(self.feature_names) - set(aligned_features_df.columns)
for c in missing_cols:
aligned_features_df[c] = 0
# Ensure the order of columns matches the training data
aligned_features_df = aligned_features_df[self.feature_names]
# Now we no longer need to one-hot encode 'Age' since it was not one-hot encoded during training
return aligned_features_df
def predict(self, text):
logging.info("Making a prediction")
features_df_encoded = self.extract_features(text)
predictions = self.model.predict_proba(features_df_encoded)
# Create a dictionary to store the probabilities for each disease class
disease_probabilities = {}
for disease_class, probability in zip(self.model.classes_, predictions[0]):
# Map probabilities from 0 to 100 and round to two decimal places
mapped_probability = round(probability * 100, 2)
disease_probabilities[disease_class.lower()] = mapped_probability
# Sort the disease probabilities in descending order and select the top three
sorted_probabilities = dict(sorted(disease_probabilities.items(), key=lambda item: item[1], reverse=True)[:3])
# Generate JSON output
json_output = {
"StatusCode": 200,
"Message": "Prediction successfully computed.",
"Data": []
}
count = 0
for key, value in sorted_probabilities.items():
# Append each key and value as a dictionary to ensure proper JSON object format
json_output["Data"].append({"disease": key, "probability": value})
count += 1 # Increment the counter
if count == 3:
break
# Return the JSON object
return json.dumps(json_output)
# Example usage:
if __name__ == "__main__":
if len(sys.argv) != 2:
logging.error("Incorrect number of arguments provided.")
json_output = {
"StatusCode": 400,
"Message": "Incorrect number of arguments provided.",
"Data": []
}
print(json.dumps(json_output))
sys.exit(1)
logging.basicConfig(filename='disease_prediction.log', level=logging.INFO,
format='%(asctime)s %(levelname)s:%(message)s')
text = sys.argv[1]
script_directory = os.path.dirname(os.path.abspath(__file__))
csv_path = os.path.join(script_directory, 'dataset.csv')
predictor = DiseasePredictor(csv_path)
prediction_json = predictor.predict(text)
print(prediction_json)