import numpy as np from sklearn.model_selection import train_test_split from sklearn.ensemble import RandomForestClassifier from sklearn.neighbors import KNeighborsClassifier from sklearn.metrics import roc_curve, roc_auc_score import matplotlib.pyplot as plt # Set a random seed for reproducibility np.random.seed(42) # Generate random data num_samples = 1000 num_features = 10 # Generate features (X) X = np.random.randn(num_samples, num_features) # Generate labels (y) y = np.random.randint(2, size=num_samples) # Standardize the features X = (X - np.mean(X, axis=0)) / np.std(X, axis=0) # Split the data into train and test sets X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42) # Fit the train set using random forest classifier rf_clf = RandomForestClassifier() rf_clf.fit(X_train, y_train) # Fit the train set using K-nearest neighbors (KNN) classifier knn_clf = KNeighborsClassifier() knn_clf.fit(X_train, y_train) # Predict probabilities for the positive class (class 1) y_train_rf_probs = rf_clf.predict_proba(X_train)[:, 1] y_train_knn_probs = knn_clf.predict_proba(X_train)[:, 1] # Calculate the false positive rate (FPR), true positive rate (TPR), and thresholds for the ROC curve rf_fpr, rf_tpr, rf_thresholds = roc_curve(y_train, y_train_rf_probs) knn_fpr, knn_tpr, knn_thresholds = roc_curve(y_train, y_train_knn_probs) # Calculate the AUC score for the ROC curve rf_auc = roc_auc_score(y_train, y_train_rf_probs) knn_auc = roc_auc_score(y_train, y_train_knn_probs) # Plot the ROC curve plt.plot(rf_fpr, rf_tpr, label=f"Random Forest (AUC = {rf_auc:.2f})") plt.plot(knn_fpr, knn_tpr, label=f"KNN (AUC = {knn_auc:.2f})") plt.plot([0, 1], [0, 1], 'k--') # Diagonal line for random classifier plt.xlabel('False Positive Rate') plt.ylabel('True Positive Rate') plt.title('ROC Curve - Train Set') plt.legend(loc='lower right') plt.show()