# -*- coding: utf-8 -*- """Inclass-Lab Automatically generated by Colab. Original file is located at https://colab.research.google.com/drive/13N7lQHvv4_LxKgcerumJ-18e5GfdBggm """ """In-Class Lab: Comparing Classification Models (Breast Cancer Dataset) Learning Objectives --- By the end of this lab, you should be able to: * Train multiple models (KNN, Logistic Regression, Decision Tree) * Compare performance across various scenarios * Understand the effect of scaling and evaluation methods * Interpret confusion matrix & classification report """ #################################### # BLOCK 1: ''' Load dataset Explain features/target Do train_test_split ''' #################################### from sklearn.datasets import load_breast_cancer data = load_breast_cancer() X = data.data y = data.target print(data.target_names) print(X.shape) #0 - malignant #1 - benigh ############################### # BLOCK 2: IMPORT LIBRARIES FOR CLASSIFIERS ############################### from sklearn.neighbors import KNeighborsClassifier from sklearn.linear_model import LogisticRegression from sklearn.tree import DecisionTreeClassifier ######################################### # BLOCK 3: IMPORT MODUL FOR DATA SPLIT ######################################### from sklearn.model_selection import train_test_split #################################### # BLOCK 4: SPLIT THE DATASET #################################### X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42) #################################### # BLOCK 5 #TASK 1: Train 3 models (no scaling) #################################### knn = KNeighborsClassifier(n_neighbors=5) logreg = LogisticRegression(max_iter=5000) tree = DecisionTreeClassifier() knn.fit(X_train, y_train) logreg.fit(X_train, y_train) tree.fit(X_train, y_train) #################################### # BLOCK 6 # Task 2: Evaluate accuracy ''' * Which model performs best? * Are the results very different? ''' #################################### print("KNN:", knn.score(X_test, y_test)) print("LogReg:", logreg.score(X_test, y_test)) print("Tree:", tree.score(X_test, y_test)) #################################### #BLOCK 7 # Part 2: Evaluation Metrics # Task 3: Confusion Matrix from sklearn.metrics import confusion_matrix print(confusion_matrix(y_test, knn.predict(X_test))) ##################################### # BLOCK 8: CLASSIFICATION FREPORT ''' * Which model has better recall? * Which is better for detecting cancer? ''' ##################################### from sklearn.metrics import classification_report print(classification_report(y_test, knn.predict(X_test))) ######################################## # BLOCK 9: # Part 3: Scaling Effect # Task 5: Apply scaling using pipeline ######################################## from sklearn.pipeline import Pipeline from sklearn.preprocessing import StandardScaler from sklearn.metrics import accuracy_score, confusion_matrix, classification_report knn_scaled = Pipeline([ ('scaler', StandardScaler()), ('knn', KNeighborsClassifier(n_neighbors=5)) ]) logreg_scaled = Pipeline([ ('scaler', StandardScaler()), ('logreg', LogisticRegression(max_iter=5000)) ]) tree_scaled = Pipeline([ ('scaler', StandardScaler()), ('tree', DecisionTreeClassifier(random_state=42)) ]) knn_scaled.fit(X_train, y_train) logreg_scaled.fit(X_train, y_train) tree_scaled.fit(X_train, y_train) knn_pred_scaled = knn_scaled.predict(X_test) logreg_pred_scaled = logreg_scaled.predict(X_test) tree_pred_scaled = tree_scaled.predict(X_test) print("\n=== RESULTS WITH SCALING ===") print("KNN Accuracy:", accuracy_score(y_test, knn_pred_scaled)) print("Logistic Regression Accuracy:", accuracy_score(y_test, logreg_pred_scaled)) print("Decision Tree Accuracy:", accuracy_score(y_test, tree_pred_scaled)) ##################################### # BLOCK 10: CLASSIFICATION REPORT ##################################### print("\n=== CLASSIFICATION REPORTS (WITH SCALING) ===") print("\nKNN Report:") print(classification_report(y_test, knn_pred_scaled)) print("\nLogistic Regression Report:") print(classification_report(y_test, logreg_pred_scaled)) print("\nDecision Tree Report:") print(classification_report(y_test, tree_pred_scaled)) #################################### # BLOCK 11: CONFUSION MATRIX #################################### print("\n=== CONFUSION MATRICES (WITH SCALING) ===") print("\nKNN Confusion Matrix:") print(confusion_matrix(y_test, knn_pred_scaled)) print("\nLogistic Regression Confusion Matrix:") print(confusion_matrix(y_test, logreg_pred_scaled)) print("\nDecision Tree Confusion Matrix:") print(confusion_matrix(y_test, tree_pred_scaled)) ############################################### # BLOCK 12: MODWLS WITH CROSS-VALIDATION ############################################### from sklearn.model_selection import cross_val_score from sklearn.preprocessing import StandardScaler knn_pipeline = Pipeline([ ('scaler', StandardScaler()), ('knn', KNeighborsClassifier(n_neighbors=5)) ]) logreg_pipeline = Pipeline([ ('scaler', StandardScaler()), ('logreg', LogisticRegression(max_iter=5000)) ]) tree_model = DecisionTreeClassifier(random_state=42) knn_cv_scores = cross_val_score(knn_pipeline, X, y, cv=5, scoring='accuracy') logreg_cv_scores = cross_val_score(logreg_pipeline, X, y, cv=5, scoring='accuracy') tree_cv_scores = cross_val_score(tree_model, X, y, cv=5, scoring='accuracy') print("=== Cross-Validation Results ===") print("\nKNN") print("CV Scores:", knn_cv_scores) print("Mean Accuracy:", np.mean(knn_cv_scores)) print("Standard Deviation:", np.std(knn_cv_scores)) print("\nLogistic Regression") print("CV Scores:", logreg_cv_scores) print("Mean Accuracy:", np.mean(logreg_cv_scores)) print("Standard Deviation:", np.std(logreg_cv_scores)) print("\nDecision Tree") print("CV Scores:", tree_cv_scores) print("Mean Accuracy:", np.mean(tree_cv_scores)) print("Standard Deviation:", np.std(tree_cv_scores)) #################################### # BLOCK 13: CONFUSION MATRIX #################################### from sklearn.model_selection import cross_val_predict from sklearn.metrics import confusion_matrix, classification_report knn_pred = cross_val_predict(knn_pipeline, X, y, cv=5) logreg_pred = cross_val_predict(logreg_pipeline, X, y, cv=5) tree_pred = cross_val_predict(tree_model, X, y, cv=5) print("\n=== Confusion Matrices ===") print("\nKNN") print(confusion_matrix(y, knn_pred)) print("\nLogistic Regression") print(confusion_matrix(y, logreg_pred)) print("\nDecision Tree") print(confusion_matrix(y, tree_pred)) print("\n=== Classification Reports ===") print("\nKNN") print(classification_report(y, knn_pred)) print("\nLogistic Regression") print(classification_report(y, logreg_pred)) print("\nDecision Tree") print(classification_report(y, tree_pred)) ########################################### # BLOCK 14: LETS GET THE BEST VALUE FOR K # task: use the k value to recompute the previous KNN model and evaluate the performance ########################################## from sklearn.model_selection import cross_val_score import numpy as np k_values = range(1, 21) scores = [] for k in k_values: knn = KNeighborsClassifier(n_neighbors=k) cv_scores = cross_val_score(knn, X, y, cv=5) scores.append(np.mean(cv_scores)) best_k = k_values[np.argmax(scores)] print("Best k:", best_k) print("Best CV Accuracy:", max(scores))