From d06c904f4bb39b5c245e58054774fc0e46b40e52 Mon Sep 17 00:00:00 2001 From: Benjamin Date: Wed, 1 Apr 2026 10:37:07 -0400 Subject: [PATCH] in class lab template added --- Classification Lab/inclass_lab.py | 268 ++++++++++++++++++++++++++++++ 1 file changed, 268 insertions(+) create mode 100644 Classification Lab/inclass_lab.py diff --git a/Classification Lab/inclass_lab.py b/Classification Lab/inclass_lab.py new file mode 100644 index 0000000..dbe687f --- /dev/null +++ b/Classification Lab/inclass_lab.py @@ -0,0 +1,268 @@ +# -*- coding: utf-8 -*- +"""Inclass-Lab + +Automatically generated by Colab. + +Original file is located at + https://colab.research.google.com/drive/13N7lQHvv4_LxKgcerumJ-18e5GfdBggm +""" + + + +"""In-Class Lab: Comparing Classification Models (Breast Cancer Dataset) + +Learning Objectives + +--- + + +By the end of this lab, you should be able to: + + + +* Train multiple models (KNN, Logistic Regression, Decision Tree) +* Compare performance across various scenarios +* Understand the effect of scaling and evaluation methods +* Interpret confusion matrix & classification report + + + + + + +""" + +#################################### +# BLOCK 1: +''' +Load dataset +Explain features/target +Do train_test_split +''' +#################################### +from sklearn.datasets import load_breast_cancer +data = load_breast_cancer() +X = data.data +y = data.target + +print(data.target_names) +print(X.shape) +#0 - malignant +#1 - benigh + +############################### +# BLOCK 2: IMPORT LIBRARIES FOR CLASSIFIERS +############################### +from sklearn.neighbors import KNeighborsClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.tree import DecisionTreeClassifier + +######################################### +# BLOCK 3: IMPORT MODUL FOR DATA SPLIT +######################################### +from sklearn.model_selection import train_test_split + +#################################### +# BLOCK 4: SPLIT THE DATASET +#################################### +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42) + +#################################### +# BLOCK 5 +#TASK 1: Train 3 models (no scaling) +#################################### +knn = KNeighborsClassifier(n_neighbors=5) +logreg = LogisticRegression(max_iter=5000) +tree = DecisionTreeClassifier() + +knn.fit(X_train, y_train) +logreg.fit(X_train, y_train) +tree.fit(X_train, y_train) + +#################################### +# BLOCK 6 +# Task 2: Evaluate accuracy +''' +* Which model performs best? +* Are the results very different? +''' +#################################### +print("KNN:", knn.score(X_test, y_test)) +print("LogReg:", logreg.score(X_test, y_test)) +print("Tree:", tree.score(X_test, y_test)) + +#################################### +#BLOCK 7 +# Part 2: Evaluation Metrics +# Task 3: Confusion Matrix + +from sklearn.metrics import confusion_matrix +print(confusion_matrix(y_test, knn.predict(X_test))) + +##################################### +# BLOCK 8: CLASSIFICATION FREPORT +''' +* Which model has better recall? +* Which is better for detecting cancer? +''' +##################################### +from sklearn.metrics import classification_report +print(classification_report(y_test, knn.predict(X_test))) + +######################################## +# BLOCK 9: +# Part 3: Scaling Effect +# Task 5: Apply scaling using pipeline +######################################## + +from sklearn.pipeline import Pipeline +from sklearn.preprocessing import StandardScaler +from sklearn.metrics import accuracy_score, confusion_matrix, classification_report + +knn_scaled = Pipeline([ + ('scaler', StandardScaler()), + ('knn', KNeighborsClassifier(n_neighbors=5)) +]) + +logreg_scaled = Pipeline([ + ('scaler', StandardScaler()), + ('logreg', LogisticRegression(max_iter=5000)) +]) + +tree_scaled = Pipeline([ + ('scaler', StandardScaler()), + ('tree', DecisionTreeClassifier(random_state=42)) +]) + +knn_scaled.fit(X_train, y_train) +logreg_scaled.fit(X_train, y_train) +tree_scaled.fit(X_train, y_train) + +knn_pred_scaled = knn_scaled.predict(X_test) +logreg_pred_scaled = logreg_scaled.predict(X_test) +tree_pred_scaled = tree_scaled.predict(X_test) + +print("\n=== RESULTS WITH SCALING ===") +print("KNN Accuracy:", accuracy_score(y_test, knn_pred_scaled)) +print("Logistic Regression Accuracy:", accuracy_score(y_test, logreg_pred_scaled)) +print("Decision Tree Accuracy:", accuracy_score(y_test, tree_pred_scaled)) + +##################################### +# BLOCK 10: CLASSIFICATION REPORT +##################################### +print("\n=== CLASSIFICATION REPORTS (WITH SCALING) ===") + +print("\nKNN Report:") +print(classification_report(y_test, knn_pred_scaled)) + +print("\nLogistic Regression Report:") +print(classification_report(y_test, logreg_pred_scaled)) + +print("\nDecision Tree Report:") +print(classification_report(y_test, tree_pred_scaled)) + +#################################### +# BLOCK 11: CONFUSION MATRIX +#################################### +print("\n=== CONFUSION MATRICES (WITH SCALING) ===") + +print("\nKNN Confusion Matrix:") +print(confusion_matrix(y_test, knn_pred_scaled)) + +print("\nLogistic Regression Confusion Matrix:") +print(confusion_matrix(y_test, logreg_pred_scaled)) + +print("\nDecision Tree Confusion Matrix:") +print(confusion_matrix(y_test, tree_pred_scaled)) + + + +############################################### +# BLOCK 12: MODWLS WITH CROSS-VALIDATION +############################################### +from sklearn.model_selection import cross_val_score +from sklearn.preprocessing import StandardScaler + +knn_pipeline = Pipeline([ + ('scaler', StandardScaler()), + ('knn', KNeighborsClassifier(n_neighbors=5)) +]) + +logreg_pipeline = Pipeline([ + ('scaler', StandardScaler()), + ('logreg', LogisticRegression(max_iter=5000)) +]) + +tree_model = DecisionTreeClassifier(random_state=42) + +knn_cv_scores = cross_val_score(knn_pipeline, X, y, cv=5, scoring='accuracy') +logreg_cv_scores = cross_val_score(logreg_pipeline, X, y, cv=5, scoring='accuracy') +tree_cv_scores = cross_val_score(tree_model, X, y, cv=5, scoring='accuracy') + +print("=== Cross-Validation Results ===") + +print("\nKNN") +print("CV Scores:", knn_cv_scores) +print("Mean Accuracy:", np.mean(knn_cv_scores)) +print("Standard Deviation:", np.std(knn_cv_scores)) + +print("\nLogistic Regression") +print("CV Scores:", logreg_cv_scores) +print("Mean Accuracy:", np.mean(logreg_cv_scores)) +print("Standard Deviation:", np.std(logreg_cv_scores)) + +print("\nDecision Tree") +print("CV Scores:", tree_cv_scores) +print("Mean Accuracy:", np.mean(tree_cv_scores)) +print("Standard Deviation:", np.std(tree_cv_scores)) + +#################################### +# BLOCK 13: CONFUSION MATRIX +#################################### +from sklearn.model_selection import cross_val_predict +from sklearn.metrics import confusion_matrix, classification_report + +knn_pred = cross_val_predict(knn_pipeline, X, y, cv=5) +logreg_pred = cross_val_predict(logreg_pipeline, X, y, cv=5) +tree_pred = cross_val_predict(tree_model, X, y, cv=5) + +print("\n=== Confusion Matrices ===") +print("\nKNN") +print(confusion_matrix(y, knn_pred)) + +print("\nLogistic Regression") +print(confusion_matrix(y, logreg_pred)) + +print("\nDecision Tree") +print(confusion_matrix(y, tree_pred)) + +print("\n=== Classification Reports ===") +print("\nKNN") +print(classification_report(y, knn_pred)) + +print("\nLogistic Regression") +print(classification_report(y, logreg_pred)) + +print("\nDecision Tree") +print(classification_report(y, tree_pred)) + +########################################### +# BLOCK 14: LETS GET THE BEST VALUE FOR K +# task: use the k value to recompute the previous KNN model and evaluate the performance +########################################## +from sklearn.model_selection import cross_val_score +import numpy as np + +k_values = range(1, 21) +scores = [] + +for k in k_values: + knn = KNeighborsClassifier(n_neighbors=k) + cv_scores = cross_val_score(knn, X, y, cv=5) + scores.append(np.mean(cv_scores)) + +best_k = k_values[np.argmax(scores)] + +print("Best k:", best_k) +print("Best CV Accuracy:", max(scores)) +