Files
DCS-211/Classification Lab/inclass_lab.py
2026-04-01 10:37:07 -04:00

269 lines
7.3 KiB
Python

# -*- coding: utf-8 -*-
"""Inclass-Lab
Automatically generated by Colab.
Original file is located at
https://colab.research.google.com/drive/13N7lQHvv4_LxKgcerumJ-18e5GfdBggm
"""
"""In-Class Lab: Comparing Classification Models (Breast Cancer Dataset)
Learning Objectives
---
By the end of this lab, you should be able to:
* Train multiple models (KNN, Logistic Regression, Decision Tree)
* Compare performance across various scenarios
* Understand the effect of scaling and evaluation methods
* Interpret confusion matrix & classification report
"""
####################################
# BLOCK 1:
'''
Load dataset
Explain features/target
Do train_test_split
'''
####################################
from sklearn.datasets import load_breast_cancer
data = load_breast_cancer()
X = data.data
y = data.target
print(data.target_names)
print(X.shape)
#0 - malignant
#1 - benigh
###############################
# BLOCK 2: IMPORT LIBRARIES FOR CLASSIFIERS
###############################
from sklearn.neighbors import KNeighborsClassifier
from sklearn.linear_model import LogisticRegression
from sklearn.tree import DecisionTreeClassifier
#########################################
# BLOCK 3: IMPORT MODUL FOR DATA SPLIT
#########################################
from sklearn.model_selection import train_test_split
####################################
# BLOCK 4: SPLIT THE DATASET
####################################
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
####################################
# BLOCK 5
#TASK 1: Train 3 models (no scaling)
####################################
knn = KNeighborsClassifier(n_neighbors=5)
logreg = LogisticRegression(max_iter=5000)
tree = DecisionTreeClassifier()
knn.fit(X_train, y_train)
logreg.fit(X_train, y_train)
tree.fit(X_train, y_train)
####################################
# BLOCK 6
# Task 2: Evaluate accuracy
'''
* Which model performs best?
* Are the results very different?
'''
####################################
print("KNN:", knn.score(X_test, y_test))
print("LogReg:", logreg.score(X_test, y_test))
print("Tree:", tree.score(X_test, y_test))
####################################
#BLOCK 7
# Part 2: Evaluation Metrics
# Task 3: Confusion Matrix
from sklearn.metrics import confusion_matrix
print(confusion_matrix(y_test, knn.predict(X_test)))
#####################################
# BLOCK 8: CLASSIFICATION FREPORT
'''
* Which model has better recall?
* Which is better for detecting cancer?
'''
#####################################
from sklearn.metrics import classification_report
print(classification_report(y_test, knn.predict(X_test)))
########################################
# BLOCK 9:
# Part 3: Scaling Effect
# Task 5: Apply scaling using pipeline
########################################
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import StandardScaler
from sklearn.metrics import accuracy_score, confusion_matrix, classification_report
knn_scaled = Pipeline([
('scaler', StandardScaler()),
('knn', KNeighborsClassifier(n_neighbors=5))
])
logreg_scaled = Pipeline([
('scaler', StandardScaler()),
('logreg', LogisticRegression(max_iter=5000))
])
tree_scaled = Pipeline([
('scaler', StandardScaler()),
('tree', DecisionTreeClassifier(random_state=42))
])
knn_scaled.fit(X_train, y_train)
logreg_scaled.fit(X_train, y_train)
tree_scaled.fit(X_train, y_train)
knn_pred_scaled = knn_scaled.predict(X_test)
logreg_pred_scaled = logreg_scaled.predict(X_test)
tree_pred_scaled = tree_scaled.predict(X_test)
print("\n=== RESULTS WITH SCALING ===")
print("KNN Accuracy:", accuracy_score(y_test, knn_pred_scaled))
print("Logistic Regression Accuracy:", accuracy_score(y_test, logreg_pred_scaled))
print("Decision Tree Accuracy:", accuracy_score(y_test, tree_pred_scaled))
#####################################
# BLOCK 10: CLASSIFICATION REPORT
#####################################
print("\n=== CLASSIFICATION REPORTS (WITH SCALING) ===")
print("\nKNN Report:")
print(classification_report(y_test, knn_pred_scaled))
print("\nLogistic Regression Report:")
print(classification_report(y_test, logreg_pred_scaled))
print("\nDecision Tree Report:")
print(classification_report(y_test, tree_pred_scaled))
####################################
# BLOCK 11: CONFUSION MATRIX
####################################
print("\n=== CONFUSION MATRICES (WITH SCALING) ===")
print("\nKNN Confusion Matrix:")
print(confusion_matrix(y_test, knn_pred_scaled))
print("\nLogistic Regression Confusion Matrix:")
print(confusion_matrix(y_test, logreg_pred_scaled))
print("\nDecision Tree Confusion Matrix:")
print(confusion_matrix(y_test, tree_pred_scaled))
###############################################
# BLOCK 12: MODWLS WITH CROSS-VALIDATION
###############################################
from sklearn.model_selection import cross_val_score
from sklearn.preprocessing import StandardScaler
knn_pipeline = Pipeline([
('scaler', StandardScaler()),
('knn', KNeighborsClassifier(n_neighbors=5))
])
logreg_pipeline = Pipeline([
('scaler', StandardScaler()),
('logreg', LogisticRegression(max_iter=5000))
])
tree_model = DecisionTreeClassifier(random_state=42)
knn_cv_scores = cross_val_score(knn_pipeline, X, y, cv=5, scoring='accuracy')
logreg_cv_scores = cross_val_score(logreg_pipeline, X, y, cv=5, scoring='accuracy')
tree_cv_scores = cross_val_score(tree_model, X, y, cv=5, scoring='accuracy')
print("=== Cross-Validation Results ===")
print("\nKNN")
print("CV Scores:", knn_cv_scores)
print("Mean Accuracy:", np.mean(knn_cv_scores))
print("Standard Deviation:", np.std(knn_cv_scores))
print("\nLogistic Regression")
print("CV Scores:", logreg_cv_scores)
print("Mean Accuracy:", np.mean(logreg_cv_scores))
print("Standard Deviation:", np.std(logreg_cv_scores))
print("\nDecision Tree")
print("CV Scores:", tree_cv_scores)
print("Mean Accuracy:", np.mean(tree_cv_scores))
print("Standard Deviation:", np.std(tree_cv_scores))
####################################
# BLOCK 13: CONFUSION MATRIX
####################################
from sklearn.model_selection import cross_val_predict
from sklearn.metrics import confusion_matrix, classification_report
knn_pred = cross_val_predict(knn_pipeline, X, y, cv=5)
logreg_pred = cross_val_predict(logreg_pipeline, X, y, cv=5)
tree_pred = cross_val_predict(tree_model, X, y, cv=5)
print("\n=== Confusion Matrices ===")
print("\nKNN")
print(confusion_matrix(y, knn_pred))
print("\nLogistic Regression")
print(confusion_matrix(y, logreg_pred))
print("\nDecision Tree")
print(confusion_matrix(y, tree_pred))
print("\n=== Classification Reports ===")
print("\nKNN")
print(classification_report(y, knn_pred))
print("\nLogistic Regression")
print(classification_report(y, logreg_pred))
print("\nDecision Tree")
print(classification_report(y, tree_pred))
###########################################
# BLOCK 14: LETS GET THE BEST VALUE FOR K
# task: use the k value to recompute the previous KNN model and evaluate the performance
##########################################
from sklearn.model_selection import cross_val_score
import numpy as np
k_values = range(1, 21)
scores = []
for k in k_values:
knn = KNeighborsClassifier(n_neighbors=k)
cv_scores = cross_val_score(knn, X, y, cv=5)
scores.append(np.mean(cv_scores))
best_k = k_values[np.argmax(scores)]
print("Best k:", best_k)
print("Best CV Accuracy:", max(scores))