Files
DCS-211/Classification Lab/inclass_lab.py
2026-04-01 11:33:35 -04:00

285 lines
8.1 KiB
Python

# -*- coding: utf-8 -*-
"""Inclass-Lab
Automatically generated by Colab.
Original file is located at
https://colab.research.google.com/drive/13N7lQHvv4_LxKgcerumJ-18e5GfdBggm
"""
"""In-Class Lab: Comparing Classification Models (Breast Cancer Dataset)
Learning Objectives
---
By the end of this lab, you should be able to:
* Train multiple models (KNN, Logistic Regression, Decision Tree)
* Compare performance across various scenarios
* Understand the effect of scaling and evaluation methods
* Interpret confusion matrix & classification report
"""
####################################
# BLOCK 1:
'''
Load dataset
Explain features/target
Do train_test_split
'''
####################################
from sklearn.datasets import load_breast_cancer
data = load_breast_cancer()
X = data.data
y = data.target
print(data.target_names)
print(X.shape)
#0 - malignant
#1 - benigh
'''
Answer: The dataset contains 569 samples with 30 features each. The target variable has two classes: 0 (malignant) and 1 (benign). The goal is to classify tumors as either malignant or benign based on the features provided.
'''
###############################
# BLOCK 2: IMPORT LIBRARIES FOR CLASSIFIERS
###############################
from sklearn.neighbors import KNeighborsClassifier
from sklearn.linear_model import LogisticRegression
from sklearn.tree import DecisionTreeClassifier
import numpy as np
#########################################
# BLOCK 3: IMPORT MODUL FOR DATA SPLIT
#########################################
from sklearn.model_selection import train_test_split
####################################
# BLOCK 4: SPLIT THE DATASET
####################################
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
####################################
# BLOCK 5
#TASK 1: Train 3 models (no scaling)
####################################
knn = KNeighborsClassifier(n_neighbors=5)
logreg = LogisticRegression(max_iter=5000)
tree = DecisionTreeClassifier()
knn.fit(X_train, y_train)
logreg.fit(X_train, y_train)
tree.fit(X_train, y_train)
####################################
# BLOCK 6
# Task 2: Evaluate accuracy
'''
* Which model performs best?
* Are the results very different?
'''
####################################
print("KNN:", knn.score(X_test, y_test))
print("LogReg:", logreg.score(X_test, y_test))
print("Tree:", tree.score(X_test, y_test))
'''
Answer: The results are as follows:
KNN: 0.956140350877193
LogReg: 0.956140350877193
Tree: 0.9298245614035088
They are all very similar, however LogReg and KNN perform slightly better than the Decision Tree in terms of accuracy on the test set. The difference is not very large, but it suggests that KNN and Logistic Regression may be more effective for this particular dataset.
'''
####################################
#BLOCK 7
# Part 2: Evaluation Metrics
# Task 3: Confusion Matrix
from sklearn.metrics import confusion_matrix
print(confusion_matrix(y_test, knn.predict(X_test)))
#####################################
# BLOCK 8: CLASSIFICATION FREPORT
'''
* Which model has better recall?
* Which is better for detecting cancer?
'''
'''
Answer: The recall for the model in Block 8 is 0.88 for malignant, so it correctly identifies 88% of cancer cases.
'''
#####################################
from sklearn.metrics import classification_report
print(classification_report(y_test, knn.predict(X_test)))
########################################
# BLOCK 9:
# Part 3: Scaling Effect
# Task 5: Apply scaling using pipeline
########################################
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import StandardScaler
from sklearn.metrics import accuracy_score, confusion_matrix, classification_report
knn_scaled = Pipeline([
('scaler', StandardScaler()),
('knn', KNeighborsClassifier(n_neighbors=5))
])
logreg_scaled = Pipeline([
('scaler', StandardScaler()),
('logreg', LogisticRegression(max_iter=5000))
])
tree_scaled = Pipeline([
('scaler', StandardScaler()),
('tree', DecisionTreeClassifier(random_state=42))
])
knn_scaled.fit(X_train, y_train)
logreg_scaled.fit(X_train, y_train)
tree_scaled.fit(X_train, y_train)
knn_pred_scaled = knn_scaled.predict(X_test)
logreg_pred_scaled = logreg_scaled.predict(X_test)
tree_pred_scaled = tree_scaled.predict(X_test)
print("\n=== RESULTS WITH SCALING ===")
print("KNN Accuracy:", accuracy_score(y_test, knn_pred_scaled))
print("Logistic Regression Accuracy:", accuracy_score(y_test, logreg_pred_scaled))
print("Decision Tree Accuracy:", accuracy_score(y_test, tree_pred_scaled))
#####################################
# BLOCK 10: CLASSIFICATION REPORT
#####################################
print("\n=== CLASSIFICATION REPORTS (WITH SCALING) ===")
print("\nKNN Report:")
print(classification_report(y_test, knn_pred_scaled))
print("\nLogistic Regression Report:")
print(classification_report(y_test, logreg_pred_scaled))
print("\nDecision Tree Report:")
print(classification_report(y_test, tree_pred_scaled))
####################################
# BLOCK 11: CONFUSION MATRIX
####################################
print("\n=== CONFUSION MATRICES (WITH SCALING) ===")
print("\nKNN Confusion Matrix:")
print(confusion_matrix(y_test, knn_pred_scaled))
print("\nLogistic Regression Confusion Matrix:")
print(confusion_matrix(y_test, logreg_pred_scaled))
print("\nDecision Tree Confusion Matrix:")
print(confusion_matrix(y_test, tree_pred_scaled))
###############################################
# BLOCK 12: MODWLS WITH CROSS-VALIDATION
###############################################
from sklearn.model_selection import cross_val_score
from sklearn.preprocessing import StandardScaler
knn_pipeline = Pipeline([
('scaler', StandardScaler()),
('knn', KNeighborsClassifier(n_neighbors=5))
])
logreg_pipeline = Pipeline([
('scaler', StandardScaler()),
('logreg', LogisticRegression(max_iter=5000))
])
tree_model = DecisionTreeClassifier(random_state=42)
knn_cv_scores = cross_val_score(knn_pipeline, X, y, cv=5, scoring='accuracy')
logreg_cv_scores = cross_val_score(logreg_pipeline, X, y, cv=5, scoring='accuracy')
tree_cv_scores = cross_val_score(tree_model, X, y, cv=5, scoring='accuracy')
print("=== Cross-Validation Results ===")
print("\nKNN")
print("CV Scores:", knn_cv_scores)
print("Mean Accuracy:", np.mean(knn_cv_scores))
print("Standard Deviation:", np.std(knn_cv_scores))
print("\nLogistic Regression")
print("CV Scores:", logreg_cv_scores)
print("Mean Accuracy:", np.mean(logreg_cv_scores))
print("Standard Deviation:", np.std(logreg_cv_scores))
print("\nDecision Tree")
print("CV Scores:", tree_cv_scores)
print("Mean Accuracy:", np.mean(tree_cv_scores))
print("Standard Deviation:", np.std(tree_cv_scores))
####################################
# BLOCK 13: CONFUSION MATRIX
####################################
from sklearn.model_selection import cross_val_predict
from sklearn.metrics import confusion_matrix, classification_report
knn_pred = cross_val_predict(knn_pipeline, X, y, cv=5)
logreg_pred = cross_val_predict(logreg_pipeline, X, y, cv=5)
tree_pred = cross_val_predict(tree_model, X, y, cv=5)
print("\n=== Confusion Matrices ===")
print("\nKNN")
print(confusion_matrix(y, knn_pred))
print("\nLogistic Regression")
print(confusion_matrix(y, logreg_pred))
print("\nDecision Tree")
print(confusion_matrix(y, tree_pred))
print("\n=== Classification Reports ===")
print("\nKNN")
print(classification_report(y, knn_pred))
print("\nLogistic Regression")
print(classification_report(y, logreg_pred))
print("\nDecision Tree")
print(classification_report(y, tree_pred))
###########################################
# BLOCK 14: LETS GET THE BEST VALUE FOR K
# task: use the k value to recompute the previous KNN model and evaluate the performance
##########################################
from sklearn.model_selection import cross_val_score
import numpy as np
k_values = range(1, 21)
scores = []
for k in k_values:
knn = KNeighborsClassifier(n_neighbors=k)
cv_scores = cross_val_score(knn, X, y, cv=5)
scores.append(np.mean(cv_scores))
best_k = k_values[np.argmax(scores)]
print("Best k:", best_k)
print("Best CV Accuracy:", max(scores))