diff --git a/HW 5/Final_HW5.py b/HW 5/Final_HW5.py new file mode 100644 index 0000000..7668371 --- /dev/null +++ b/HW 5/Final_HW5.py @@ -0,0 +1,764 @@ +# %% +############################ +# BLOCK 1: IMPORTS +############################ + +# libraries! +import numpy as np # numpy is Python's "array" library +import pandas as pd # Pandas is Python's "data" library ("dataframe" == spreadsheet) + +# %% + + +# %% +############################# +# BLOCK 2: VARIABLES LISTING +############################# + +# for reference, as you work throughout, come back and list all the variable names +# here along with what that variable holds + +# Variable: Contents +# ------------------- +# [FILL IN VARS BELOW] +# + +# %% +#################################### +#from sklearn.datasets import load_iris +#data = load_iris() +#df = pd.DataFrame(data.data, columns=data.feature_names) +#df + +# %% +############################# +# BLOCK 3: READING DATA +############################# + +# let's read in our flower data... +# +filename = 'iris.csv' +df = pd.read_csv(filename, header=0) # encoding="latin1" et al. +print(f"{filename}: file read into a pandas DataFrame.") +df.head() + +# %% +############################# +# BLOCK 4: SETTING OPTIONS +############################# + +# +# a dataframe is a "spreadsheet in Python" +# (this one seems to have an extra column!) +# +pd.set_option('display.max_rows', 8) # None for no limit; default: 10 +pd.set_option('display.min_rows', 8) # None for no limit; default: 10 +# let's view it! +df + +# %% +############################## +# BLOCK 5: USING DF'S .info() +############################## + +# +# let's look at our pandas DataFrame's info (Aargh: that extra column!) +# +df.info() + +# %% +############################# +# BLOCK 6: CLEANING DATA +############################# + +# +# let's drop that last column (dropping is usually by _name_): +# +# if you want a list of the column names use df.columns +#col5name = df.columns[5] # get column name at index 5 + +df_clean = df.drop(columns=['junk']) # drop by name is typical, but what else is possible? +df_clean.info() # Is the bad last column gone? + +# %% +############################## +# BLOCK 7: FEATURE NAMES DICT +############################## + +# +# let's keep our column names in variables, for reference; +# in machine learning contexts, these are referred to as "features" +# +features = df_clean.columns # "list" of columns +print(f"FEATURES: {features}\n") + # It's a "pandas" list, called an Index + # use it just as a Python list of strings: +print(f"First feature: {features[0]}\n") + +# let's create a dictionary to look up any column index by name +feature_name_to_index = {} +for i, name in enumerate(features): + feature_name_to_index[name] = i # using the name (as key), assign the value (i) +print(f"feature_name_to_index: {feature_name_to_index}") + +# %% +############################## +# BLOCK 8: INSPECTING DATA +############################## + +# +# let's look at our cleaned-up dataframe... +# +df_clean.info() +print('=' * 70) + +# +# Notice that the non-null count is _different_ for irisname! +# Why? Show a table and inspect... +print(df_clean) +print('=' * 70) + +# or more to the point, grab that column and inspect: +print(f"Irisname entries: {df_clean['irisname'].unique()}") + +# also note how the last two rows in the df have problems, among others... + + +# %% +df.columns + +# %% +############################## +# BLOCK 9: USING DF'S dropna +############################## + +# +# typically, after dropping columns that we don't want, +# we drop rows with missing data (other approaches are possible, too) +# +df_clean = df_clean.dropna() # this removes all rows with nan items +df_clean.info() +print('=' * 70) +df_clean + +# +# notice that _all_ of the rows now have 144 non-null items +# also, the first and last rows (among others) aren't valid data... +# we'll handle that next + +# %% +################################ +# BLOCK 10: REMOVING BOGUS DATA +################################ + +# or more to the point, grab that column and inspect: +print(f"Irisname entries: {df_clean['irisname'].unique()}") +print(df_clean['irisname'] == 'alieniris') # what does this show? +print(df_clean['irisname'] != 'alieniris') # what does this show? + +# define a final version of the DataFrame by pulling out the +# bad alieniris data (remember that you can pass a boolean +# Series to select data that you want) +df_final = df_clean[df_clean['irisname'] != 'alieniris'].copy() +# ^^^^ YOU NEED TO WRITE CODE HERE... + +print(df_final.shape) +df_final + +# %% +########################################## +# BLOCK 11: CONVERT SPECIES NAME TO INDEX +########################################## + +# all of scikit-learn's ML routines need numbers, not strings +# ... even for categories/classifications (like species!) +# so, we will convert the flower-species to numbers: + +species_names = df_final['irisname'].unique() +species_name_to_index = { species_names[i]:i for i in range(len(species_names))} +print(species_name_to_index) +print(type(species_name_to_index)) + +def convertSpecies(species_name: str) -> int: + ''' return the species index (a unique integer/category) ''' + #print(f"converting {species_name}...") + return species_name_to_index[species_name] + +# Let's try it out... +for name in species_names: + print(f"{name} maps to {convertSpecies(name)}") + +# %% +########################################## +# BLOCK 12: USING DF'S .apply +########################################## + +# +# we can "apply" our new convertSpecies function to a whole column +# +# (The following will issue a "SettingWithCopyWarning" here...) +# see https://pandas.pydata.org/pandas-docs/stable/user_guide/indexing.html#returning-a-view-versus-a-copy +# +# >>> ADD CODE HERE TO KEEP THE WARNING FROM HAPPENING <<< + +df_final.loc[:, 'irisname'] = df_final['irisname'].apply(convertSpecies) +#df_final.loc[:, 'irisname'] = df_final['irisname'].apply(convertSpecies) +# Don't run this twice! Why?! What's "KeyError: 0"? +# (of course, you can always go back and re-establish definitions of df_final) + + + +# %% +########################################## +# BLOCK 13: CONFIRMING FINAL DATAFRAME +########################################## + +# +# let's see it! (this is safe to run many times...) +# +df_final # print(df_final.tostring()) # for _all_ rows... + +# %% +########################################## +# BLOCK 14: CONVERTING TO NUMPY FORMAT +########################################## + +# +# let's convert our dataframe to a numpy array, named A +# Our ML library, scikit-learn operates entirely on numpy arrays. +# +#A = df_final.values +A = df_final.to_numpy() # better -- self-documenting! + +print(f"type of A: {type(A)}") +print(df_final.head()) +print(f"A[0] = {A[0]}") + +# %% +################################################## +# BLOCK 15: CONVERTING NUMPY ARRAY TO ALL FLOATS +################################################## + +# +# let's convert to make sure it's all floating-point, so we can multiply and divide +# +A = A.astype('float64') # so many: www.tutorialspoint.com/numpy/numpy_data_types.htm +A + +# %% +########################################## +# BLOCK 16: USING NUMPY'S .shape +########################################## + +# +# nice to have num_rows and num_cols variables handy... +# +num_rows, num_cols = A.shape +print(f"\nThe dataset has {num_rows} rows and {num_cols} cols") +print(A) + +# %% +########################################## +# BLOCK 17: PRINTING FLOWER INFO +########################################## + +# let's use all of our previously-defined variables, to reinforce names... + +# choose a row index (particular flower) arbitrarily: +flower = 132 +print(f"flower #{flower} data is {A[flower]}") + +for i in range(len(features)): + col_name = features[i] + if col_name != 'irisname': + print(f" Its {col_name} is {A[flower][i]}") + else: + species_num = int(A[flower][i]) + species_name = species_names[species_num] + print(f" Its {col_name} is {species_name} ({species_num})") + +# %% +########################################## +# BLOCK 18: WRITING OUR OWN 1-NN FUNCTION +########################################## + +# +# We don't have to use scikit-learn to implement NN! +# + +# +# data-driven predictive model (1-nearest-neighbor) +# + +# Python functions are first-class objects! +dist = np.linalg.norm # built in to numpy... but what does norm do? + +num_rows, num_cols = A.shape # data size + +def predictiveModel( features: list[float] ) -> str: + """ input: a list of four features + [ sepallen, sepalwid, petallen, petalwid ] + output: the predicted species of iris, from + setosa (0), versicolor (1), virginica (2) + """ + our_features = np.asarray(features) # make a numpy array + + closest_flower = A[0] + closest_features = A[0,0:4] + closest_distance = dist(our_features - closest_features) + + for i in range(1, num_rows, 1): + current_flower = A[i] + current_features = A[i,0:4] + current_distance = dist(our_features - current_features) + + if current_distance < closest_distance: + closest_distance = current_distance # remember closest! + closest_flower = current_flower + + # done comparing with every flower in the dataset + predicted_species = int(round(closest_flower[4])) # what type is closest_flower? + name = species_names[predicted_species] + return f"{name} ({predicted_species})" + +# +# Try it! +# +# features = eval(input("Enter new features: ")) +# +features = [ 4.6, 3.6, 3.0, 1.2 ] +result = predictiveModel( features ) +print(f"I predict {result} from features {features}") + +# %% +########################################## +# BLOCK 19: COMMENTS ON SCIKIT-LEARN kNN +########################################## + +# +# but, we don't have to write our own ... because +# +# we want knn for any k (not just k=1 as above) +# we want an already-debugged algorithm! +# we want to ask iris-related questions instead of implementation ones... +# + +# %% +################################################ +# BLOCK 20: DEFINIING FEATURES & LABELS FOR kNN +################################################ + +print("+++ Start of data definitions +++\n") + +X_all = A[:,0:4] # X (features) ... is all rows, columns 0, 1, 2, 3 +y_all = A[:,4] # y (labels) ... is all rows, column 4 only + # (look back at slide 20) +print(f"X_all (just features) is \n {X_all}") +print(f"y_all (just labels) is \n {y_all}") + +# %% +################################################ +# BLOCK 21: REWEIGHTING FEATURES +################################################ + +# +# we can re-weight different features here... +# + +col_weights = { # could be called feature weight... + 'sepallen':1.0, + 'sepalwid':1.0, + 'petallen':1.0, + 'petalwid':1.0, +} + +for col_name in col_weights: + i = feature_name_to_index[col_name] # get the column index, i, of the column name + weight = col_weights[col_name] # from the dictionary above + print(f"Weighting {col_name} by {weight}") + # weighting == "multiplying" + X_all[:,i] *= weight # multiply by the weight to give this column ("feature") + +# %% +################################################ +# BLOCK 22: PERMUTING THE DATA +################################################ + +# +# +# we scramble the data, to give a different TRAIN/TEST split each time... +# +indices = np.random.permutation(len(y_all)) # indices is a permutation-list + +# we scramble both X and y, necessarily with the same permutation +X_labeled = X_all[indices] # we apply the _same_ permutation to each! +y_labeled = y_all[indices] # again... +print(X_labeled) # note that X_labeled and y_labeled are permuted identically +print(y_labeled) + +# %% +################################################ +# BLOCK 23: DEFINING TRAIN VS TEST SETS +################################################ + +# +# We next separate into test data and training data ... +# + We will train on the training data... +# + We will _not_ look at the testing data when building the model +# +# Then, afterward, we will test on the testing data -- and see how well we do! +# + +# +# a common convention: train on 80%, test on 20% Let's define the TEST_PERCENT +# +num_rows = X_labeled.shape[0] # the number of labeled rows +test_percent = 0.20 +test_size = int(test_percent * num_rows) # no harm in rounding down + +X_test = X_labeled[:test_size] # first section are for testing +y_test = y_labeled[:test_size] + +X_train = X_labeled[test_size:] # all the rest are for training +y_train = y_labeled[test_size:] + +num_train_rows = len(y_train) +num_test_rows = len(y_test) +print(f"total rows: {num_rows}; training with {num_train_rows} rows; testing with {num_test_rows} rows" ) +print(f"\t(sanity check: {num_train_rows} + {num_test_rows} = {num_train_rows + num_test_rows})") + +# %% +####################################################### +# BLOCK 24: FIRST ATTEMPT TO BUILD & TRAIN A kNN MODEL +####################################################### + +# +# +++ This is the "Model-building and Model-training Cell" +# +# Create a kNN model and train it! +# +from sklearn.neighbors import KNeighborsClassifier + +k = 84 # we don't know what k to use, so we guess for now! (this will _not_ be a good value) +knn_model = KNeighborsClassifier(n_neighbors = k) # here, k is the "k" in kNN + +# we train the model (it's one line!) +knn_model.fit(X_train, y_train) # yay! trained! +print("Created and trained a knn classifier with k =", k) + +# %% +################################################ +# BLOCK 25: TEST THE kNN MODEL +################################################ + +# +# +++ This is the "Model-testing Cell" +# +# Now, let's see how well we did on our "held-out data" (the testing data) +# + +# We run our test set! +predicted_labels = knn_model.predict(X_test) +actual_labels = y_test + +# Let's print them so we can compare... +print("Predicted labels:", predicted_labels) +print("Actual labels :", actual_labels) + +# And, some overall results +num_correct = sum(predicted_labels == actual_labels) +total = len(actual_labels) +print(f"\nResults on test set: {num_correct} correct out of {total} total.") + +# %% +################################################ +# BLOCK 26: PRETTY-PRINT PREDICTED LABELS +################################################ + +# +# Let's print these more helpfully, in a vertical table +# + +def compareLabels(predicted_labels: np.ndarray, actual_labels: np.ndarray) -> int: + ''' a more neatly formatted comparison, returning the number correct ''' + num_labels = len(predicted_labels) + num_correct = 0 + + for i in range(num_labels): + predicted = int(round(predicted_labels[i])) # round-to-int protects from float imprecision + actual = int(round(actual_labels[i])) + result = "incorrect" + if predicted == actual: # if they match, + result = "" # no longer incorrect + num_correct += 1 # and we count a match! + + # note the justification formatting: + # :>3d right justifies integers (d) to width 3 + # :<12s left justifies strings (s) to width 12 + print(f"row {i:>3d} : ", end = "") + print(f"{species_names[predicted]:>12s} ", end = "") + print(f"{species_names[actual]:<12s} {result}") + + print() + print(f"Correct: {num_correct} out of {num_labels}") + return num_correct + +# +# let's try it out! +# + +compareLabels(predicted_labels,actual_labels) + +# %% +################################################ +# BLOCK 27: USE THE FIRST-ATTEMPT kNN MODEL +################################################ + +# +# Ok! We have our knn model, we could just use it... +# + +# +# data-driven predictive model (k-nearest-neighbor), using scikit-learn +# + +def predictiveModel( features: list[float] ) -> str: + ''' input: a list of four features + [ sepallen, sepalwid, petallen, petalwid ] + output: the predicted species of iris, from + setosa (0), versicolor (1), virginica (2) + ''' + our_features = np.asarray([features]) # extra brackets needed + predicted_species = knn_model.predict(our_features) + print(f"predicted_species = {predicted_species}") + + predicted_species = int(round(predicted_species[0])) # unpack one element + name = species_names[predicted_species] + return f"{name} ({predicted_species})" + +# +# Try it! +# +# features = eval(input("Enter new features: ")) +# +features = [6.7,3.3,5.7,2.1] # [5.8,2.7,4.1,1.0] [4.6,3.6,3.0,2.2] [6.7,3.3,5.7,2.1] +result = predictiveModel( features ) +print(f"I predict {result} from features {features}") + +# %% +################################################ +# BLOCK 28: COMMENTS ON CHOICE OF BEST k +################################################ + +# +# Except, we didn't really explore whether this was the BEST model we could build! +# +# +# We used k = 84 (a neighborhood size of 84 flowers) +# In a dataset of only 140ish flowers, with three species, this seems like a bad idea! +# +# Perhaps we should try ALL the neighborhood sizes in their own TRAIN/TEST split +# and see which neighborhood size works the best, for irises, at least... +# + +# %% +################################################ +# BLOCK 29: USING CROSS VALIDATION +################################################ + +# +# to do this, we use "cross validation" +# (see slide 45) +# + +import seaborn as sns +import matplotlib.pyplot as plt +from sklearn.model_selection import cross_val_score + +# +# cross-validation splits the training set into two pieces: +# + model-building and model-validation. We'll use "build" and "validate" +# + +max_k = 85 +all_accuracies = [] +best_k = 1 +best_accuracy = -1 + +for k in range(1, max_k, 1): + knn_cv_model = KNeighborsClassifier(n_neighbors=k) + cv_scores = cross_val_score(knn_cv_model, X_train, y_train, cv=5) + this_cv_accuracy = cv_scores.mean() + print(f"k: {k:2d} cv accuracy: {this_cv_accuracy:7.4f}") + all_accuracies.append(this_cv_accuracy) + if this_cv_accuracy > best_accuracy: + best_accuracy = this_cv_accuracy + best_k = k +# assign best value of k to best_k +#best_k = k # *** AT THE MOMENT THIS IS INCORRECT *** +# you'll need to modify the loop above to find and remember the real best_k + +print(f"best_k = {best_k} yields the highest average cv accuracy.") # print the best one + +plt.figure(figsize=(10, 6)) +sns.lineplot(x=range(1,len(all_accuracies)+1), y=all_accuracies) +plt.xlabel("k value (index)") +plt.ylabel("Cross-validated accuracy") +plt.title("kNN Accuracy vs k Value") +plt.show() + +# %% +################################################ +# BLOCK 30: RE-BUILD & RE-TRAIN USING BEST k +################################################ + +# +# Now, we re-create and re-run the "Model-building and -training Cell" +# +# Now, using best_k instead of the original, randomly-guessed value How does it do?! +# +from sklearn.neighbors import KNeighborsClassifier +knn_model_tuned = KNeighborsClassifier(n_neighbors = best_k) # here, we use the best_k + +# we train the model (one line!) +knn_model_tuned.fit(X_train, y_train) # yay! trained! +print(f"Created + trained a knn classifier, now tuned with a (best) k of {best_k}") + +# %% +################################################ +# BLOCK 31: RE-TEST THE BEST-k MODEL +################################################ + +# +# Re-create and re-run the "Model-testing Cell" How does it do with best_k?! +# +predicted_labels = knn_model_tuned.predict(X_test) +actual_labels = y_test + +# Let's print them so we can compare... +print("Predicted labels:", predicted_labels) +print("Actual labels:", actual_labels) +print() +# and, we'll print our nicer table... +compareLabels(predicted_labels,actual_labels) + +# %% +################################################ +# BLOCK 32: TRYING DIFFERENT PERMUTATIONS +################################################ + +# +# Before moving on, go back and choose a new permutation to give +# new testing and training sets, and go back through the steps +# for identifying the "best" k (starting at Block 22 and re-doing +# up through Block 31). +# +# Repeat this several times. +# Do you get the same value for k each time? Why or why not? +# + +# %% +#################################################### +# BLOCK 33: REBUILD & TRAIN USING BEST k & ALL DATA +#################################################### + +# Ok! Now we have tuned knn to use the "best" value of k... +# +# And, we should really use ALL available data to train our final predictive model +# (not split into test and train as before) +# + +knn_model_final = KNeighborsClassifier(n_neighbors=best_k) # here, we use the best_k +knn_model_final.fit(X_all, y_all) # yay! trained! +print(f"Created + trained a 'final' knn classifier, with a (best) k of {best_k}") + +# %% +#################################################### +# BLOCK 34: TRYING FINAL MODEL "IN THE WILD" +#################################################### + +# +# final predictive model (k-nearest-neighbor), with tuned k + ALL data incorporated +# + +def predictiveModel( features: list[float] ) -> str: + ''' input: a list of four features + [ sepallen, sepalwid, petallen, petalwid ] + output: the predicted species of iris, from + setosa (0), versicolor (1), virginica (2) + ''' + our_features = np.asarray([features]) # extra brackets needed + predicted_species = knn_model_final.predict(our_features) + + predicted_species = int(round(predicted_species[0])) # unpack one element + name = species_names[predicted_species] + return f"{name} ({predicted_species})" + +# +# Try it on several! +# +# features = eval(input("Enter new features: ")) +# +features = [6.7,3.3,5.7,2.1] # [5.8,2.7,4.1,1.0] [4.6,3.6,3.0,2.2] [6.7,3.3,5.7,2.1] +result = predictiveModel( features ) +print(f"I predict {result} from features {features}") + +# %% +################################################################ +# BLOCK 35a: HOW DOES THE MODEL PERFORM ON MEAN OF EACH FLOWER? +################################################################ + +# let's recall the species names and their order in the list +print(species_names) +print('=' * 70) + +# and let's recall what the final DataFrame looks like, noting that +# the irisname was converted to the corresponding index in species_names +print(df_final) + +# %% +################################################################ +# BLOCK 35b: HOW DOES THE MODEL PERFORM ON MEAN OF EACH FLOWER? +################################################################ + +# let's grab the data from df_final corresponding to versicolor +versicolor_data = df_final[ df_final['irisname'] == convertSpecies('versicolor') ] +print(versicolor_data) +print('=' * 70) + +# let's print the mean of each column +print(versicolor_data.mean()) +print('=' * 70) + +# and let's grab the features means only (don't need the label) for versicolor +versicolor_features_means = np.asarray(versicolor_data.mean()[:-1]) +print(versicolor_features_means) + +# now run the predictive model (see Block 34 above) and print a corresponding +# message + +# >>> YOU ADD CODE HERE +result = predictiveModel(versicolor_features_means) +print(f"Mean versicolor features {versicolor_features_means} -> {result}") + +# %% +################################################################ +# BLOCK 35c: HOW DOES THE MODEL PERFORM ON MEAN OF EACH FLOWER? +################################################################ + +# Automate the process used in Block 35b, across all species: +# +# Loop across species names, and for each species, +# (a) pull that corresponding data from df_final +# (b) compute the means of the features, storing as a numpy array +# (c) call the predictive model using those features +# (d) print a corresponding message + +# >> YOU ADD CODE HERE -- YOU SHOULD NEED NO MORE THAN 5 LINES OF CODE +for name in species_names: + species_data = df_final[df_final['irisname'] == convertSpecies(name)] + species_features_means = np.asarray(species_data.mean()[:-1]) + result = predictiveModel(species_features_means) + print(f"Mean {name} features {species_features_means} -> {result}") + + diff --git a/HW 5/Iris_knn_W2025_template.ipynb b/HW 5/Iris_knn_W2025_template.ipynb index 8f8ef94..44236f5 100644 --- a/HW 5/Iris_knn_W2025_template.ipynb +++ b/HW 5/Iris_knn_W2025_template.ipynb @@ -752,7 +752,7 @@ }, { "cell_type": "code", - "execution_count": 12, + "execution_count": null, "metadata": { "id": "h0spkxGzKPR-" },