--- title: "Palmer Penguins kNN Tuning (Python)" format: html engine: knitr --- This Quarto document runs the Python translation of `kNN_example_penguins2.qmd` through `reticulate`. ```{r} library(reticulate) py_require(c("pandas", "numpy", "scikit-learn", "matplotlib", "seaborn", "palmerpenguins")) ``` ## Load, clean, and visualize the data ```{python} import matplotlib.pyplot as plt import pandas as pd import seaborn as sns from sklearn.metrics import ConfusionMatrixDisplay, accuracy_score, confusion_matrix from sklearn.model_selection import GridSearchCV, StratifiedKFold, train_test_split from sklearn.neighbors import KNeighborsClassifier from sklearn.pipeline import Pipeline from sklearn.preprocessing import StandardScaler try: from palmerpenguins import load_penguins penguins = load_penguins() except ImportError: penguins = sns.load_dataset("penguins") penguins = penguins.drop(columns=["island", "sex"]) print(penguins["species"].value_counts().sort_index()) print(penguins.isna().sum()) print(penguins.describe(include="all")) # Equivalent to step_naomit() in the R recipe. penguins = penguins.dropna().copy() sns.scatterplot(data=penguins, x="bill_length_mm", y="bill_depth_mm", hue="species") plt.tight_layout() plt.show() sns.pairplot(penguins, hue="species") plt.show() ``` ## Split the data and define the model ```{python} X = penguins.drop(columns="species") y = penguins["species"] X_train, X_test, y_train, y_test = train_test_split( X, y, test_size=0.20, stratify=y, random_state=123 ) # Scaling is inside the pipeline so each cross-validation fold learns its own # scaling parameters, preventing information leakage. pipeline = Pipeline( [ ("scale", StandardScaler()), ("knn", KNeighborsClassifier(weights="uniform")), ] ) ``` ## Tune k with five-fold cross-validation ```{python} neighbor_grid = list(range(4, 16)) folds = StratifiedKFold(n_splits=5, shuffle=True, random_state=234) search = GridSearchCV( pipeline, param_grid={"knn__n_neighbors": neighbor_grid}, cv=folds, scoring="accuracy", return_train_score=False, ) search.fit(X_train, y_train) cv_results = pd.DataFrame(search.cv_results_)[ ["param_knn__n_neighbors", "mean_test_score", "std_test_score", "rank_test_score"] ].sort_values("rank_test_score") print(cv_results.to_string(index=False)) print(f"Best k: {search.best_params_['knn__n_neighbors']}") print(f"Best CV accuracy: {search.best_score_:.4f}") ``` ## Evaluate the finalized model ```{python} # GridSearchCV refits the best pipeline on all training data. final_model = search.best_estimator_ test_predictions = final_model.predict(X_test) print(f"Test accuracy: {accuracy_score(y_test, test_predictions):.4f}") labels = final_model.classes_ cm = confusion_matrix(y_test, test_predictions, labels=labels) print(pd.DataFrame(cm, index=labels, columns=labels)) ConfusionMatrixDisplay(cm, display_labels=labels).plot(cmap="Blues") plt.title("Testing confusion matrix") plt.tight_layout() plt.show() ```