All examples
Machine learning

KNN on Iris

Standardize training data, fit a nearest neighbor classifier, and inspect held out predictions.

Download .aner source
ml_iris.aner
// Iris: Fisher, R. (1936). UCI. https://doi.org/10.24432/C56C76
// CC BY 4.0: https://creativecommons.org/licenses/by/4.0/
// Original UCI records retained; class names mapped to zero-based IDs.
// Full attribution and variant notes: datasets/iris/README.md.
import aner.dataset;
import aner.tensor;
import aner.ml;
import aner.metrics;

fn main() -> Unit {
    let flowers = dataset.iris();
    let split_seed = 2026;
    let partitions = dataset.split(flowers, 0.2, split_seed);
    let training = dataset.train(partitions);
    let testing = dataset.test(partitions);
    // Fit preprocessing only on training rows; reuse it unchanged for test rows.
    let scaler = dataset.fit_standardizer(dataset.features(training));
    let x_train = dataset.transform(scaler, dataset.features(training));
    let x_test = dataset.transform(scaler, dataset.features(testing));
    let test_labels = dataset.targets(testing);

    print("Dataset citation"); print(dataset.citation(flowers));
    print("Dataset license"); print(dataset.license(flowers));
    print("Dataset version"); print(dataset.version(flowers));
    print("Dataset SHA256"); print(dataset.sha256(flowers));
    print("Split seed"); print(split_seed);
    print("Training rows"); print(dataset.rows(training));
    print("Test rows"); print(dataset.rows(testing));

    // Choose k before looking at the held-out results.
    let k = 5;
    let model: KNNClassifier = ml.fit_knn(x_train, dataset.targets(training), k);
    let predictions = ml.predict(model, x_test);
    print("Neighbors"); print(k);
    print("Test accuracy"); print(metrics.accuracy(predictions, test_labels));
    let confusion = metrics.confusion_matrix(predictions, test_labels, dataset.classes(flowers));
    print("Confusion matrix: true rows, predicted columns; row-major counts");
    var row = 0;
    while row < tensor.rows(confusion) {
        var col = 0;
        while col < tensor.cols(confusion) {
            print(tensor.value(confusion, row, col));
            col = col + 1;
        }
        row = row + 1;
    }
    let class_ids = ml.classes(model);
    let votes = ml.predict_proba(model, x_test);
    print("First test row: class ID then neighbor vote fraction (not calibrated)");
    var c = 0;
    while c < tensor.rows(class_ids) {
        print(tensor.value(class_ids, c, 0));
        print(tensor.value(votes, 0, c));
        c = c + 1;
    }
    // Indices refer to x_train, not the original dataset; use row_ids to map back.
    let neighbors = ml.neighbor_indices(model, x_test);
    let distances = ml.neighbor_distances(model, x_test);
    let source_rows = dataset.row_ids(training);
    print("First test row: nearest training-local row index and standardized Euclidean distance");
    // Aner currently has no Float64-to-Int64 cast, so print the training-local
    // index separately; source_rows is the complete training-to-source mapping.
    print(tensor.value(neighbors, 0, 0));
    print(tensor.value(distances, 0, 0));
    print("Training-to-source mapping rows"); print(tensor.rows(source_rows));
}

Make it your experiment.

With Aner installed, save this program as examples/ml_iris.aner inside a folder for your experiment. Open a terminal in that folder, then check and run the program.

Terminal
aner check examples/ml_iris.aner
aner run examples/ml_iris.aner

The aner command must be on your PATH. Follow the installation guide if your terminal cannot find it.

Iris and Wine are teaching datasets with their own attribution. Example outcomes are not comparative benchmarks or evidence of clinical validity.

Dataset sources & attribution
Install Aner