All examples
Machine learning

Clustering Iris

Fit K means and evaluate clusters without assuming that cluster IDs equal class labels.

Download .aner source
kmeans_iris.aner
// Iris: Fisher, R. (1936). UCI. https://doi.org/10.24432/C56C76
// CC BY 4.0: https://creativecommons.org/licenses/by/4.0/
// Original UCI records retained; class names mapped to zero-based IDs.
// Full attribution and variant notes: datasets/iris/README.md.
import aner.dataset;
import aner.tensor;
import aner.ml;
import aner.metrics;

fn main() -> Unit {
    let flowers = dataset.iris();
    let split_seed = 2026;
    let partitions = dataset.split(flowers, 0.2, split_seed);
    let training = dataset.train(partitions);
    let testing = dataset.test(partitions);
    // Fit preprocessing only on training rows; reuse it unchanged for test rows.
    let scaler = dataset.fit_standardizer(dataset.features(training));
    let x_train = dataset.transform(scaler, dataset.features(training));
    let x_test = dataset.transform(scaler, dataset.features(testing));
    let test_labels = dataset.targets(testing);

    print("Dataset citation"); print(dataset.citation(flowers));
    print("Dataset license"); print(dataset.license(flowers));
    print("Dataset version"); print(dataset.version(flowers));
    print("Dataset SHA256"); print(dataset.sha256(flowers));
    print("Split seed"); print(split_seed);
    print("Training rows"); print(dataset.rows(training));
    print("Test rows"); print(dataset.rows(testing));

    // Fit on features only. Class labels are used below only for evaluation.
    // One fixed initialization is educational; it does not find a guaranteed optimum.
    let clusters = 3;
    let initialization_seed = 42;
    let model: KMeansModel = ml.fit_kmeans(x_train, clusters, 100, 0.000001, initialization_seed);
    print("Clusters"); print(clusters);
    print("Initialization seed"); print(initialization_seed);
    print("Converged"); print(ml.converged(model));
    print("Iterations"); print(ml.iterations(model));
    print("Training inertia (sum of squared distances)"); print(ml.inertia(model));
    let assignments = ml.predict(model, x_test);
    // Cluster numbers have no class meaning: ARI compares partitions up to renaming.
    print("Test adjusted Rand index");
    print(metrics.adjusted_rand_index(assignments, test_labels));
    print("Centers in standardized feature coordinates; row-major values");
    let centers = ml.centers(model);
    var row = 0;
    while row < tensor.rows(centers) {
        var col = 0;
        while col < tensor.cols(centers) {
            print(tensor.value(centers, row, col));
            col = col + 1;
        }
        row = row + 1;
    }
}

Make it your experiment.

With Aner installed, save this program as examples/kmeans_iris.aner inside a folder for your experiment. Open a terminal in that folder, then check and run the program.

Terminal
aner check examples/kmeans_iris.aner
aner run examples/kmeans_iris.aner

The aner command must be on your PATH. Follow the installation guide if your terminal cannot find it.

Iris and Wine are teaching datasets with their own attribution. Example outcomes are not comparative benchmarks or evidence of clinical validity.

Dataset sources & attribution
Install Aner