All examples
Data

From a typed CSV to a tensor

Validate synthetic records, preserve missingness, summarize bounded batches, and explicitly collect complete rows.

Download .aner source
data_csv.aner
// Synthetic records only: these values do not represent patients or clinical rules.
// Run from the project root; CSV paths are relative to the working directory.
import aner.data;
import aner.tensor;

fn main() -> Unit {
    var schema: DataSchema = data.schema();
    schema = data.column(schema, "record_id", "String", false);
    schema = data.column(schema, "age", "Int64", true);
    schema = data.column(schema, "measurement", "Float64", true);
    schema = data.column(schema, "site", "String", false);
    schema = data.column(schema, "eligible", "Bool", false);

    // This declares a plan. No file is loaded until summarize or collect.
    // Only unquoted NA is missing; the reader does not guess or silently clean.
    let source: DataScan = data.scan_csv("examples/fixtures/visits.csv", schema, "NA");
    let adults = data.where_ge(source, "age", 18);
    let selected = data.include(data.select(adults, "measurement"), "age");
    print(data.explain(selected));

    // One MiB for managed execution buffers; two selected rows per batch.
    // All source columns are validated, including rows that do not pass filters.
    let budget = 1048576;
    let report: DataReport = data.summarize(selected, budget, 2);
    print(data.describe(report));
    print("Validated source rows"); print(data.scanned_rows(report));
    print("Selected rows"); print(data.rows(report));
    print("Missing measurements"); print(data.missing(report, "measurement"));
    print("Mean measurement among observed selected rows");
    print(data.mean(report, "measurement"));

    // Exclusion is explicit: missing measurements are not filled with zero.
    // Collection executes a new scan of the same path; it is not a snapshot.
    let complete = data.where_present(selected, "measurement");
    let table: DataTable = data.collect(complete, 100, budget, 2);
    print("Collected complete rows"); print(data.rows(table));
    print("First column"); print(data.column_name(table, 0));
    let values = data.to_tensor(table, budget);
    print("Tensor rows"); print(tensor.rows(values));
    print("Tensor columns"); print(tensor.cols(values));
    print("First measurement"); print(tensor.value(values, 0, 0));

    // Identifiers retain their original text, including leading zeros.
    let identifiers = data.collect(data.select(source, "record_id"), 100, budget, 2);
    print("First record ID"); print(data.text(identifiers, 0, "record_id"));
}

Make it your experiment.

With Aner installed, save this program as examples/data_csv.aner inside a folder for your experiment. Open a terminal in that folder, then check and run the program.

Terminal
aner check examples/data_csv.aner
aner run examples/data_csv.aner

The aner command must be on your PATH. Follow the installation guide if your terminal cannot find it.

Download visits.csv to examples/fixtures/visits.csv beneath your working directory, then run Aner from that directory. These are six synthetic records, not patient data.

Iris and Wine are teaching datasets with their own attribution. Example outcomes are not comparative benchmarks or evidence of clinical validity.

Dataset sources & attribution
Install Aner