MeiliSearch/milli/benches/utils.rs

use std::{fs::{File, create_dir_all, remove_dir_all}, time::Duration};

use heed::EnvOpenOptions;
use criterion::BenchmarkId;
use milli::{FacetCondition, Index, update::{IndexDocumentsMethod, Settings, UpdateBuilder, UpdateFormat}};

pub struct Conf<'a> {
    /// where we are going to create our database.mmdb directory
    /// each benchmark will first try to delete it and then recreate it
    pub database_name: &'a str,
    /// the dataset to be used, it must be an uncompressed csv
    pub dataset: &'a str,
    pub group_name: &'a str,
    pub queries: &'a[&'a str],
    /// here you can change which criterion are used and in which order.
    /// - if you specify something all the base configuration will be thrown out
    /// - if you don't specify anything (None) the default configuration will be kept
    pub criterion: Option<&'a [&'a str]>,
    /// the last chance to configure your database as you want
    pub configure: fn(&mut Settings),
    pub facet_condition: Option<FacetCondition>,
    /// enable or disable the optional words on the query
    pub optional_words: bool,
}

impl Conf<'_> {
    fn nop(_builder: &mut Settings) {}

    fn songs_conf(builder: &mut Settings) {
        let displayed_fields = [
            "id", "title", "album", "artist", "genre", "country", "released", "duration",
        ]
        .iter()
        .map(|s| s.to_string())
        .collect();
        builder.set_displayed_fields(displayed_fields);

        let searchable_fields = ["title", "album", "artist"]
            .iter()
            .map(|s| s.to_string())
            .collect();
        builder.set_searchable_fields(searchable_fields);
    }

    pub const BASE: Self = Conf {
        database_name: "benches.mmdb",
        dataset: "",
        group_name: "",
        queries: &[],
        criterion: None,
        configure: Self::nop,
        facet_condition: None,
        optional_words: true,
    };

    pub const BASE_SONGS: Self = Conf {
        dataset: "smol-songs",
        configure: Self::songs_conf,
        ..Self::BASE
    };
}

pub fn base_setup(conf: &Conf) -> Index {
    match remove_dir_all(&conf.database_name) {
        Ok(_) => (),
        Err(e) if e.kind() == std::io::ErrorKind::NotFound => (),
        Err(e) => panic!("{}", e),
    }
    create_dir_all(&conf.database_name).unwrap();

    let mut options = EnvOpenOptions::new();
    options.map_size(100 * 1024 * 1024 * 1024); // 100 GB
    options.max_readers(10);
    let index = Index::new(options, conf.database_name).unwrap();

    let update_builder = UpdateBuilder::new(0);
    let mut wtxn = index.write_txn().unwrap();
    let mut builder = update_builder.settings(&mut wtxn, &index);

    if let Some(criterion) = conf.criterion {
        builder.reset_faceted_fields();
        builder.reset_criteria();
        builder.reset_stop_words();

        let criterion = criterion.iter().map(|s| s.to_string()).collect();
        builder.set_criteria(criterion);
    }

    (conf.configure)(&mut builder);

    builder.execute(|_, _| ()).unwrap();
    wtxn.commit().unwrap();

    let update_builder = UpdateBuilder::new(0);
    let mut wtxn = index.write_txn().unwrap();
    let mut builder = update_builder.index_documents(&mut wtxn, &index);
    builder.update_format(UpdateFormat::Csv);
    builder.index_documents_method(IndexDocumentsMethod::ReplaceDocuments);
    // we called from cargo the current directory is supposed to be milli/milli
    let reader = File::open(conf.dataset).unwrap();
    builder.execute(reader, |_, _| ()).unwrap();
    wtxn.commit().unwrap();

    index
}

pub fn run_benches(c: &mut criterion::Criterion, confs: &[Conf]) {
    for conf in confs {
        let index = base_setup(conf);

        let mut group = c.benchmark_group(&format!("{}: {}", conf.dataset, conf.group_name));
        group.measurement_time(Duration::from_secs(10));

        for &query in conf.queries {
            group.bench_with_input(BenchmarkId::from_parameter(query), &query, |b, &query| {
                b.iter(|| {
                    let rtxn = index.read_txn().unwrap();
                    let mut search = index.search(&rtxn);
                    search.query(query).optional_words(conf.optional_words);
                    if let Some(facet_condition) = conf.facet_condition.clone() {
                        search.facet_condition(facet_condition);
                    }
                    let _ids = search.execute().unwrap();
                });
            });
        }
        group.finish();
    }
}
add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`use std::{fs::{File, create_dir_all, remove_dir_all}, time::Duration};`
push a first version of the benchmark for the typo 2021-04-01 18:54:14 +02:00
			`use heed::EnvOpenOptions;`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00			`use criterion::BenchmarkId;`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`use milli::{FacetCondition, Index, update::{IndexDocumentsMethod, Settings, UpdateBuilder, UpdateFormat}};`
push a first version of the benchmark for the typo 2021-04-01 18:54:14 +02:00
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00			`pub struct Conf<'a> {`
add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`/// where we are going to create our database.mmdb directory`
			`/// each benchmark will first try to delete it and then recreate it`
			`pub database_name: &'a str,`
			`/// the dataset to be used, it must be an uncompressed csv`
			`pub dataset: &'a str,`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00			`pub group_name: &'a str,`
			`pub queries: &'a[&'a str],`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`/// here you can change which criterion are used and in which order.`
			`/// - if you specify something all the base configuration will be thrown out`
			`/// - if you don't specify anything (None) the default configuration will be kept`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00			`pub criterion: Option<&'a [&'a str]>,`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`/// the last chance to configure your database as you want`
			`pub configure: fn(&mut Settings),`
add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`pub facet_condition: Option<FacetCondition>,`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`/// enable or disable the optional words on the query`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00			`pub optional_words: bool,`
			`}`

add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`impl Conf<'_> {`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`fn nop(_builder: &mut Settings) {}`

			`fn songs_conf(builder: &mut Settings) {`
			`let displayed_fields = [`
			`"id", "title", "album", "artist", "genre", "country", "released", "duration",`
			`]`
			`.iter()`
			`.map(\|s\| s.to_string())`
			`.collect();`
			`builder.set_displayed_fields(displayed_fields);`

			`let searchable_fields = ["title", "album", "artist"]`
			`.iter()`
			`.map(\|s\| s.to_string())`
			`.collect();`
			`builder.set_searchable_fields(searchable_fields);`
			`}`

add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`pub const BASE: Self = Conf {`
			`database_name: "benches.mmdb",`
			`dataset: "",`
			`group_name: "",`
			`queries: &[],`
			`criterion: None,`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`configure: Self::nop,`
add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`facet_condition: None,`
			`optional_words: true,`
			`};`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00
			`pub const BASE_SONGS: Self = Conf {`
			`dataset: "smol-songs",`
			`configure: Self::songs_conf,`
			`..Self::BASE`
			`};`
add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`}`

add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`pub fn base_setup(conf: &Conf) -> Index {`
			`match remove_dir_all(&conf.database_name) {`
add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`Ok(_) => (),`
			`Err(e) if e.kind() == std::io::ErrorKind::NotFound => (),`
			`Err(e) => panic!("{}", e),`
			`}`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`create_dir_all(&conf.database_name).unwrap();`
push a first version of the benchmark for the typo 2021-04-01 18:54:14 +02:00
			`let mut options = EnvOpenOptions::new();`
			`options.map_size(100 * 1024 * 1024 * 1024); // 100 GB`
			`options.max_readers(10);`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`let index = Index::new(options, conf.database_name).unwrap();`
push a first version of the benchmark for the typo 2021-04-01 18:54:14 +02:00
			`let update_builder = UpdateBuilder::new(0);`
			`let mut wtxn = index.write_txn().unwrap();`
			`let mut builder = update_builder.settings(&mut wtxn, &index);`

add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`if let Some(criterion) = conf.criterion {`
push a first version of the benchmark for the typo 2021-04-01 18:54:14 +02:00			`builder.reset_faceted_fields();`
			`builder.reset_criteria();`
			`builder.reset_stop_words();`

add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`let criterion = criterion.iter().map(\|s\| s.to_string()).collect();`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00			`builder.set_criteria(criterion);`
push a first version of the benchmark for the typo 2021-04-01 18:54:14 +02:00			`}`

add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`(conf.configure)(&mut builder);`

push a first version of the benchmark for the typo 2021-04-01 18:54:14 +02:00			`builder.execute(\|_, _\| ()).unwrap();`
			`wtxn.commit().unwrap();`

			`let update_builder = UpdateBuilder::new(0);`
			`let mut wtxn = index.write_txn().unwrap();`
			`let mut builder = update_builder.index_documents(&mut wtxn, &index);`
			`builder.update_format(UpdateFormat::Csv);`
			`builder.index_documents_method(IndexDocumentsMethod::ReplaceDocuments);`
			`// we called from cargo the current directory is supposed to be milli/milli`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`let reader = File::open(conf.dataset).unwrap();`
push a first version of the benchmark for the typo 2021-04-01 18:54:14 +02:00			`builder.execute(reader, \|_, _\| ()).unwrap();`
			`wtxn.commit().unwrap();`

			`index`
			`}`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00
			`pub fn run_benches(c: &mut criterion::Criterion, confs: &[Conf]) {`
			`for conf in confs {`
add the configuration of the searchable fields and displayed fields and a default configuration for the songs 2021-04-13 11:40:16 +02:00			`let index = base_setup(conf);`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00
add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`let mut group = c.benchmark_group(&format!("{}: {}", conf.dataset, conf.group_name));`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00			`group.measurement_time(Duration::from_secs(10));`

			`for &query in conf.queries {`
			`group.bench_with_input(BenchmarkId::from_parameter(query), &query, \|b, &query\| {`
			`b.iter(\|\| {`
			`let rtxn = index.read_txn().unwrap();`
add a bunch of queries and start the introduction of the filters and the new dataset 2021-04-13 10:44:27 +02:00			`let mut search = index.search(&rtxn);`
			`search.query(query).optional_words(conf.optional_words);`
			`if let Some(facet_condition) = conf.facet_condition.clone() {`
			`search.facet_condition(facet_condition);`
			`}`
			`let _ids = search.execute().unwrap();`
merge all the criterion only benchmarks in one file 2021-04-07 11:50:38 +02:00			`});`
			`});`
			`}`
			`group.finish();`
			`}`
			`}`