mod criterion; mod node; mod query_tokens; mod search; pub mod heed_codec; pub mod tokenizer; use std::collections::HashMap; use std::hash::BuildHasherDefault; use anyhow::Context; use csv::StringRecord; use fxhash::{FxHasher32, FxHasher64}; use heed::types::*; use heed::{PolyDatabase, Database}; pub use self::search::{Search, SearchResult}; pub use self::criterion::{Criterion, default_criteria}; use self::heed_codec::{RoaringBitmapCodec, StrBEU32Codec, CsvStringRecordCodec}; pub type FastMap4 = HashMap>; pub type FastMap8 = HashMap>; pub type SmallString32 = smallstr::SmallString<[u8; 32]>; pub type SmallVec32 = smallvec::SmallVec<[T; 32]>; pub type SmallVec16 = smallvec::SmallVec<[T; 16]>; pub type BEU32 = heed::zerocopy::U32; pub type DocumentId = u32; pub type Attribute = u32; pub type Position = u32; const WORDS_FST_KEY: &str = "words-fst"; const HEADERS_KEY: &str = "headers"; const DOCUMENTS_IDS_KEY: &str = "documents-ids"; #[derive(Clone)] pub struct Index { /// Contains many different types (e.g. the documents CSV headers). pub main: PolyDatabase, /// A word and all the positions where it appears in the whole dataset. pub word_positions: Database, /// Maps a word at a position (u32) and all the documents ids where the given word appears. pub word_position_docids: Database, /// Maps a word and a range of 4 positions, i.e. 0..4, 4..8, 12..16. pub word_four_positions_docids: Database, /// Maps a word and an attribute (u32) to all the documents ids where the given word appears. pub word_attribute_docids: Database, /// Maps the document id to the document as a CSV line. pub documents: Database, ByteSlice>, } impl Index { pub fn new(env: &heed::Env) -> anyhow::Result { Ok(Index { main: env.create_poly_database(None)?, word_positions: env.create_database(Some("word-positions"))?, word_position_docids: env.create_database(Some("word-position-docids"))?, word_four_positions_docids: env.create_database(Some("word-four-positions-docids"))?, word_attribute_docids: env.create_database(Some("word-attribute-docids"))?, documents: env.create_database(Some("documents"))?, }) } pub fn put_headers(&self, wtxn: &mut heed::RwTxn, headers: &StringRecord) -> heed::Result<()> { self.main.put::<_, Str, CsvStringRecordCodec>(wtxn, HEADERS_KEY, headers) } pub fn headers(&self, rtxn: &heed::RoTxn) -> heed::Result> { self.main.get::<_, Str, CsvStringRecordCodec>(rtxn, HEADERS_KEY) } pub fn number_of_attributes(&self, rtxn: &heed::RoTxn) -> anyhow::Result> { match self.headers(rtxn)? { Some(headers) => Ok(Some(headers.len())), None => Ok(None), } } pub fn put_fst>(&self, wtxn: &mut heed::RwTxn, fst: &fst::Set) -> anyhow::Result<()> { Ok(self.main.put::<_, Str, ByteSlice>(wtxn, WORDS_FST_KEY, fst.as_fst().as_bytes())?) } pub fn fst<'t>(&self, rtxn: &'t heed::RoTxn) -> anyhow::Result>> { match self.main.get::<_, Str, ByteSlice>(rtxn, WORDS_FST_KEY)? { Some(bytes) => Ok(Some(fst::Set::new(bytes)?)), None => Ok(None), } } /// Returns a [`Vec`] of the requested documents. Returns an error if a document is missing. pub fn documents<'t>( &self, rtxn: &'t heed::RoTxn, iter: impl IntoIterator, ) -> anyhow::Result> { let ids: Vec<_> = iter.into_iter().collect(); let mut content = Vec::new(); for id in ids.iter().cloned() { let document_content = self.documents.get(rtxn, &BEU32::new(id))? .with_context(|| format!("Could not find document {}", id))?; content.extend_from_slice(document_content); } let mut rdr = csv::ReaderBuilder::new().has_headers(false).from_reader(&content[..]); let mut documents = Vec::with_capacity(ids.len()); for (id, result) in ids.into_iter().zip(rdr.records()) { documents.push((id, result?)); } Ok(documents) } /// Returns the number of documents indexed in the database. pub fn number_of_documents<'t>(&self, rtxn: &'t heed::RoTxn) -> anyhow::Result { let docids = self.main.get::<_, Str, RoaringBitmapCodec>(rtxn, DOCUMENTS_IDS_KEY)? .with_context(|| format!("Could not find the list of documents ids"))?; Ok(docids.len() as usize) } pub fn search<'a>(&'a self, rtxn: &'a heed::RoTxn) -> Search<'a> { Search::new(rtxn, self) } }