2021-06-17 23:05:34 +08:00
|
|
|
use std::collections::btree_map::Entry;
|
2021-06-17 00:33:33 +08:00
|
|
|
use std::collections::HashMap;
|
2021-04-01 15:07:16 +08:00
|
|
|
|
2020-11-23 00:53:33 +08:00
|
|
|
use fst::IntoStreamer;
|
2022-03-24 22:22:57 +08:00
|
|
|
use heed::types::{ByteSlice, Str};
|
2022-03-25 17:49:34 +08:00
|
|
|
use heed::{BytesDecode, BytesEncode, Database};
|
2020-10-26 18:01:00 +08:00
|
|
|
use roaring::RoaringBitmap;
|
2021-11-10 03:19:49 +08:00
|
|
|
use serde::{Deserialize, Serialize};
|
2021-02-13 20:57:53 +08:00
|
|
|
use serde_json::Value;
|
2022-02-15 18:41:55 +08:00
|
|
|
use time::OffsetDateTime;
|
2020-10-26 18:01:00 +08:00
|
|
|
|
2021-06-17 00:33:33 +08:00
|
|
|
use super::ClearDocuments;
|
2021-08-20 23:32:02 +08:00
|
|
|
use crate::error::{InternalError, SerializationError, UserError};
|
|
|
|
use crate::heed_codec::facet::{
|
|
|
|
FacetLevelValueU32Codec, FacetStringLevelZeroValueCodec, FacetStringZeroBoundsValueCodec,
|
|
|
|
};
|
2021-04-21 21:43:44 +08:00
|
|
|
use crate::heed_codec::CboRoaringBitmapCodec;
|
2021-06-15 17:06:42 +08:00
|
|
|
use crate::index::{db_name, main_key};
|
2022-03-24 22:22:57 +08:00
|
|
|
use crate::{
|
|
|
|
DocumentId, ExternalDocumentsIds, FieldId, Index, Result, RoaringBitmapCodec, SmallString32,
|
|
|
|
BEU32,
|
|
|
|
};
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
pub struct DeleteDocuments<'t, 'u, 'i> {
|
2020-10-30 18:42:00 +08:00
|
|
|
wtxn: &'t mut heed::RwTxn<'i, 'u>,
|
2020-10-26 18:01:00 +08:00
|
|
|
index: &'i Index,
|
2020-11-23 00:53:33 +08:00
|
|
|
external_documents_ids: ExternalDocumentsIds<'static>,
|
2020-10-26 18:01:00 +08:00
|
|
|
documents_ids: RoaringBitmap,
|
|
|
|
}
|
|
|
|
|
2021-11-10 03:19:49 +08:00
|
|
|
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
|
|
|
pub struct DocumentDeletionResult {
|
|
|
|
pub deleted_documents: u64,
|
|
|
|
pub remaining_documents: u64,
|
|
|
|
}
|
|
|
|
|
2020-10-26 18:01:00 +08:00
|
|
|
impl<'t, 'u, 'i> DeleteDocuments<'t, 'u, 'i> {
|
|
|
|
pub fn new(
|
2020-10-30 18:42:00 +08:00
|
|
|
wtxn: &'t mut heed::RwTxn<'i, 'u>,
|
2020-10-26 18:01:00 +08:00
|
|
|
index: &'i Index,
|
2021-06-17 00:33:33 +08:00
|
|
|
) -> Result<DeleteDocuments<'t, 'u, 'i>> {
|
|
|
|
let external_documents_ids = index.external_documents_ids(wtxn)?.into_static();
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
Ok(DeleteDocuments {
|
|
|
|
wtxn,
|
|
|
|
index,
|
2020-11-22 18:54:04 +08:00
|
|
|
external_documents_ids,
|
2020-10-26 18:01:00 +08:00
|
|
|
documents_ids: RoaringBitmap::new(),
|
|
|
|
})
|
|
|
|
}
|
|
|
|
|
|
|
|
pub fn delete_document(&mut self, docid: u32) {
|
|
|
|
self.documents_ids.insert(docid);
|
|
|
|
}
|
|
|
|
|
|
|
|
pub fn delete_documents(&mut self, docids: &RoaringBitmap) {
|
2021-06-30 20:12:56 +08:00
|
|
|
self.documents_ids |= docids;
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
|
2020-11-22 18:54:04 +08:00
|
|
|
pub fn delete_external_id(&mut self, external_id: &str) -> Option<u32> {
|
2020-11-23 00:53:33 +08:00
|
|
|
let docid = self.external_documents_ids.get(external_id)?;
|
2020-10-26 18:01:00 +08:00
|
|
|
self.delete_document(docid);
|
|
|
|
Some(docid)
|
|
|
|
}
|
|
|
|
|
2021-11-10 03:19:49 +08:00
|
|
|
pub fn execute(self) -> Result<DocumentDeletionResult> {
|
2022-02-15 18:41:55 +08:00
|
|
|
self.index.set_updated_at(self.wtxn, &OffsetDateTime::now_utc())?;
|
2020-10-30 19:14:25 +08:00
|
|
|
// We retrieve the current documents ids that are in the database.
|
2020-10-26 18:01:00 +08:00
|
|
|
let mut documents_ids = self.index.documents_ids(self.wtxn)?;
|
2021-11-10 03:19:49 +08:00
|
|
|
let current_documents_ids_len = documents_ids.len();
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
// We can and must stop removing documents in a database that is empty.
|
|
|
|
if documents_ids.is_empty() {
|
2021-11-10 03:19:49 +08:00
|
|
|
return Ok(DocumentDeletionResult {
|
|
|
|
deleted_documents: 0,
|
|
|
|
remaining_documents: current_documents_ids_len,
|
|
|
|
});
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
|
2020-10-30 19:14:25 +08:00
|
|
|
// We remove the documents ids that we want to delete
|
|
|
|
// from the documents in the database and write them back.
|
2021-06-30 20:12:56 +08:00
|
|
|
documents_ids -= &self.documents_ids;
|
2020-10-26 18:01:00 +08:00
|
|
|
self.index.put_documents_ids(self.wtxn, &documents_ids)?;
|
|
|
|
|
2020-10-29 20:52:00 +08:00
|
|
|
// We can execute a ClearDocuments operation when the number of documents
|
|
|
|
// to delete is exactly the number of documents in the database.
|
|
|
|
if current_documents_ids_len == self.documents_ids.len() {
|
2021-11-03 20:12:01 +08:00
|
|
|
let remaining_documents = ClearDocuments::new(self.wtxn, self.index).execute()?;
|
2021-11-10 03:19:49 +08:00
|
|
|
return Ok(DocumentDeletionResult {
|
|
|
|
deleted_documents: current_documents_ids_len,
|
|
|
|
remaining_documents,
|
|
|
|
});
|
2020-10-29 20:52:00 +08:00
|
|
|
}
|
2020-10-28 18:17:36 +08:00
|
|
|
|
2020-10-26 18:01:00 +08:00
|
|
|
let fields_ids_map = self.index.fields_ids_map(self.wtxn)?;
|
2021-06-14 22:46:19 +08:00
|
|
|
let primary_key = self.index.primary_key(self.wtxn)?.ok_or_else(|| {
|
2021-06-15 17:06:42 +08:00
|
|
|
InternalError::DatabaseMissingEntry {
|
|
|
|
db_name: db_name::MAIN,
|
|
|
|
key: Some(main_key::PRIMARY_KEY_KEY),
|
|
|
|
}
|
2021-06-14 22:46:19 +08:00
|
|
|
})?;
|
2021-07-22 23:11:17 +08:00
|
|
|
|
2021-11-10 03:19:49 +08:00
|
|
|
// Since we already checked if the DB was empty, if we can't find the primary key, then
|
|
|
|
// something is wrong, and we must return an error.
|
2021-07-22 23:11:17 +08:00
|
|
|
let id_field = match fields_ids_map.id(primary_key) {
|
|
|
|
Some(field) => field,
|
2021-11-10 03:19:49 +08:00
|
|
|
None => return Err(UserError::MissingPrimaryKey.into()),
|
2021-07-22 23:11:17 +08:00
|
|
|
};
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
let Index {
|
2020-10-30 17:56:35 +08:00
|
|
|
env: _env,
|
2020-10-26 18:01:00 +08:00
|
|
|
main: _main,
|
|
|
|
word_docids,
|
2022-03-24 22:22:57 +08:00
|
|
|
exact_word_docids,
|
2021-02-03 17:30:33 +08:00
|
|
|
word_prefix_docids,
|
2022-03-25 17:49:34 +08:00
|
|
|
exact_word_prefix_docids,
|
2020-10-26 18:01:00 +08:00
|
|
|
docid_word_positions,
|
|
|
|
word_pair_proximity_docids,
|
2021-05-27 21:27:41 +08:00
|
|
|
field_id_word_count_docids,
|
2021-02-10 17:28:15 +08:00
|
|
|
word_prefix_pair_proximity_docids,
|
2021-10-05 17:18:42 +08:00
|
|
|
word_position_docids,
|
|
|
|
word_prefix_position_docids,
|
2021-04-21 21:43:44 +08:00
|
|
|
facet_id_f64_docids,
|
|
|
|
facet_id_string_docids,
|
|
|
|
field_id_docid_facet_f64s,
|
|
|
|
field_id_docid_facet_strings,
|
2020-10-26 18:01:00 +08:00
|
|
|
documents,
|
|
|
|
} = self.index;
|
|
|
|
|
2021-04-01 15:07:16 +08:00
|
|
|
// Number of fields for each document that has been deleted.
|
|
|
|
let mut fields_ids_distribution_diff = HashMap::new();
|
|
|
|
|
2020-11-22 18:54:04 +08:00
|
|
|
// Retrieve the words and the external documents ids contained in the documents.
|
2020-10-26 18:01:00 +08:00
|
|
|
let mut words = Vec::new();
|
2020-11-22 18:54:04 +08:00
|
|
|
let mut external_ids = Vec::new();
|
2020-10-30 19:14:25 +08:00
|
|
|
for docid in &self.documents_ids {
|
2020-10-26 18:01:00 +08:00
|
|
|
// We create an iterator to be able to get the content and delete the document
|
|
|
|
// content itself. It's faster to acquire a cursor to get and delete,
|
|
|
|
// as we avoid traversing the LMDB B-Tree two times but only once.
|
|
|
|
let key = BEU32::new(docid);
|
|
|
|
let mut iter = documents.range_mut(self.wtxn, &(key..=key))?;
|
|
|
|
if let Some((_key, obkv)) = iter.next().transpose()? {
|
2021-04-01 15:07:16 +08:00
|
|
|
for (field_id, _) in obkv.iter() {
|
|
|
|
*fields_ids_distribution_diff.entry(field_id).or_default() += 1;
|
|
|
|
}
|
|
|
|
|
2020-10-26 18:01:00 +08:00
|
|
|
if let Some(content) = obkv.get(id_field) {
|
2021-02-13 20:57:53 +08:00
|
|
|
let external_id = match serde_json::from_slice(content).unwrap() {
|
|
|
|
Value::String(string) => SmallString32::from(string.as_str()),
|
|
|
|
Value::Number(number) => SmallString32::from(number.to_string()),
|
2021-06-17 00:33:33 +08:00
|
|
|
document_id => {
|
|
|
|
return Err(UserError::InvalidDocumentId { document_id }.into())
|
|
|
|
}
|
2021-02-13 20:57:53 +08:00
|
|
|
};
|
2020-11-22 18:54:04 +08:00
|
|
|
external_ids.push(external_id);
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
drop(iter);
|
|
|
|
|
2021-04-01 15:07:16 +08:00
|
|
|
// We iterate through the words positions of the document id,
|
2020-10-26 18:01:00 +08:00
|
|
|
// retrieve the word and delete the positions.
|
|
|
|
let mut iter = docid_word_positions.prefix_iter_mut(self.wtxn, &(docid, ""))?;
|
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let ((_docid, word), _positions) = result?;
|
|
|
|
// This boolean will indicate if we must remove this word from the words FST.
|
2020-10-29 21:32:32 +08:00
|
|
|
words.push((SmallString32::from(word), false));
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2021-06-17 21:16:20 +08:00
|
|
|
let mut field_distribution = self.index.field_distribution(self.wtxn)?;
|
2021-04-01 15:07:16 +08:00
|
|
|
|
|
|
|
// We use pre-calculated number of fields occurrences that needs to be deleted
|
|
|
|
// to reflect deleted documents.
|
|
|
|
// If all field occurrences are removed, delete the entry from distribution.
|
|
|
|
// Otherwise, insert new number of occurrences (current_count - count_diff).
|
|
|
|
for (field_id, count_diff) in fields_ids_distribution_diff {
|
|
|
|
let field_name = fields_ids_map.name(field_id).unwrap();
|
2021-06-17 21:16:20 +08:00
|
|
|
if let Entry::Occupied(mut entry) = field_distribution.entry(field_name.to_string()) {
|
2021-04-01 15:07:16 +08:00
|
|
|
match entry.get().checked_sub(count_diff) {
|
|
|
|
Some(0) | None => entry.remove(),
|
2021-06-17 00:33:33 +08:00
|
|
|
Some(count) => entry.insert(count),
|
2021-04-01 15:07:16 +08:00
|
|
|
};
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2021-06-17 21:16:20 +08:00
|
|
|
self.index.put_field_distribution(self.wtxn, &field_distribution)?;
|
2021-04-01 15:07:16 +08:00
|
|
|
|
2020-11-22 18:54:04 +08:00
|
|
|
// We create the FST map of the external ids that we must delete.
|
|
|
|
external_ids.sort_unstable();
|
2022-03-15 00:13:07 +08:00
|
|
|
let external_ids_to_delete = fst::Set::from_iter(external_ids)?;
|
2020-10-26 18:01:00 +08:00
|
|
|
|
2020-11-23 00:53:33 +08:00
|
|
|
// We acquire the current external documents ids map...
|
|
|
|
let mut new_external_documents_ids = self.index.external_documents_ids(self.wtxn)?;
|
|
|
|
// ...and remove the to-delete external ids.
|
|
|
|
new_external_documents_ids.delete_ids(external_ids_to_delete)?;
|
2020-10-26 18:01:00 +08:00
|
|
|
|
2020-11-22 18:54:04 +08:00
|
|
|
// We write the new external ids into the main database.
|
2020-11-23 00:53:33 +08:00
|
|
|
let new_external_documents_ids = new_external_documents_ids.into_static();
|
2020-11-22 18:54:04 +08:00
|
|
|
self.index.put_external_documents_ids(self.wtxn, &new_external_documents_ids)?;
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
// Maybe we can improve the get performance of the words
|
|
|
|
// if we sort the words first, keeping the LMDB pages in cache.
|
|
|
|
words.sort_unstable();
|
|
|
|
|
|
|
|
// We iterate over the words and delete the documents ids
|
|
|
|
// from the word docids database.
|
|
|
|
for (word, must_remove) in &mut words {
|
2022-03-24 22:22:57 +08:00
|
|
|
remove_from_word_docids(
|
|
|
|
self.wtxn,
|
|
|
|
word_docids,
|
|
|
|
word.as_str(),
|
|
|
|
must_remove,
|
|
|
|
&self.documents_ids,
|
|
|
|
)?;
|
|
|
|
|
|
|
|
remove_from_word_docids(
|
|
|
|
self.wtxn,
|
|
|
|
exact_word_docids,
|
|
|
|
word.as_str(),
|
|
|
|
must_remove,
|
|
|
|
&self.documents_ids,
|
|
|
|
)?;
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
// We construct an FST set that contains the words to delete from the words FST.
|
2021-06-17 00:33:33 +08:00
|
|
|
let words_to_delete =
|
|
|
|
words.iter().filter_map(
|
|
|
|
|(word, must_remove)| {
|
|
|
|
if *must_remove {
|
2022-03-15 00:13:07 +08:00
|
|
|
Some(word.as_str())
|
2021-06-17 00:33:33 +08:00
|
|
|
} else {
|
|
|
|
None
|
|
|
|
}
|
|
|
|
},
|
|
|
|
);
|
2020-10-26 18:01:00 +08:00
|
|
|
let words_to_delete = fst::Set::from_iter(words_to_delete)?;
|
|
|
|
|
|
|
|
let new_words_fst = {
|
|
|
|
// We retrieve the current words FST from the database.
|
|
|
|
let words_fst = self.index.words_fst(self.wtxn)?;
|
|
|
|
let difference = words_fst.op().add(&words_to_delete).difference();
|
|
|
|
|
2020-11-22 18:54:04 +08:00
|
|
|
// We stream the new external ids that does no more contains the to-delete external ids.
|
2020-10-26 18:01:00 +08:00
|
|
|
let mut new_words_fst_builder = fst::SetBuilder::memory();
|
|
|
|
new_words_fst_builder.extend_stream(difference.into_stream())?;
|
|
|
|
|
|
|
|
// We create an words FST set from the above builder.
|
|
|
|
new_words_fst_builder.into_set()
|
|
|
|
};
|
|
|
|
|
|
|
|
// We write the new words FST into the main database.
|
|
|
|
self.index.put_words_fst(self.wtxn, &new_words_fst)?;
|
|
|
|
|
2022-03-25 17:49:34 +08:00
|
|
|
let prefixes_to_delete =
|
|
|
|
remove_from_word_prefix_docids(self.wtxn, word_prefix_docids, &self.documents_ids)?;
|
2021-02-17 18:22:25 +08:00
|
|
|
|
2022-03-25 17:49:34 +08:00
|
|
|
let exact_prefix_to_delete = remove_from_word_prefix_docids(
|
|
|
|
self.wtxn,
|
|
|
|
exact_word_prefix_docids,
|
|
|
|
&self.documents_ids,
|
|
|
|
)?;
|
|
|
|
|
|
|
|
let all_prefixes_to_delete = prefixes_to_delete.op().add(&exact_prefix_to_delete).union();
|
2021-02-17 18:22:25 +08:00
|
|
|
|
|
|
|
// We compute the new prefix FST and write it only if there is a change.
|
2022-03-25 17:49:34 +08:00
|
|
|
if !prefixes_to_delete.is_empty() || !exact_prefix_to_delete.is_empty() {
|
2021-02-17 18:22:25 +08:00
|
|
|
let new_words_prefixes_fst = {
|
|
|
|
// We retrieve the current words prefixes FST from the database.
|
|
|
|
let words_prefixes_fst = self.index.words_prefixes_fst(self.wtxn)?;
|
2022-03-25 17:49:34 +08:00
|
|
|
let difference =
|
|
|
|
words_prefixes_fst.op().add(all_prefixes_to_delete.into_stream()).difference();
|
2021-02-17 18:22:25 +08:00
|
|
|
|
|
|
|
// We stream the new external ids that does no more contains the to-delete external ids.
|
|
|
|
let mut new_words_prefixes_fst_builder = fst::SetBuilder::memory();
|
|
|
|
new_words_prefixes_fst_builder.extend_stream(difference.into_stream())?;
|
|
|
|
|
|
|
|
// We create an words FST set from the above builder.
|
|
|
|
new_words_prefixes_fst_builder.into_set()
|
|
|
|
};
|
|
|
|
|
|
|
|
// We write the new words prefixes FST into the main database.
|
|
|
|
self.index.put_words_prefixes_fst(self.wtxn, &new_words_prefixes_fst)?;
|
|
|
|
}
|
|
|
|
|
2021-02-10 17:35:25 +08:00
|
|
|
// We delete the documents ids from the word prefix pair proximity database docids
|
|
|
|
// and remove the empty pairs too.
|
|
|
|
let db = word_prefix_pair_proximity_docids.remap_key_type::<ByteSlice>();
|
|
|
|
let mut iter = db.iter_mut(self.wtxn)?;
|
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (key, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= &self.documents_ids;
|
2021-02-10 17:35:25 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-02-10 17:35:25 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let key = key.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &docids)? };
|
2021-02-10 17:35:25 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
drop(iter);
|
2021-02-17 18:22:25 +08:00
|
|
|
|
2020-10-28 01:50:09 +08:00
|
|
|
// We delete the documents ids that are under the pairs of words,
|
|
|
|
// it is faster and use no memory to iterate over all the words pairs than
|
|
|
|
// to compute the cartesian product of every words of the deleted documents.
|
2021-06-17 00:33:33 +08:00
|
|
|
let mut iter =
|
|
|
|
word_pair_proximity_docids.remap_key_type::<ByteSlice>().iter_mut(self.wtxn)?;
|
2020-10-28 01:50:09 +08:00
|
|
|
while let Some(result) = iter.next() {
|
2020-11-19 18:17:53 +08:00
|
|
|
let (bytes, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= &self.documents_ids;
|
2020-10-28 01:50:09 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2020-11-19 18:17:53 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let bytes = bytes.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&bytes, &docids)? };
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2020-11-11 23:04:04 +08:00
|
|
|
drop(iter);
|
|
|
|
|
2021-03-17 22:47:41 +08:00
|
|
|
// We delete the documents ids that are under the word level position docids.
|
2021-10-05 17:18:42 +08:00
|
|
|
let mut iter = word_position_docids.iter_mut(self.wtxn)?.remap_key_type::<ByteSlice>();
|
2021-03-17 22:47:41 +08:00
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (bytes, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= &self.documents_ids;
|
2021-03-17 22:47:41 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-03-17 22:47:41 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let bytes = bytes.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&bytes, &docids)? };
|
2021-03-17 22:47:41 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
drop(iter);
|
|
|
|
|
2021-03-25 18:10:12 +08:00
|
|
|
// We delete the documents ids that are under the word prefix level position docids.
|
2021-06-17 00:33:33 +08:00
|
|
|
let mut iter =
|
2021-10-05 17:18:42 +08:00
|
|
|
word_prefix_position_docids.iter_mut(self.wtxn)?.remap_key_type::<ByteSlice>();
|
2021-03-25 18:10:12 +08:00
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (bytes, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= &self.documents_ids;
|
2021-03-25 18:10:12 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-03-25 18:10:12 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let bytes = bytes.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&bytes, &docids)? };
|
2021-03-25 18:10:12 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
drop(iter);
|
|
|
|
|
2021-06-01 23:04:10 +08:00
|
|
|
// Remove the documents ids from the field id word count database.
|
2021-05-27 21:27:41 +08:00
|
|
|
let mut iter = field_id_word_count_docids.iter_mut(self.wtxn)?;
|
|
|
|
while let Some((key, mut docids)) = iter.next().transpose()? {
|
|
|
|
let previous_len = docids.len();
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= &self.documents_ids;
|
2021-05-27 21:27:41 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-05-27 21:27:41 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-07-21 16:35:35 +08:00
|
|
|
let key = key.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &docids)? };
|
2021-05-27 21:27:41 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
drop(iter);
|
|
|
|
|
2021-08-25 20:58:36 +08:00
|
|
|
if let Some(mut rtree) = self.index.geo_rtree(self.wtxn)? {
|
2021-08-26 23:49:50 +08:00
|
|
|
let mut geo_faceted_doc_ids = self.index.geo_faceted_documents_ids(self.wtxn)?;
|
|
|
|
|
2021-09-09 18:20:08 +08:00
|
|
|
let (points_to_remove, docids_to_remove): (Vec<_>, RoaringBitmap) = rtree
|
2021-08-25 20:58:36 +08:00
|
|
|
.iter()
|
2021-12-14 19:21:24 +08:00
|
|
|
.filter(|&point| self.documents_ids.contains(point.data.0))
|
2021-08-25 20:58:36 +08:00
|
|
|
.cloned()
|
2021-12-14 19:21:24 +08:00
|
|
|
.map(|point| (point, point.data.0))
|
2021-09-09 18:20:08 +08:00
|
|
|
.unzip();
|
2021-08-25 20:58:36 +08:00
|
|
|
points_to_remove.iter().for_each(|point| {
|
|
|
|
rtree.remove(&point);
|
|
|
|
});
|
2021-09-09 18:20:08 +08:00
|
|
|
geo_faceted_doc_ids -= docids_to_remove;
|
2021-08-25 20:58:36 +08:00
|
|
|
|
|
|
|
self.index.put_geo_rtree(self.wtxn, &rtree)?;
|
2021-08-26 23:49:50 +08:00
|
|
|
self.index.put_geo_faceted_documents_ids(self.wtxn, &geo_faceted_doc_ids)?;
|
2021-08-25 20:58:36 +08:00
|
|
|
}
|
|
|
|
|
2021-04-21 21:43:44 +08:00
|
|
|
// We delete the documents ids that are under the facet field id values.
|
2021-07-17 18:50:01 +08:00
|
|
|
remove_docids_from_facet_field_id_number_docids(
|
2021-04-21 21:43:44 +08:00
|
|
|
self.wtxn,
|
|
|
|
facet_id_f64_docids,
|
|
|
|
&self.documents_ids,
|
|
|
|
)?;
|
|
|
|
|
2021-07-17 18:50:01 +08:00
|
|
|
remove_docids_from_facet_field_id_string_docids(
|
2021-04-21 21:43:44 +08:00
|
|
|
self.wtxn,
|
|
|
|
facet_id_string_docids,
|
|
|
|
&self.documents_ids,
|
|
|
|
)?;
|
|
|
|
|
|
|
|
// Remove the documents ids from the faceted documents ids.
|
2021-04-28 23:58:16 +08:00
|
|
|
for field_id in self.index.faceted_fields_ids(self.wtxn)? {
|
|
|
|
// Remove docids from the number faceted documents ids
|
|
|
|
let mut docids = self.index.number_faceted_documents_ids(self.wtxn, field_id)?;
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= &self.documents_ids;
|
2021-04-28 23:58:16 +08:00
|
|
|
self.index.put_number_faceted_documents_ids(self.wtxn, field_id, &docids)?;
|
2021-04-21 21:43:44 +08:00
|
|
|
|
|
|
|
remove_docids_from_field_id_docid_facet_value(
|
|
|
|
self.wtxn,
|
|
|
|
field_id_docid_facet_f64s,
|
|
|
|
field_id,
|
|
|
|
&self.documents_ids,
|
|
|
|
|(_fid, docid, _value)| docid,
|
|
|
|
)?;
|
|
|
|
|
2021-04-28 23:58:16 +08:00
|
|
|
// Remove docids from the string faceted documents ids
|
|
|
|
let mut docids = self.index.string_faceted_documents_ids(self.wtxn, field_id)?;
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= &self.documents_ids;
|
2021-04-28 23:58:16 +08:00
|
|
|
self.index.put_string_faceted_documents_ids(self.wtxn, field_id, &docids)?;
|
|
|
|
|
2021-04-21 21:43:44 +08:00
|
|
|
remove_docids_from_field_id_docid_facet_value(
|
|
|
|
self.wtxn,
|
|
|
|
field_id_docid_facet_strings,
|
|
|
|
field_id,
|
|
|
|
&self.documents_ids,
|
|
|
|
|(_fid, docid, _value)| docid,
|
|
|
|
)?;
|
|
|
|
}
|
|
|
|
|
2021-11-10 03:19:49 +08:00
|
|
|
Ok(DocumentDeletionResult {
|
|
|
|
deleted_documents: self.documents_ids.len(),
|
|
|
|
remaining_documents: documents_ids.len(),
|
|
|
|
})
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
}
|
2021-02-13 21:04:23 +08:00
|
|
|
|
2022-03-25 17:49:34 +08:00
|
|
|
fn remove_from_word_prefix_docids(
|
|
|
|
txn: &mut heed::RwTxn,
|
|
|
|
db: &Database<Str, RoaringBitmapCodec>,
|
|
|
|
to_remove: &RoaringBitmap,
|
|
|
|
) -> Result<fst::Set<Vec<u8>>> {
|
|
|
|
let mut prefixes_to_delete = fst::SetBuilder::memory();
|
|
|
|
|
|
|
|
// We iterate over the word prefix docids database and remove the deleted documents ids
|
|
|
|
// from every docids lists. We register the empty prefixes in an fst Set for futur deletion.
|
|
|
|
let mut iter = db.iter_mut(txn)?;
|
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (prefix, mut docids) = result?;
|
|
|
|
let prefix = prefix.to_owned();
|
|
|
|
let previous_len = docids.len();
|
|
|
|
docids -= to_remove;
|
|
|
|
if docids.is_empty() {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
|
|
|
prefixes_to_delete.insert(prefix)?;
|
|
|
|
} else if docids.len() != previous_len {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&prefix, &docids)? };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(prefixes_to_delete.into_set())
|
|
|
|
}
|
|
|
|
|
2022-03-24 22:22:57 +08:00
|
|
|
fn remove_from_word_docids(
|
|
|
|
txn: &mut heed::RwTxn,
|
|
|
|
db: &heed::Database<Str, RoaringBitmapCodec>,
|
|
|
|
word: &str,
|
|
|
|
must_remove: &mut bool,
|
|
|
|
to_remove: &RoaringBitmap,
|
|
|
|
) -> Result<()> {
|
|
|
|
// We create an iterator to be able to get the content and delete the word docids.
|
|
|
|
// It's faster to acquire a cursor to get and delete or put, as we avoid traversing
|
|
|
|
// the LMDB B-Tree two times but only once.
|
|
|
|
let mut iter = db.prefix_iter_mut(txn, &word)?;
|
|
|
|
if let Some((key, mut docids)) = iter.next().transpose()? {
|
|
|
|
if key == word {
|
|
|
|
let previous_len = docids.len();
|
|
|
|
docids -= to_remove;
|
|
|
|
if docids.is_empty() {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
|
|
|
*must_remove = true;
|
|
|
|
} else if docids.len() != previous_len {
|
|
|
|
let key = key.to_owned();
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &docids)? };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
2022-04-05 20:14:15 +08:00
|
|
|
|
2022-03-24 22:22:57 +08:00
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
|
2021-07-15 16:19:35 +08:00
|
|
|
fn remove_docids_from_field_id_docid_facet_value<'a, C, K, F, DC, V>(
|
2021-04-21 21:43:44 +08:00
|
|
|
wtxn: &'a mut heed::RwTxn,
|
2021-07-15 16:19:35 +08:00
|
|
|
db: &heed::Database<C, DC>,
|
2021-04-21 21:43:44 +08:00
|
|
|
field_id: FieldId,
|
|
|
|
to_remove: &RoaringBitmap,
|
|
|
|
convert: F,
|
|
|
|
) -> heed::Result<()>
|
|
|
|
where
|
2021-07-15 16:19:35 +08:00
|
|
|
C: heed::BytesDecode<'a, DItem = K>,
|
|
|
|
DC: heed::BytesDecode<'a, DItem = V>,
|
2021-04-21 21:43:44 +08:00
|
|
|
F: Fn(K) -> DocumentId,
|
|
|
|
{
|
2021-07-06 17:31:24 +08:00
|
|
|
let mut iter = db
|
|
|
|
.remap_key_type::<ByteSlice>()
|
|
|
|
.prefix_iter_mut(wtxn, &field_id.to_be_bytes())?
|
|
|
|
.remap_key_type::<C>();
|
2021-04-21 21:43:44 +08:00
|
|
|
|
|
|
|
while let Some(result) = iter.next() {
|
2021-07-15 16:19:35 +08:00
|
|
|
let (key, _) = result?;
|
2021-04-21 21:43:44 +08:00
|
|
|
if to_remove.contains(convert(key)) {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-04-21 21:43:44 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
|
2021-08-20 23:32:02 +08:00
|
|
|
fn remove_docids_from_facet_field_id_string_docids<'a, C, D>(
|
2021-07-17 18:50:01 +08:00
|
|
|
wtxn: &'a mut heed::RwTxn,
|
2021-08-20 23:32:02 +08:00
|
|
|
db: &heed::Database<C, D>,
|
2021-07-17 18:50:01 +08:00
|
|
|
to_remove: &RoaringBitmap,
|
2021-08-20 23:32:02 +08:00
|
|
|
) -> crate::Result<()> {
|
|
|
|
let db_name = Some(crate::index::db_name::FACET_ID_STRING_DOCIDS);
|
|
|
|
let mut iter = db.remap_types::<ByteSlice, ByteSlice>().iter_mut(wtxn)?;
|
2021-07-17 18:50:01 +08:00
|
|
|
while let Some(result) = iter.next() {
|
2021-08-20 23:32:02 +08:00
|
|
|
let (key, val) = result?;
|
|
|
|
match FacetLevelValueU32Codec::bytes_decode(key) {
|
|
|
|
Some(_) => {
|
|
|
|
// If we are able to parse this key it means it is a facet string group
|
|
|
|
// level key. We must then parse the value using the appropriate codec.
|
|
|
|
let (group, mut docids) =
|
|
|
|
FacetStringZeroBoundsValueCodec::<CboRoaringBitmapCodec>::bytes_decode(val)
|
|
|
|
.ok_or_else(|| SerializationError::Decoding { db_name })?;
|
|
|
|
|
|
|
|
let previous_len = docids.len();
|
|
|
|
docids -= to_remove;
|
|
|
|
if docids.is_empty() {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
|
|
|
} else if docids.len() != previous_len {
|
|
|
|
let key = key.to_owned();
|
|
|
|
let val = &(group, docids);
|
|
|
|
let value_bytes =
|
|
|
|
FacetStringZeroBoundsValueCodec::<CboRoaringBitmapCodec>::bytes_encode(val)
|
|
|
|
.ok_or_else(|| SerializationError::Encoding { db_name })?;
|
|
|
|
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &value_bytes)? };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
None => {
|
|
|
|
// The key corresponds to a level zero facet string.
|
|
|
|
let (original_value, mut docids) =
|
2021-08-16 19:36:30 +08:00
|
|
|
FacetStringLevelZeroValueCodec::bytes_decode(val)
|
2021-08-20 23:32:02 +08:00
|
|
|
.ok_or_else(|| SerializationError::Decoding { db_name })?;
|
|
|
|
|
|
|
|
let previous_len = docids.len();
|
|
|
|
docids -= to_remove;
|
|
|
|
if docids.is_empty() {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
|
|
|
} else if docids.len() != previous_len {
|
|
|
|
let key = key.to_owned();
|
|
|
|
let val = &(original_value, docids);
|
2021-08-16 19:36:30 +08:00
|
|
|
let value_bytes = FacetStringLevelZeroValueCodec::bytes_encode(val)
|
|
|
|
.ok_or_else(|| SerializationError::Encoding { db_name })?;
|
2021-08-20 23:32:02 +08:00
|
|
|
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &value_bytes)? };
|
|
|
|
}
|
|
|
|
}
|
2021-07-17 18:50:01 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
|
|
|
|
fn remove_docids_from_facet_field_id_number_docids<'a, C>(
|
2021-04-21 21:43:44 +08:00
|
|
|
wtxn: &'a mut heed::RwTxn,
|
|
|
|
db: &heed::Database<C, CboRoaringBitmapCodec>,
|
|
|
|
to_remove: &RoaringBitmap,
|
|
|
|
) -> heed::Result<()>
|
|
|
|
where
|
|
|
|
C: heed::BytesDecode<'a> + heed::BytesEncode<'a>,
|
|
|
|
{
|
|
|
|
let mut iter = db.remap_key_type::<ByteSlice>().iter_mut(wtxn)?;
|
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (bytes, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= to_remove;
|
2021-04-21 21:43:44 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-04-21 21:43:44 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let bytes = bytes.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&bytes, &docids)? };
|
2021-04-21 21:43:44 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
|
2021-02-13 21:04:23 +08:00
|
|
|
#[cfg(test)]
|
|
|
|
mod tests {
|
2021-08-26 23:49:50 +08:00
|
|
|
use std::collections::HashSet;
|
|
|
|
|
2021-08-20 23:32:02 +08:00
|
|
|
use big_s::S;
|
2021-02-13 21:04:23 +08:00
|
|
|
use heed::EnvOpenOptions;
|
2021-08-20 23:32:02 +08:00
|
|
|
use maplit::hashset;
|
2021-02-13 21:04:23 +08:00
|
|
|
|
|
|
|
use super::*;
|
2021-12-08 21:12:07 +08:00
|
|
|
use crate::update::{IndexDocuments, IndexDocumentsConfig, IndexerConfig, Settings};
|
2021-11-06 08:32:12 +08:00
|
|
|
use crate::Filter;
|
2021-02-13 21:04:23 +08:00
|
|
|
|
|
|
|
#[test]
|
|
|
|
fn delete_documents_with_numbers_as_primary_key() {
|
|
|
|
let path = tempfile::tempdir().unwrap();
|
|
|
|
let mut options = EnvOpenOptions::new();
|
|
|
|
options.map_size(10 * 1024 * 1024); // 10 MB
|
|
|
|
let index = Index::new(options, &path).unwrap();
|
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2021-08-31 17:44:15 +08:00
|
|
|
let content = documents!([
|
2021-02-13 21:04:23 +08:00
|
|
|
{ "id": 0, "name": "kevin", "object": { "key1": "value1", "key2": "value2" } },
|
|
|
|
{ "id": 1, "name": "kevina", "array": ["I", "am", "fine"] },
|
|
|
|
{ "id": 2, "name": "benoit", "array_of_object": [{ "wow": "amazing" }] }
|
2021-08-31 17:44:15 +08:00
|
|
|
]);
|
2021-12-08 21:12:07 +08:00
|
|
|
let config = IndexerConfig::default();
|
|
|
|
let indexing_config = IndexDocumentsConfig::default();
|
|
|
|
let mut builder = IndexDocuments::new(&mut wtxn, &index, &config, indexing_config, |_| ());
|
|
|
|
builder.add_documents(content).unwrap();
|
|
|
|
builder.execute().unwrap();
|
2021-02-13 21:04:23 +08:00
|
|
|
|
|
|
|
// delete those documents, ids are synchronous therefore 0, 1, and 2.
|
2021-11-03 20:12:01 +08:00
|
|
|
let mut builder = DeleteDocuments::new(&mut wtxn, &index).unwrap();
|
2021-02-13 21:04:23 +08:00
|
|
|
builder.delete_document(0);
|
|
|
|
builder.delete_document(1);
|
|
|
|
builder.delete_document(2);
|
|
|
|
builder.execute().unwrap();
|
|
|
|
|
|
|
|
wtxn.commit().unwrap();
|
2021-04-01 15:07:16 +08:00
|
|
|
|
|
|
|
let rtxn = index.read_txn().unwrap();
|
|
|
|
|
2021-06-17 21:16:20 +08:00
|
|
|
assert!(index.field_distribution(&rtxn).unwrap().is_empty());
|
2021-02-13 21:04:23 +08:00
|
|
|
}
|
2021-06-08 23:33:29 +08:00
|
|
|
|
|
|
|
#[test]
|
|
|
|
fn delete_documents_with_strange_primary_key() {
|
|
|
|
let path = tempfile::tempdir().unwrap();
|
|
|
|
let mut options = EnvOpenOptions::new();
|
|
|
|
options.map_size(10 * 1024 * 1024); // 10 MB
|
|
|
|
let index = Index::new(options, &path).unwrap();
|
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2021-08-31 17:44:15 +08:00
|
|
|
let content = documents!([
|
2021-06-08 23:33:29 +08:00
|
|
|
{ "mysuperid": 0, "name": "kevin" },
|
|
|
|
{ "mysuperid": 1, "name": "kevina" },
|
|
|
|
{ "mysuperid": 2, "name": "benoit" }
|
2021-08-31 17:44:15 +08:00
|
|
|
]);
|
2021-12-08 21:12:07 +08:00
|
|
|
|
|
|
|
let config = IndexerConfig::default();
|
|
|
|
let indexing_config = IndexDocumentsConfig::default();
|
|
|
|
let mut builder = IndexDocuments::new(&mut wtxn, &index, &config, indexing_config, |_| ());
|
|
|
|
builder.add_documents(content).unwrap();
|
|
|
|
builder.execute().unwrap();
|
2021-06-08 23:33:29 +08:00
|
|
|
|
|
|
|
// Delete not all of the documents but some of them.
|
2021-11-03 20:12:01 +08:00
|
|
|
let mut builder = DeleteDocuments::new(&mut wtxn, &index).unwrap();
|
2021-06-08 23:33:29 +08:00
|
|
|
builder.delete_external_id("0");
|
|
|
|
builder.delete_external_id("1");
|
|
|
|
builder.execute().unwrap();
|
|
|
|
|
|
|
|
wtxn.commit().unwrap();
|
|
|
|
}
|
2021-08-20 23:32:02 +08:00
|
|
|
|
|
|
|
#[test]
|
|
|
|
fn delete_documents_with_filterable_attributes() {
|
|
|
|
let path = tempfile::tempdir().unwrap();
|
|
|
|
let mut options = EnvOpenOptions::new();
|
|
|
|
options.map_size(10 * 1024 * 1024); // 10 MB
|
|
|
|
let index = Index::new(options, &path).unwrap();
|
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2021-12-08 21:12:07 +08:00
|
|
|
let config = IndexerConfig::default();
|
|
|
|
let mut builder = Settings::new(&mut wtxn, &index, &config);
|
2021-08-20 23:32:02 +08:00
|
|
|
builder.set_primary_key(S("docid"));
|
|
|
|
builder.set_filterable_fields(hashset! { S("label") });
|
2021-11-03 20:12:01 +08:00
|
|
|
builder.execute(|_| ()).unwrap();
|
2021-08-20 23:32:02 +08:00
|
|
|
|
2021-08-31 17:44:15 +08:00
|
|
|
let content = documents!([
|
2021-08-20 23:32:02 +08:00
|
|
|
{"docid":"1_4","label":"sign"},
|
|
|
|
{"docid":"1_5","label":"letter"},
|
|
|
|
{"docid":"1_7","label":"abstract,cartoon,design,pattern"},
|
|
|
|
{"docid":"1_36","label":"drawing,painting,pattern"},
|
|
|
|
{"docid":"1_37","label":"art,drawing,outdoor"},
|
|
|
|
{"docid":"1_38","label":"aquarium,art,drawing"},
|
|
|
|
{"docid":"1_39","label":"abstract"},
|
|
|
|
{"docid":"1_40","label":"cartoon"},
|
|
|
|
{"docid":"1_41","label":"art,drawing"},
|
|
|
|
{"docid":"1_42","label":"art,pattern"},
|
|
|
|
{"docid":"1_43","label":"abstract,art,drawing,pattern"},
|
|
|
|
{"docid":"1_44","label":"drawing"},
|
|
|
|
{"docid":"1_45","label":"art"},
|
|
|
|
{"docid":"1_46","label":"abstract,colorfulness,pattern"},
|
|
|
|
{"docid":"1_47","label":"abstract,pattern"},
|
|
|
|
{"docid":"1_52","label":"abstract,cartoon"},
|
|
|
|
{"docid":"1_57","label":"abstract,drawing,pattern"},
|
|
|
|
{"docid":"1_58","label":"abstract,art,cartoon"},
|
|
|
|
{"docid":"1_68","label":"design"},
|
|
|
|
{"docid":"1_69","label":"geometry"}
|
2021-08-31 17:44:15 +08:00
|
|
|
]);
|
2021-12-08 21:12:07 +08:00
|
|
|
|
|
|
|
let config = IndexerConfig::default();
|
|
|
|
let indexing_config = IndexDocumentsConfig::default();
|
|
|
|
let mut builder = IndexDocuments::new(&mut wtxn, &index, &config, indexing_config, |_| ());
|
|
|
|
builder.add_documents(content).unwrap();
|
|
|
|
builder.execute().unwrap();
|
2021-08-20 23:32:02 +08:00
|
|
|
|
|
|
|
// Delete not all of the documents but some of them.
|
2021-11-03 20:12:01 +08:00
|
|
|
let mut builder = DeleteDocuments::new(&mut wtxn, &index).unwrap();
|
2021-08-20 23:32:02 +08:00
|
|
|
builder.delete_external_id("1_4");
|
|
|
|
builder.execute().unwrap();
|
|
|
|
|
2021-12-09 18:13:12 +08:00
|
|
|
let filter = Filter::from_str("label = sign").unwrap().unwrap();
|
2021-08-20 23:32:02 +08:00
|
|
|
let results = index.search(&wtxn).filter(filter).execute().unwrap();
|
|
|
|
assert!(results.documents_ids.is_empty());
|
|
|
|
|
|
|
|
wtxn.commit().unwrap();
|
|
|
|
}
|
2021-08-26 19:27:32 +08:00
|
|
|
|
|
|
|
#[test]
|
|
|
|
fn delete_documents_with_geo_points() {
|
|
|
|
let path = tempfile::tempdir().unwrap();
|
|
|
|
let mut options = EnvOpenOptions::new();
|
|
|
|
options.map_size(10 * 1024 * 1024); // 10 MB
|
|
|
|
let index = Index::new(options, &path).unwrap();
|
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2021-12-08 21:12:07 +08:00
|
|
|
let config = IndexerConfig::default();
|
|
|
|
let mut builder = Settings::new(&mut wtxn, &index, &config);
|
2021-08-26 19:27:32 +08:00
|
|
|
builder.set_primary_key(S("id"));
|
2021-08-30 21:47:33 +08:00
|
|
|
builder.set_filterable_fields(hashset!(S("_geo")));
|
|
|
|
builder.set_sortable_fields(hashset!(S("_geo")));
|
2021-11-03 20:12:01 +08:00
|
|
|
builder.execute(|_| ()).unwrap();
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2021-08-31 17:44:15 +08:00
|
|
|
let content = documents!([
|
2021-08-26 19:27:32 +08:00
|
|
|
{"id":"1","city":"Lille", "_geo": { "lat": 50.629973371633746, "lng": 3.0569447399419570 } },
|
|
|
|
{"id":"2","city":"Mons-en-Barœul", "_geo": { "lat": 50.641586120121050, "lng": 3.1106593480348670 } },
|
|
|
|
{"id":"3","city":"Hellemmes", "_geo": { "lat": 50.631220965518080, "lng": 3.1106399673339933 } },
|
|
|
|
{"id":"4","city":"Villeneuve-d'Ascq", "_geo": { "lat": 50.622468098014565, "lng": 3.1476425513437140 } },
|
|
|
|
{"id":"5","city":"Hem", "_geo": { "lat": 50.655250871381355, "lng": 3.1897297266244130 } },
|
|
|
|
{"id":"6","city":"Roubaix", "_geo": { "lat": 50.692473451896710, "lng": 3.1763326737747650 } },
|
|
|
|
{"id":"7","city":"Tourcoing", "_geo": { "lat": 50.726397466736480, "lng": 3.1541653659578670 } },
|
|
|
|
{"id":"8","city":"Mouscron", "_geo": { "lat": 50.745325554908610, "lng": 3.2206407854429853 } },
|
|
|
|
{"id":"9","city":"Tournai", "_geo": { "lat": 50.605342528602630, "lng": 3.3758586941351414 } },
|
|
|
|
{"id":"10","city":"Ghent", "_geo": { "lat": 51.053777403679035, "lng": 3.6957733119926930 } },
|
|
|
|
{"id":"11","city":"Brussels", "_geo": { "lat": 50.846640974544690, "lng": 4.3370663564281840 } },
|
|
|
|
{"id":"12","city":"Charleroi", "_geo": { "lat": 50.409570138889480, "lng": 4.4347354315085520 } },
|
|
|
|
{"id":"13","city":"Mons", "_geo": { "lat": 50.450294178855420, "lng": 3.9623722870904690 } },
|
|
|
|
{"id":"14","city":"Valenciennes", "_geo": { "lat": 50.351817774473545, "lng": 3.5326283646928800 } },
|
|
|
|
{"id":"15","city":"Arras", "_geo": { "lat": 50.284487528579950, "lng": 2.7637515844478160 } },
|
|
|
|
{"id":"16","city":"Cambrai", "_geo": { "lat": 50.179340577906700, "lng": 3.2189409952502930 } },
|
|
|
|
{"id":"17","city":"Bapaume", "_geo": { "lat": 50.111276127236400, "lng": 2.8547894666083120 } },
|
|
|
|
{"id":"18","city":"Amiens", "_geo": { "lat": 49.931472529669996, "lng": 2.2710499758317080 } },
|
|
|
|
{"id":"19","city":"Compiègne", "_geo": { "lat": 49.444980887725656, "lng": 2.7913841281529015 } },
|
|
|
|
{"id":"20","city":"Paris", "_geo": { "lat": 48.902100060895480, "lng": 2.3708400867406930 } }
|
2021-08-31 17:44:15 +08:00
|
|
|
]);
|
2021-08-26 19:27:32 +08:00
|
|
|
let external_ids_to_delete = ["5", "6", "7", "12", "17", "19"];
|
|
|
|
|
2021-12-08 21:12:07 +08:00
|
|
|
let indexing_config = IndexDocumentsConfig::default();
|
|
|
|
|
|
|
|
let mut builder = IndexDocuments::new(&mut wtxn, &index, &config, indexing_config, |_| ());
|
|
|
|
builder.add_documents(content).unwrap();
|
|
|
|
builder.execute().unwrap();
|
2021-08-26 19:27:32 +08:00
|
|
|
|
|
|
|
let external_document_ids = index.external_documents_ids(&wtxn).unwrap();
|
|
|
|
let ids_to_delete: Vec<u32> = external_ids_to_delete
|
|
|
|
.iter()
|
|
|
|
.map(|id| external_document_ids.get(id.as_bytes()).unwrap())
|
|
|
|
.collect();
|
|
|
|
|
|
|
|
// Delete some documents.
|
2021-11-03 20:12:01 +08:00
|
|
|
let mut builder = DeleteDocuments::new(&mut wtxn, &index).unwrap();
|
2021-08-26 19:27:32 +08:00
|
|
|
external_ids_to_delete.iter().for_each(|id| drop(builder.delete_external_id(id)));
|
|
|
|
builder.execute().unwrap();
|
|
|
|
|
|
|
|
wtxn.commit().unwrap();
|
|
|
|
|
|
|
|
let rtxn = index.read_txn().unwrap();
|
|
|
|
let rtree = index.geo_rtree(&rtxn).unwrap().unwrap();
|
2021-08-26 23:49:50 +08:00
|
|
|
let geo_faceted_doc_ids = index.geo_faceted_documents_ids(&rtxn).unwrap();
|
2021-08-26 19:27:32 +08:00
|
|
|
|
|
|
|
let all_geo_ids = rtree.iter().map(|point| point.data).collect::<Vec<_>>();
|
2021-08-26 23:49:50 +08:00
|
|
|
let all_geo_documents = index
|
2021-12-14 19:21:24 +08:00
|
|
|
.documents(&rtxn, all_geo_ids.iter().map(|(id, _)| id).copied())
|
2021-08-26 23:49:50 +08:00
|
|
|
.unwrap()
|
|
|
|
.iter()
|
|
|
|
.map(|(id, _)| *id)
|
|
|
|
.collect::<HashSet<_>>();
|
|
|
|
|
|
|
|
let all_geo_faceted_ids = geo_faceted_doc_ids.iter().collect::<Vec<_>>();
|
|
|
|
let all_geo_faceted_documents = index
|
|
|
|
.documents(&rtxn, all_geo_faceted_ids.iter().copied())
|
|
|
|
.unwrap()
|
|
|
|
.iter()
|
|
|
|
.map(|(id, _)| *id)
|
|
|
|
.collect::<HashSet<_>>();
|
|
|
|
|
|
|
|
assert_eq!(
|
|
|
|
all_geo_documents, all_geo_faceted_documents,
|
|
|
|
"There is an inconsistency between the geo_faceted database and the rtree"
|
|
|
|
);
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2021-08-26 23:49:50 +08:00
|
|
|
for id in all_geo_documents.iter() {
|
2021-08-26 19:27:32 +08:00
|
|
|
assert!(!ids_to_delete.contains(&id), "The document {} was supposed to be deleted", id);
|
|
|
|
}
|
|
|
|
|
|
|
|
assert_eq!(
|
|
|
|
all_geo_ids.len(),
|
|
|
|
all_geo_documents.len(),
|
|
|
|
"We deleted documents that were not supposed to be deleted"
|
|
|
|
);
|
|
|
|
}
|
2021-02-13 21:04:23 +08:00
|
|
|
}
|