2021-06-17 23:05:34 +08:00
|
|
|
use std::collections::btree_map::Entry;
|
2022-09-21 21:53:39 +08:00
|
|
|
use std::collections::{HashMap, HashSet};
|
2021-04-01 15:07:16 +08:00
|
|
|
|
2020-11-23 00:53:33 +08:00
|
|
|
use fst::IntoStreamer;
|
2022-09-08 19:28:17 +08:00
|
|
|
use heed::types::{ByteSlice, DecodeIgnore, Str};
|
2022-09-01 14:34:26 +08:00
|
|
|
use heed::Database;
|
2020-10-26 18:01:00 +08:00
|
|
|
use roaring::RoaringBitmap;
|
2021-11-10 03:19:49 +08:00
|
|
|
use serde::{Deserialize, Serialize};
|
2022-02-15 18:41:55 +08:00
|
|
|
use time::OffsetDateTime;
|
2020-10-26 18:01:00 +08:00
|
|
|
|
2022-09-21 21:53:39 +08:00
|
|
|
use super::facet::delete::FacetsDelete;
|
|
|
|
use super::ClearDocuments;
|
2022-12-20 17:37:50 +08:00
|
|
|
use crate::error::InternalError;
|
2022-09-01 14:34:26 +08:00
|
|
|
use crate::facet::FacetType;
|
2022-09-21 21:53:39 +08:00
|
|
|
use crate::heed_codec::facet::FieldDocIdFacetCodec;
|
2021-04-21 21:43:44 +08:00
|
|
|
use crate::heed_codec::CboRoaringBitmapCodec;
|
2022-03-24 22:22:57 +08:00
|
|
|
use crate::{
|
2022-09-21 21:53:39 +08:00
|
|
|
ExternalDocumentsIds, FieldId, FieldIdMapMissingEntry, Index, Result, RoaringBitmapCodec,
|
|
|
|
SmallString32, BEU32,
|
2022-03-24 22:22:57 +08:00
|
|
|
};
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
pub struct DeleteDocuments<'t, 'u, 'i> {
|
2020-10-30 18:42:00 +08:00
|
|
|
wtxn: &'t mut heed::RwTxn<'i, 'u>,
|
2020-10-26 18:01:00 +08:00
|
|
|
index: &'i Index,
|
2020-11-23 00:53:33 +08:00
|
|
|
external_documents_ids: ExternalDocumentsIds<'static>,
|
2022-06-13 23:59:34 +08:00
|
|
|
to_delete_docids: RoaringBitmap,
|
2022-12-19 16:47:54 +08:00
|
|
|
strategy: DeletionStrategy,
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
|
2022-12-13 17:15:22 +08:00
|
|
|
/// Result of a [`DeleteDocuments`] operation.
|
2021-11-10 03:19:49 +08:00
|
|
|
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
|
|
|
pub struct DocumentDeletionResult {
|
|
|
|
pub deleted_documents: u64,
|
|
|
|
pub remaining_documents: u64,
|
|
|
|
}
|
2022-12-13 17:15:22 +08:00
|
|
|
|
2022-12-19 16:47:54 +08:00
|
|
|
/// Strategy for deleting documents.
|
|
|
|
///
|
|
|
|
/// - Soft-deleted documents are simply marked as deleted without being actually removed from DB.
|
|
|
|
/// - Hard-deleted documents are definitely suppressed from the DB.
|
|
|
|
///
|
|
|
|
/// Soft-deleted documents trade disk space for runtime performance.
|
|
|
|
///
|
|
|
|
/// Note that any of these variants can be used at any given moment for any indexation in a database.
|
|
|
|
/// For instance, you can use an [`AlwaysSoft`] followed by an [`AlwaysHard`] option without issue.
|
|
|
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Default)]
|
|
|
|
pub enum DeletionStrategy {
|
|
|
|
#[default]
|
2022-12-20 01:23:50 +08:00
|
|
|
/// Definitely suppress documents according to the number or size of soft-deleted documents
|
2022-12-19 16:47:54 +08:00
|
|
|
Dynamic,
|
|
|
|
/// Never definitely suppress documents
|
|
|
|
AlwaysSoft,
|
|
|
|
/// Always definitely suppress documents
|
|
|
|
AlwaysHard,
|
|
|
|
}
|
|
|
|
|
|
|
|
impl std::fmt::Display for DeletionStrategy {
|
|
|
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
|
|
|
match self {
|
|
|
|
DeletionStrategy::Dynamic => write!(f, "dynamic"),
|
|
|
|
DeletionStrategy::AlwaysSoft => write!(f, "always_soft"),
|
|
|
|
DeletionStrategy::AlwaysHard => write!(f, "always_hard"),
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2022-12-13 17:15:22 +08:00
|
|
|
/// Result of a [`DeleteDocuments`] operation, used for internal purposes.
|
|
|
|
///
|
|
|
|
/// It is a superset of the [`DocumentDeletionResult`] structure, giving
|
|
|
|
/// additional information about the algorithm used to delete the documents.
|
2022-12-12 19:42:55 +08:00
|
|
|
#[derive(Debug)]
|
2022-12-13 17:15:22 +08:00
|
|
|
pub(crate) struct DetailedDocumentDeletionResult {
|
2022-12-12 19:42:55 +08:00
|
|
|
pub deleted_documents: u64,
|
|
|
|
pub remaining_documents: u64,
|
2022-12-13 17:15:22 +08:00
|
|
|
pub soft_deletion_used: bool,
|
2022-12-12 19:42:55 +08:00
|
|
|
}
|
2021-11-10 03:19:49 +08:00
|
|
|
|
2020-10-26 18:01:00 +08:00
|
|
|
impl<'t, 'u, 'i> DeleteDocuments<'t, 'u, 'i> {
|
|
|
|
pub fn new(
|
2020-10-30 18:42:00 +08:00
|
|
|
wtxn: &'t mut heed::RwTxn<'i, 'u>,
|
2020-10-26 18:01:00 +08:00
|
|
|
index: &'i Index,
|
2021-06-17 00:33:33 +08:00
|
|
|
) -> Result<DeleteDocuments<'t, 'u, 'i>> {
|
|
|
|
let external_documents_ids = index.external_documents_ids(wtxn)?.into_static();
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
Ok(DeleteDocuments {
|
|
|
|
wtxn,
|
|
|
|
index,
|
2020-11-22 18:54:04 +08:00
|
|
|
external_documents_ids,
|
2022-06-13 23:59:34 +08:00
|
|
|
to_delete_docids: RoaringBitmap::new(),
|
2022-12-19 16:47:54 +08:00
|
|
|
strategy: Default::default(),
|
2020-10-26 18:01:00 +08:00
|
|
|
})
|
|
|
|
}
|
|
|
|
|
2022-12-19 16:47:54 +08:00
|
|
|
pub fn strategy(&mut self, strategy: DeletionStrategy) {
|
|
|
|
self.strategy = strategy;
|
2022-09-21 23:16:11 +08:00
|
|
|
}
|
|
|
|
|
2020-10-26 18:01:00 +08:00
|
|
|
pub fn delete_document(&mut self, docid: u32) {
|
2022-06-13 23:59:34 +08:00
|
|
|
self.to_delete_docids.insert(docid);
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
pub fn delete_documents(&mut self, docids: &RoaringBitmap) {
|
2022-06-13 23:59:34 +08:00
|
|
|
self.to_delete_docids |= docids;
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
|
2020-11-22 18:54:04 +08:00
|
|
|
pub fn delete_external_id(&mut self, external_id: &str) -> Option<u32> {
|
2020-11-23 00:53:33 +08:00
|
|
|
let docid = self.external_documents_ids.get(external_id)?;
|
2020-10-26 18:01:00 +08:00
|
|
|
self.delete_document(docid);
|
|
|
|
Some(docid)
|
|
|
|
}
|
2022-12-12 19:42:55 +08:00
|
|
|
pub fn execute(self) -> Result<DocumentDeletionResult> {
|
|
|
|
let DetailedDocumentDeletionResult {
|
|
|
|
deleted_documents,
|
|
|
|
remaining_documents,
|
2022-12-13 17:15:22 +08:00
|
|
|
soft_deletion_used: _,
|
2022-12-12 19:42:55 +08:00
|
|
|
} = self.execute_inner()?;
|
|
|
|
|
|
|
|
Ok(DocumentDeletionResult { deleted_documents, remaining_documents })
|
|
|
|
}
|
|
|
|
pub(crate) fn execute_inner(mut self) -> Result<DetailedDocumentDeletionResult> {
|
2022-02-15 18:41:55 +08:00
|
|
|
self.index.set_updated_at(self.wtxn, &OffsetDateTime::now_utc())?;
|
2022-08-31 14:10:45 +08:00
|
|
|
|
2020-10-30 19:14:25 +08:00
|
|
|
// We retrieve the current documents ids that are in the database.
|
2020-10-26 18:01:00 +08:00
|
|
|
let mut documents_ids = self.index.documents_ids(self.wtxn)?;
|
2022-06-13 23:59:34 +08:00
|
|
|
let mut soft_deleted_docids = self.index.soft_deleted_documents_ids(self.wtxn)?;
|
2021-11-10 03:19:49 +08:00
|
|
|
let current_documents_ids_len = documents_ids.len();
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
// We can and must stop removing documents in a database that is empty.
|
|
|
|
if documents_ids.is_empty() {
|
2022-06-13 23:59:34 +08:00
|
|
|
// but if there was still documents to delete we clear the database entirely
|
|
|
|
if !soft_deleted_docids.is_empty() {
|
|
|
|
ClearDocuments::new(self.wtxn, self.index).execute()?;
|
|
|
|
}
|
2022-12-12 19:42:55 +08:00
|
|
|
return Ok(DetailedDocumentDeletionResult {
|
|
|
|
deleted_documents: 0,
|
|
|
|
remaining_documents: 0,
|
2022-12-13 17:15:22 +08:00
|
|
|
soft_deletion_used: false,
|
2022-12-12 19:42:55 +08:00
|
|
|
});
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
|
2020-10-30 19:14:25 +08:00
|
|
|
// We remove the documents ids that we want to delete
|
|
|
|
// from the documents in the database and write them back.
|
2022-06-13 23:59:34 +08:00
|
|
|
documents_ids -= &self.to_delete_docids;
|
2020-10-26 18:01:00 +08:00
|
|
|
self.index.put_documents_ids(self.wtxn, &documents_ids)?;
|
|
|
|
|
2020-10-29 20:52:00 +08:00
|
|
|
// We can execute a ClearDocuments operation when the number of documents
|
|
|
|
// to delete is exactly the number of documents in the database.
|
2022-06-13 23:59:34 +08:00
|
|
|
if current_documents_ids_len == self.to_delete_docids.len() {
|
2021-11-03 20:12:01 +08:00
|
|
|
let remaining_documents = ClearDocuments::new(self.wtxn, self.index).execute()?;
|
2022-12-12 19:42:55 +08:00
|
|
|
return Ok(DetailedDocumentDeletionResult {
|
2021-11-10 03:19:49 +08:00
|
|
|
deleted_documents: current_documents_ids_len,
|
|
|
|
remaining_documents,
|
2022-12-13 17:15:22 +08:00
|
|
|
soft_deletion_used: false,
|
2021-11-10 03:19:49 +08:00
|
|
|
});
|
2020-10-29 20:52:00 +08:00
|
|
|
}
|
2020-10-28 18:17:36 +08:00
|
|
|
|
2020-10-26 18:01:00 +08:00
|
|
|
let fields_ids_map = self.index.fields_ids_map(self.wtxn)?;
|
2022-06-13 23:59:34 +08:00
|
|
|
let mut field_distribution = self.index.field_distribution(self.wtxn)?;
|
|
|
|
|
|
|
|
// we update the field distribution
|
|
|
|
for docid in self.to_delete_docids.iter() {
|
|
|
|
let key = BEU32::new(docid);
|
|
|
|
let document =
|
|
|
|
self.index.documents.get(self.wtxn, &key)?.ok_or(
|
|
|
|
InternalError::DatabaseMissingEntry { db_name: "documents", key: None },
|
|
|
|
)?;
|
|
|
|
for (fid, _value) in document.iter() {
|
|
|
|
let field_name =
|
|
|
|
fields_ids_map.name(fid).ok_or(FieldIdMapMissingEntry::FieldId {
|
|
|
|
field_id: fid,
|
|
|
|
process: "delete documents",
|
|
|
|
})?;
|
|
|
|
if let Entry::Occupied(mut entry) = field_distribution.entry(field_name.to_string())
|
|
|
|
{
|
|
|
|
match entry.get().checked_sub(1) {
|
|
|
|
Some(0) | None => entry.remove(),
|
|
|
|
Some(count) => entry.insert(count),
|
|
|
|
};
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
self.index.put_field_distribution(self.wtxn, &field_distribution)?;
|
|
|
|
|
|
|
|
soft_deleted_docids |= &self.to_delete_docids;
|
|
|
|
|
2022-12-20 17:37:50 +08:00
|
|
|
// We always soft-delete the documents, even if they will be permanently
|
|
|
|
// deleted immediately after.
|
|
|
|
self.index.put_soft_deleted_documents_ids(self.wtxn, &soft_deleted_docids)?;
|
|
|
|
|
2022-12-19 16:38:59 +08:00
|
|
|
// decide for a hard or soft deletion depending on the strategy
|
|
|
|
let soft_deletion = match self.strategy {
|
|
|
|
DeletionStrategy::Dynamic => {
|
2022-12-15 19:04:46 +08:00
|
|
|
// decide to keep the soft deleted in the DB for now if they meet 2 criteria:
|
|
|
|
// 1. There is less than a fixed rate of 50% of soft-deleted to actual documents, *and*
|
|
|
|
// 2. Soft-deleted occupy an average of less than a fixed size on disk
|
|
|
|
|
2022-12-19 16:38:59 +08:00
|
|
|
let size_used = self.index.used_size()?;
|
|
|
|
let nb_documents = self.index.number_of_documents(self.wtxn)?;
|
|
|
|
let nb_soft_deleted = soft_deleted_docids.len();
|
|
|
|
|
2022-12-15 19:04:46 +08:00
|
|
|
(nb_soft_deleted < nb_documents) && {
|
|
|
|
const SOFT_DELETED_SIZE_BYTE_THRESHOLD: u64 = 1_073_741_824; // 1GiB
|
|
|
|
|
|
|
|
// nb_documents + nb_soft_deleted !=0 because if nb_documents is 0 we short-circuit earlier, and then we moved the documents to delete
|
|
|
|
// from the documents_docids to the soft_deleted_docids.
|
|
|
|
let estimated_document_size = size_used / (nb_documents + nb_soft_deleted);
|
|
|
|
let estimated_size_used_by_soft_deleted =
|
|
|
|
estimated_document_size * nb_soft_deleted;
|
|
|
|
estimated_size_used_by_soft_deleted < SOFT_DELETED_SIZE_BYTE_THRESHOLD
|
|
|
|
}
|
2022-12-19 16:38:59 +08:00
|
|
|
}
|
|
|
|
DeletionStrategy::AlwaysSoft => true,
|
|
|
|
DeletionStrategy::AlwaysHard => false,
|
|
|
|
};
|
|
|
|
|
|
|
|
if soft_deletion {
|
|
|
|
// Keep the soft-deleted in the DB
|
2022-12-12 19:42:55 +08:00
|
|
|
return Ok(DetailedDocumentDeletionResult {
|
2022-06-13 23:59:34 +08:00
|
|
|
deleted_documents: self.to_delete_docids.len(),
|
|
|
|
remaining_documents: documents_ids.len(),
|
2022-12-13 17:15:22 +08:00
|
|
|
soft_deletion_used: true,
|
2022-06-13 23:59:34 +08:00
|
|
|
});
|
|
|
|
}
|
|
|
|
|
|
|
|
self.to_delete_docids = soft_deleted_docids;
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
let Index {
|
2020-10-30 17:56:35 +08:00
|
|
|
env: _env,
|
2020-10-26 18:01:00 +08:00
|
|
|
main: _main,
|
|
|
|
word_docids,
|
2022-03-24 22:22:57 +08:00
|
|
|
exact_word_docids,
|
2021-02-03 17:30:33 +08:00
|
|
|
word_prefix_docids,
|
2022-03-25 17:49:34 +08:00
|
|
|
exact_word_prefix_docids,
|
2020-10-26 18:01:00 +08:00
|
|
|
docid_word_positions,
|
|
|
|
word_pair_proximity_docids,
|
2021-05-27 21:27:41 +08:00
|
|
|
field_id_word_count_docids,
|
2021-02-10 17:28:15 +08:00
|
|
|
word_prefix_pair_proximity_docids,
|
2022-09-14 21:33:13 +08:00
|
|
|
prefix_word_pair_proximity_docids,
|
2021-10-05 17:18:42 +08:00
|
|
|
word_position_docids,
|
|
|
|
word_prefix_position_docids,
|
2022-09-01 14:34:26 +08:00
|
|
|
facet_id_f64_docids: _,
|
|
|
|
facet_id_string_docids: _,
|
2022-09-08 19:28:17 +08:00
|
|
|
field_id_docid_facet_f64s: _,
|
|
|
|
field_id_docid_facet_strings: _,
|
2022-10-12 19:28:36 +08:00
|
|
|
script_language_docids,
|
2022-09-08 19:28:17 +08:00
|
|
|
facet_id_exists_docids,
|
2023-03-08 23:14:00 +08:00
|
|
|
facet_id_is_null_docids,
|
2020-10-26 18:01:00 +08:00
|
|
|
documents,
|
|
|
|
} = self.index;
|
|
|
|
|
2022-12-20 17:37:50 +08:00
|
|
|
// Retrieve the words contained in the documents.
|
2020-10-26 18:01:00 +08:00
|
|
|
let mut words = Vec::new();
|
2022-06-13 23:59:34 +08:00
|
|
|
for docid in &self.to_delete_docids {
|
2022-12-20 17:37:50 +08:00
|
|
|
documents.delete(self.wtxn, &BEU32::new(docid))?;
|
2020-10-26 18:01:00 +08:00
|
|
|
|
2022-12-20 17:37:50 +08:00
|
|
|
// We iterate through the words positions of the document id, retrieve the word and delete the positions.
|
|
|
|
// We create an iterator to be able to get the content and delete the key-value itself.
|
|
|
|
// It's faster to acquire a cursor to get and delete, as we avoid traversing the LMDB B-Tree two times but only once.
|
2020-10-26 18:01:00 +08:00
|
|
|
let mut iter = docid_word_positions.prefix_iter_mut(self.wtxn, &(docid, ""))?;
|
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let ((_docid, word), _positions) = result?;
|
|
|
|
// This boolean will indicate if we must remove this word from the words FST.
|
2020-10-29 21:32:32 +08:00
|
|
|
words.push((SmallString32::from(word), false));
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
}
|
2020-11-23 00:53:33 +08:00
|
|
|
// We acquire the current external documents ids map...
|
2022-12-20 17:37:50 +08:00
|
|
|
// Note that its soft-deleted document ids field will be equal to the `to_delete_docids`
|
2020-11-23 00:53:33 +08:00
|
|
|
let mut new_external_documents_ids = self.index.external_documents_ids(self.wtxn)?;
|
2022-12-20 17:37:50 +08:00
|
|
|
// We then remove the soft-deleted docids from it
|
|
|
|
new_external_documents_ids.delete_soft_deleted_documents_ids_from_fsts()?;
|
|
|
|
// and write it back to the main database.
|
2020-11-23 00:53:33 +08:00
|
|
|
let new_external_documents_ids = new_external_documents_ids.into_static();
|
2020-11-22 18:54:04 +08:00
|
|
|
self.index.put_external_documents_ids(self.wtxn, &new_external_documents_ids)?;
|
2020-10-26 18:01:00 +08:00
|
|
|
|
|
|
|
// Maybe we can improve the get performance of the words
|
|
|
|
// if we sort the words first, keeping the LMDB pages in cache.
|
|
|
|
words.sort_unstable();
|
|
|
|
|
|
|
|
// We iterate over the words and delete the documents ids
|
|
|
|
// from the word docids database.
|
|
|
|
for (word, must_remove) in &mut words {
|
2022-03-24 22:22:57 +08:00
|
|
|
remove_from_word_docids(
|
|
|
|
self.wtxn,
|
|
|
|
word_docids,
|
|
|
|
word.as_str(),
|
|
|
|
must_remove,
|
2022-06-13 23:59:34 +08:00
|
|
|
&self.to_delete_docids,
|
2022-03-24 22:22:57 +08:00
|
|
|
)?;
|
|
|
|
|
|
|
|
remove_from_word_docids(
|
|
|
|
self.wtxn,
|
|
|
|
exact_word_docids,
|
|
|
|
word.as_str(),
|
|
|
|
must_remove,
|
2022-06-13 23:59:34 +08:00
|
|
|
&self.to_delete_docids,
|
2022-03-24 22:22:57 +08:00
|
|
|
)?;
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
// We construct an FST set that contains the words to delete from the words FST.
|
2021-06-17 00:33:33 +08:00
|
|
|
let words_to_delete =
|
|
|
|
words.iter().filter_map(
|
|
|
|
|(word, must_remove)| {
|
|
|
|
if *must_remove {
|
2022-03-15 00:13:07 +08:00
|
|
|
Some(word.as_str())
|
2021-06-17 00:33:33 +08:00
|
|
|
} else {
|
|
|
|
None
|
|
|
|
}
|
|
|
|
},
|
|
|
|
);
|
2020-10-26 18:01:00 +08:00
|
|
|
let words_to_delete = fst::Set::from_iter(words_to_delete)?;
|
|
|
|
|
|
|
|
let new_words_fst = {
|
|
|
|
// We retrieve the current words FST from the database.
|
|
|
|
let words_fst = self.index.words_fst(self.wtxn)?;
|
|
|
|
let difference = words_fst.op().add(&words_to_delete).difference();
|
|
|
|
|
2020-11-22 18:54:04 +08:00
|
|
|
// We stream the new external ids that does no more contains the to-delete external ids.
|
2020-10-26 18:01:00 +08:00
|
|
|
let mut new_words_fst_builder = fst::SetBuilder::memory();
|
|
|
|
new_words_fst_builder.extend_stream(difference.into_stream())?;
|
|
|
|
|
|
|
|
// We create an words FST set from the above builder.
|
|
|
|
new_words_fst_builder.into_set()
|
|
|
|
};
|
|
|
|
|
|
|
|
// We write the new words FST into the main database.
|
|
|
|
self.index.put_words_fst(self.wtxn, &new_words_fst)?;
|
|
|
|
|
2022-03-25 17:49:34 +08:00
|
|
|
let prefixes_to_delete =
|
2022-06-13 23:59:34 +08:00
|
|
|
remove_from_word_prefix_docids(self.wtxn, word_prefix_docids, &self.to_delete_docids)?;
|
2021-02-17 18:22:25 +08:00
|
|
|
|
2022-03-25 17:49:34 +08:00
|
|
|
let exact_prefix_to_delete = remove_from_word_prefix_docids(
|
|
|
|
self.wtxn,
|
|
|
|
exact_word_prefix_docids,
|
2022-06-13 23:59:34 +08:00
|
|
|
&self.to_delete_docids,
|
2022-03-25 17:49:34 +08:00
|
|
|
)?;
|
|
|
|
|
|
|
|
let all_prefixes_to_delete = prefixes_to_delete.op().add(&exact_prefix_to_delete).union();
|
2021-02-17 18:22:25 +08:00
|
|
|
|
|
|
|
// We compute the new prefix FST and write it only if there is a change.
|
2022-03-25 17:49:34 +08:00
|
|
|
if !prefixes_to_delete.is_empty() || !exact_prefix_to_delete.is_empty() {
|
2021-02-17 18:22:25 +08:00
|
|
|
let new_words_prefixes_fst = {
|
|
|
|
// We retrieve the current words prefixes FST from the database.
|
|
|
|
let words_prefixes_fst = self.index.words_prefixes_fst(self.wtxn)?;
|
2022-03-25 17:49:34 +08:00
|
|
|
let difference =
|
|
|
|
words_prefixes_fst.op().add(all_prefixes_to_delete.into_stream()).difference();
|
2021-02-17 18:22:25 +08:00
|
|
|
|
|
|
|
// We stream the new external ids that does no more contains the to-delete external ids.
|
|
|
|
let mut new_words_prefixes_fst_builder = fst::SetBuilder::memory();
|
|
|
|
new_words_prefixes_fst_builder.extend_stream(difference.into_stream())?;
|
|
|
|
|
|
|
|
// We create an words FST set from the above builder.
|
|
|
|
new_words_prefixes_fst_builder.into_set()
|
|
|
|
};
|
|
|
|
|
|
|
|
// We write the new words prefixes FST into the main database.
|
|
|
|
self.index.put_words_prefixes_fst(self.wtxn, &new_words_prefixes_fst)?;
|
|
|
|
}
|
|
|
|
|
2022-09-14 21:33:13 +08:00
|
|
|
for db in [word_prefix_pair_proximity_docids, prefix_word_pair_proximity_docids] {
|
|
|
|
// We delete the documents ids from the word prefix pair proximity database docids
|
|
|
|
// and remove the empty pairs too.
|
|
|
|
let db = db.remap_key_type::<ByteSlice>();
|
|
|
|
let mut iter = db.iter_mut(self.wtxn)?;
|
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (key, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
|
|
|
docids -= &self.to_delete_docids;
|
|
|
|
if docids.is_empty() {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
|
|
|
} else if docids.len() != previous_len {
|
|
|
|
let key = key.to_owned();
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &docids)? };
|
|
|
|
}
|
2021-02-10 17:35:25 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2020-10-28 01:50:09 +08:00
|
|
|
// We delete the documents ids that are under the pairs of words,
|
|
|
|
// it is faster and use no memory to iterate over all the words pairs than
|
|
|
|
// to compute the cartesian product of every words of the deleted documents.
|
2021-06-17 00:33:33 +08:00
|
|
|
let mut iter =
|
|
|
|
word_pair_proximity_docids.remap_key_type::<ByteSlice>().iter_mut(self.wtxn)?;
|
2020-10-28 01:50:09 +08:00
|
|
|
while let Some(result) = iter.next() {
|
2020-11-19 18:17:53 +08:00
|
|
|
let (bytes, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2022-06-13 23:59:34 +08:00
|
|
|
docids -= &self.to_delete_docids;
|
2020-10-28 01:50:09 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2020-11-19 18:17:53 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let bytes = bytes.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&bytes, &docids)? };
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2020-11-11 23:04:04 +08:00
|
|
|
drop(iter);
|
|
|
|
|
2021-03-17 22:47:41 +08:00
|
|
|
// We delete the documents ids that are under the word level position docids.
|
2021-10-05 17:18:42 +08:00
|
|
|
let mut iter = word_position_docids.iter_mut(self.wtxn)?.remap_key_type::<ByteSlice>();
|
2021-03-17 22:47:41 +08:00
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (bytes, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2022-06-13 23:59:34 +08:00
|
|
|
docids -= &self.to_delete_docids;
|
2021-03-17 22:47:41 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-03-17 22:47:41 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let bytes = bytes.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&bytes, &docids)? };
|
2021-03-17 22:47:41 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
drop(iter);
|
|
|
|
|
2021-03-25 18:10:12 +08:00
|
|
|
// We delete the documents ids that are under the word prefix level position docids.
|
2021-06-17 00:33:33 +08:00
|
|
|
let mut iter =
|
2021-10-05 17:18:42 +08:00
|
|
|
word_prefix_position_docids.iter_mut(self.wtxn)?.remap_key_type::<ByteSlice>();
|
2021-03-25 18:10:12 +08:00
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (bytes, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2022-06-13 23:59:34 +08:00
|
|
|
docids -= &self.to_delete_docids;
|
2021-03-25 18:10:12 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-03-25 18:10:12 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let bytes = bytes.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&bytes, &docids)? };
|
2021-03-25 18:10:12 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
drop(iter);
|
|
|
|
|
2021-06-01 23:04:10 +08:00
|
|
|
// Remove the documents ids from the field id word count database.
|
2021-05-27 21:27:41 +08:00
|
|
|
let mut iter = field_id_word_count_docids.iter_mut(self.wtxn)?;
|
|
|
|
while let Some((key, mut docids)) = iter.next().transpose()? {
|
|
|
|
let previous_len = docids.len();
|
2022-06-13 23:59:34 +08:00
|
|
|
docids -= &self.to_delete_docids;
|
2021-05-27 21:27:41 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-05-27 21:27:41 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-07-21 16:35:35 +08:00
|
|
|
let key = key.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &docids)? };
|
2021-05-27 21:27:41 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
drop(iter);
|
|
|
|
|
2021-08-25 20:58:36 +08:00
|
|
|
if let Some(mut rtree) = self.index.geo_rtree(self.wtxn)? {
|
2021-08-26 23:49:50 +08:00
|
|
|
let mut geo_faceted_doc_ids = self.index.geo_faceted_documents_ids(self.wtxn)?;
|
|
|
|
|
2021-09-09 18:20:08 +08:00
|
|
|
let (points_to_remove, docids_to_remove): (Vec<_>, RoaringBitmap) = rtree
|
2021-08-25 20:58:36 +08:00
|
|
|
.iter()
|
2022-06-13 23:59:34 +08:00
|
|
|
.filter(|&point| self.to_delete_docids.contains(point.data.0))
|
2021-08-25 20:58:36 +08:00
|
|
|
.cloned()
|
2021-12-14 19:21:24 +08:00
|
|
|
.map(|point| (point, point.data.0))
|
2021-09-09 18:20:08 +08:00
|
|
|
.unzip();
|
2021-08-25 20:58:36 +08:00
|
|
|
points_to_remove.iter().for_each(|point| {
|
2022-10-25 03:34:13 +08:00
|
|
|
rtree.remove(point);
|
2021-08-25 20:58:36 +08:00
|
|
|
});
|
2021-09-09 18:20:08 +08:00
|
|
|
geo_faceted_doc_ids -= docids_to_remove;
|
2021-08-25 20:58:36 +08:00
|
|
|
|
|
|
|
self.index.put_geo_rtree(self.wtxn, &rtree)?;
|
2021-08-26 23:49:50 +08:00
|
|
|
self.index.put_geo_faceted_documents_ids(self.wtxn, &geo_faceted_doc_ids)?;
|
2021-08-25 20:58:36 +08:00
|
|
|
}
|
|
|
|
|
2022-09-01 14:34:26 +08:00
|
|
|
for facet_type in [FacetType::Number, FacetType::String] {
|
2022-09-21 21:53:39 +08:00
|
|
|
let mut affected_facet_values = HashMap::new();
|
2022-09-08 19:28:17 +08:00
|
|
|
for field_id in self.index.faceted_fields_ids(self.wtxn)? {
|
|
|
|
// Remove docids from the number faceted documents ids
|
|
|
|
let mut docids =
|
|
|
|
self.index.faceted_documents_ids(self.wtxn, field_id, facet_type)?;
|
|
|
|
docids -= &self.to_delete_docids;
|
|
|
|
self.index.put_faceted_documents_ids(self.wtxn, field_id, facet_type, &docids)?;
|
|
|
|
|
2022-09-21 21:53:39 +08:00
|
|
|
let facet_values = remove_docids_from_field_id_docid_facet_value(
|
2022-10-27 22:58:13 +08:00
|
|
|
self.index,
|
2022-09-08 19:28:17 +08:00
|
|
|
self.wtxn,
|
|
|
|
facet_type,
|
|
|
|
field_id,
|
|
|
|
&self.to_delete_docids,
|
|
|
|
)?;
|
2022-09-21 21:53:39 +08:00
|
|
|
if !facet_values.is_empty() {
|
|
|
|
affected_facet_values.insert(field_id, facet_values);
|
|
|
|
}
|
2022-09-08 19:28:17 +08:00
|
|
|
}
|
2022-09-21 21:53:39 +08:00
|
|
|
FacetsDelete::new(
|
|
|
|
self.index,
|
|
|
|
facet_type,
|
|
|
|
affected_facet_values,
|
|
|
|
&self.to_delete_docids,
|
|
|
|
)
|
|
|
|
.execute(self.wtxn)?;
|
2022-09-01 14:34:26 +08:00
|
|
|
}
|
|
|
|
|
2022-10-15 05:25:09 +08:00
|
|
|
// Remove the documents ids from the script language database.
|
|
|
|
let mut iter = script_language_docids.iter_mut(self.wtxn)?;
|
|
|
|
while let Some((key, mut docids)) = iter.next().transpose()? {
|
|
|
|
let previous_len = docids.len();
|
|
|
|
docids -= &self.to_delete_docids;
|
|
|
|
if docids.is_empty() {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
|
|
|
} else if docids.len() != previous_len {
|
|
|
|
let key = key.to_owned();
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &docids)? };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
drop(iter);
|
2022-08-31 14:10:45 +08:00
|
|
|
// We delete the documents ids that are under the facet field id values.
|
|
|
|
remove_docids_from_facet_id_exists_docids(
|
|
|
|
self.wtxn,
|
|
|
|
facet_id_exists_docids,
|
2022-06-13 23:59:34 +08:00
|
|
|
&self.to_delete_docids,
|
2021-04-21 21:43:44 +08:00
|
|
|
)?;
|
|
|
|
|
2023-03-08 23:14:00 +08:00
|
|
|
// We delete the documents ids that are under the facet field id values.
|
|
|
|
remove_docids_from_facet_id_exists_docids(
|
|
|
|
self.wtxn,
|
|
|
|
facet_id_is_null_docids,
|
|
|
|
&self.to_delete_docids,
|
|
|
|
)?;
|
|
|
|
|
2022-12-20 17:37:50 +08:00
|
|
|
self.index.put_soft_deleted_documents_ids(self.wtxn, &RoaringBitmap::new())?;
|
|
|
|
|
2022-12-12 19:42:55 +08:00
|
|
|
Ok(DetailedDocumentDeletionResult {
|
2022-06-13 23:59:34 +08:00
|
|
|
deleted_documents: self.to_delete_docids.len(),
|
2021-11-10 03:19:49 +08:00
|
|
|
remaining_documents: documents_ids.len(),
|
2022-12-13 17:15:22 +08:00
|
|
|
soft_deletion_used: false,
|
2021-11-10 03:19:49 +08:00
|
|
|
})
|
2020-10-26 18:01:00 +08:00
|
|
|
}
|
|
|
|
}
|
2021-02-13 21:04:23 +08:00
|
|
|
|
2022-03-25 17:49:34 +08:00
|
|
|
fn remove_from_word_prefix_docids(
|
|
|
|
txn: &mut heed::RwTxn,
|
|
|
|
db: &Database<Str, RoaringBitmapCodec>,
|
|
|
|
to_remove: &RoaringBitmap,
|
|
|
|
) -> Result<fst::Set<Vec<u8>>> {
|
|
|
|
let mut prefixes_to_delete = fst::SetBuilder::memory();
|
|
|
|
|
|
|
|
// We iterate over the word prefix docids database and remove the deleted documents ids
|
|
|
|
// from every docids lists. We register the empty prefixes in an fst Set for futur deletion.
|
|
|
|
let mut iter = db.iter_mut(txn)?;
|
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (prefix, mut docids) = result?;
|
|
|
|
let prefix = prefix.to_owned();
|
|
|
|
let previous_len = docids.len();
|
|
|
|
docids -= to_remove;
|
|
|
|
if docids.is_empty() {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
|
|
|
prefixes_to_delete.insert(prefix)?;
|
|
|
|
} else if docids.len() != previous_len {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&prefix, &docids)? };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(prefixes_to_delete.into_set())
|
|
|
|
}
|
|
|
|
|
2022-03-24 22:22:57 +08:00
|
|
|
fn remove_from_word_docids(
|
|
|
|
txn: &mut heed::RwTxn,
|
|
|
|
db: &heed::Database<Str, RoaringBitmapCodec>,
|
|
|
|
word: &str,
|
|
|
|
must_remove: &mut bool,
|
|
|
|
to_remove: &RoaringBitmap,
|
|
|
|
) -> Result<()> {
|
|
|
|
// We create an iterator to be able to get the content and delete the word docids.
|
|
|
|
// It's faster to acquire a cursor to get and delete or put, as we avoid traversing
|
|
|
|
// the LMDB B-Tree two times but only once.
|
2022-10-25 03:34:13 +08:00
|
|
|
let mut iter = db.prefix_iter_mut(txn, word)?;
|
2022-03-24 22:22:57 +08:00
|
|
|
if let Some((key, mut docids)) = iter.next().transpose()? {
|
|
|
|
if key == word {
|
|
|
|
let previous_len = docids.len();
|
|
|
|
docids -= to_remove;
|
|
|
|
if docids.is_empty() {
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
|
|
|
*must_remove = true;
|
|
|
|
} else if docids.len() != previous_len {
|
|
|
|
let key = key.to_owned();
|
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&key, &docids)? };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
2022-04-05 20:14:15 +08:00
|
|
|
|
2022-03-24 22:22:57 +08:00
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
|
2023-01-31 00:18:02 +08:00
|
|
|
fn remove_docids_from_field_id_docid_facet_value(
|
2023-01-31 18:11:49 +08:00
|
|
|
index: &Index,
|
|
|
|
wtxn: &mut heed::RwTxn,
|
2022-09-08 19:28:17 +08:00
|
|
|
facet_type: FacetType,
|
2021-04-21 21:43:44 +08:00
|
|
|
field_id: FieldId,
|
|
|
|
to_remove: &RoaringBitmap,
|
2022-09-21 21:53:39 +08:00
|
|
|
) -> heed::Result<HashSet<Vec<u8>>> {
|
2022-09-08 19:28:17 +08:00
|
|
|
let db = match facet_type {
|
|
|
|
FacetType::String => {
|
|
|
|
index.field_id_docid_facet_strings.remap_types::<ByteSlice, DecodeIgnore>()
|
|
|
|
}
|
|
|
|
FacetType::Number => {
|
|
|
|
index.field_id_docid_facet_f64s.remap_types::<ByteSlice, DecodeIgnore>()
|
|
|
|
}
|
|
|
|
};
|
2022-09-21 21:53:39 +08:00
|
|
|
let mut all_affected_facet_values = HashSet::default();
|
2021-07-06 17:31:24 +08:00
|
|
|
let mut iter = db
|
|
|
|
.prefix_iter_mut(wtxn, &field_id.to_be_bytes())?
|
2022-09-21 21:53:39 +08:00
|
|
|
.remap_key_type::<FieldDocIdFacetCodec<ByteSlice>>();
|
2021-04-21 21:43:44 +08:00
|
|
|
|
|
|
|
while let Some(result) = iter.next() {
|
2022-09-21 21:53:39 +08:00
|
|
|
let ((_, docid, facet_value), _) = result?;
|
2022-09-08 19:28:17 +08:00
|
|
|
if to_remove.contains(docid) {
|
2022-09-21 21:53:39 +08:00
|
|
|
if !all_affected_facet_values.contains(facet_value) {
|
|
|
|
all_affected_facet_values.insert(facet_value.to_owned());
|
|
|
|
}
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-04-21 21:43:44 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2022-09-21 21:53:39 +08:00
|
|
|
Ok(all_affected_facet_values)
|
2021-04-21 21:43:44 +08:00
|
|
|
}
|
|
|
|
|
2022-08-31 14:10:45 +08:00
|
|
|
fn remove_docids_from_facet_id_exists_docids<'a, C>(
|
2021-04-21 21:43:44 +08:00
|
|
|
wtxn: &'a mut heed::RwTxn,
|
|
|
|
db: &heed::Database<C, CboRoaringBitmapCodec>,
|
|
|
|
to_remove: &RoaringBitmap,
|
|
|
|
) -> heed::Result<()>
|
|
|
|
where
|
|
|
|
C: heed::BytesDecode<'a> + heed::BytesEncode<'a>,
|
|
|
|
{
|
|
|
|
let mut iter = db.remap_key_type::<ByteSlice>().iter_mut(wtxn)?;
|
|
|
|
while let Some(result) = iter.next() {
|
|
|
|
let (bytes, mut docids) = result?;
|
|
|
|
let previous_len = docids.len();
|
2021-06-30 20:12:56 +08:00
|
|
|
docids -= to_remove;
|
2021-04-21 21:43:44 +08:00
|
|
|
if docids.is_empty() {
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.del_current()? };
|
2021-04-21 21:43:44 +08:00
|
|
|
} else if docids.len() != previous_len {
|
2021-06-28 22:19:02 +08:00
|
|
|
let bytes = bytes.to_owned();
|
2021-06-29 00:26:20 +08:00
|
|
|
// safety: we don't keep references from inside the LMDB database.
|
|
|
|
unsafe { iter.put_current(&bytes, &docids)? };
|
2021-04-21 21:43:44 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
Ok(())
|
|
|
|
}
|
|
|
|
|
2021-02-13 21:04:23 +08:00
|
|
|
#[cfg(test)]
|
|
|
|
mod tests {
|
2021-08-20 23:32:02 +08:00
|
|
|
use big_s::S;
|
2022-08-02 21:13:06 +08:00
|
|
|
use heed::RwTxn;
|
2021-08-20 23:32:02 +08:00
|
|
|
use maplit::hashset;
|
2021-02-13 21:04:23 +08:00
|
|
|
|
|
|
|
use super::*;
|
2022-08-02 21:13:06 +08:00
|
|
|
use crate::index::tests::TempIndex;
|
2022-08-25 20:51:50 +08:00
|
|
|
use crate::{db_snap, Filter};
|
2021-02-13 21:04:23 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
fn delete_documents<'t>(
|
|
|
|
wtxn: &mut RwTxn<'t, '_>,
|
|
|
|
index: &'t Index,
|
|
|
|
external_ids: &[&str],
|
2022-12-19 16:47:29 +08:00
|
|
|
strategy: DeletionStrategy,
|
2022-06-13 22:39:33 +08:00
|
|
|
) -> Vec<u32> {
|
2022-10-10 21:28:03 +08:00
|
|
|
let external_document_ids = index.external_documents_ids(wtxn).unwrap();
|
2022-06-13 22:39:33 +08:00
|
|
|
let ids_to_delete: Vec<u32> = external_ids
|
|
|
|
.iter()
|
|
|
|
.map(|id| external_document_ids.get(id.as_bytes()).unwrap())
|
|
|
|
.collect();
|
|
|
|
|
|
|
|
// Delete some documents.
|
|
|
|
let mut builder = DeleteDocuments::new(wtxn, index).unwrap();
|
2022-12-19 16:47:29 +08:00
|
|
|
builder.strategy(strategy);
|
2023-01-18 01:01:26 +08:00
|
|
|
external_ids.iter().for_each(|id| {
|
|
|
|
builder.delete_external_id(id);
|
|
|
|
});
|
2022-06-13 22:39:33 +08:00
|
|
|
builder.execute().unwrap();
|
|
|
|
|
|
|
|
ids_to_delete
|
|
|
|
}
|
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
fn delete_documents_with_numbers_as_primary_key_(deletion_strategy: DeletionStrategy) {
|
2022-08-02 21:13:06 +08:00
|
|
|
let index = TempIndex::new();
|
2021-02-13 21:04:23 +08:00
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2022-08-02 21:13:06 +08:00
|
|
|
index
|
|
|
|
.add_documents_using_wtxn(
|
|
|
|
&mut wtxn,
|
|
|
|
documents!([
|
|
|
|
{ "id": 0, "name": "kevin", "object": { "key1": "value1", "key2": "value2" } },
|
|
|
|
{ "id": 1, "name": "kevina", "array": ["I", "am", "fine"] },
|
|
|
|
{ "id": 2, "name": "benoit", "array_of_object": [{ "wow": "amazing" }] }
|
|
|
|
]),
|
|
|
|
)
|
|
|
|
.unwrap();
|
2021-02-13 21:04:23 +08:00
|
|
|
|
|
|
|
// delete those documents, ids are synchronous therefore 0, 1, and 2.
|
2021-11-03 20:12:01 +08:00
|
|
|
let mut builder = DeleteDocuments::new(&mut wtxn, &index).unwrap();
|
2021-02-13 21:04:23 +08:00
|
|
|
builder.delete_document(0);
|
|
|
|
builder.delete_document(1);
|
|
|
|
builder.delete_document(2);
|
2022-12-19 16:47:29 +08:00
|
|
|
builder.strategy(deletion_strategy);
|
2021-02-13 21:04:23 +08:00
|
|
|
builder.execute().unwrap();
|
|
|
|
|
|
|
|
wtxn.commit().unwrap();
|
2021-04-01 15:07:16 +08:00
|
|
|
|
2022-09-22 20:01:13 +08:00
|
|
|
// All these snapshots should be empty since the database was cleared
|
2022-12-19 16:47:29 +08:00
|
|
|
db_snap!(index, documents_ids, deletion_strategy);
|
|
|
|
db_snap!(index, word_docids, deletion_strategy);
|
|
|
|
db_snap!(index, word_pair_proximity_docids, deletion_strategy);
|
|
|
|
db_snap!(index, facet_id_exists_docids, deletion_strategy);
|
|
|
|
db_snap!(index, soft_deleted_documents_ids, deletion_strategy);
|
2022-08-25 20:51:50 +08:00
|
|
|
|
2021-04-01 15:07:16 +08:00
|
|
|
let rtxn = index.read_txn().unwrap();
|
|
|
|
|
2021-06-17 21:16:20 +08:00
|
|
|
assert!(index.field_distribution(&rtxn).unwrap().is_empty());
|
2021-02-13 21:04:23 +08:00
|
|
|
}
|
2021-06-08 23:33:29 +08:00
|
|
|
|
|
|
|
#[test]
|
2022-09-22 20:01:13 +08:00
|
|
|
fn delete_documents_with_numbers_as_primary_key() {
|
2022-12-19 16:47:29 +08:00
|
|
|
delete_documents_with_numbers_as_primary_key_(DeletionStrategy::AlwaysHard);
|
|
|
|
delete_documents_with_numbers_as_primary_key_(DeletionStrategy::AlwaysSoft);
|
2022-09-22 20:01:13 +08:00
|
|
|
}
|
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
fn delete_documents_with_strange_primary_key_(strategy: DeletionStrategy) {
|
2022-08-02 21:13:06 +08:00
|
|
|
let index = TempIndex::new();
|
2021-06-08 23:33:29 +08:00
|
|
|
|
2022-08-25 20:51:50 +08:00
|
|
|
index
|
|
|
|
.update_settings(|settings| settings.set_searchable_fields(vec!["name".to_string()]))
|
|
|
|
.unwrap();
|
|
|
|
|
2021-06-08 23:33:29 +08:00
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2022-08-02 21:13:06 +08:00
|
|
|
index
|
|
|
|
.add_documents_using_wtxn(
|
|
|
|
&mut wtxn,
|
|
|
|
documents!([
|
|
|
|
{ "mysuperid": 0, "name": "kevin" },
|
|
|
|
{ "mysuperid": 1, "name": "kevina" },
|
|
|
|
{ "mysuperid": 2, "name": "benoit" }
|
|
|
|
]),
|
|
|
|
)
|
|
|
|
.unwrap();
|
2022-08-25 20:51:50 +08:00
|
|
|
wtxn.commit().unwrap();
|
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2021-06-08 23:33:29 +08:00
|
|
|
|
|
|
|
// Delete not all of the documents but some of them.
|
2021-11-03 20:12:01 +08:00
|
|
|
let mut builder = DeleteDocuments::new(&mut wtxn, &index).unwrap();
|
2021-06-08 23:33:29 +08:00
|
|
|
builder.delete_external_id("0");
|
|
|
|
builder.delete_external_id("1");
|
2022-12-19 16:47:29 +08:00
|
|
|
builder.strategy(strategy);
|
2021-06-08 23:33:29 +08:00
|
|
|
builder.execute().unwrap();
|
|
|
|
wtxn.commit().unwrap();
|
2022-08-25 20:51:50 +08:00
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
db_snap!(index, documents_ids, strategy);
|
|
|
|
db_snap!(index, word_docids, strategy);
|
|
|
|
db_snap!(index, word_pair_proximity_docids, strategy);
|
|
|
|
db_snap!(index, soft_deleted_documents_ids, strategy);
|
2021-06-08 23:33:29 +08:00
|
|
|
}
|
2021-08-20 23:32:02 +08:00
|
|
|
|
|
|
|
#[test]
|
2022-09-22 20:01:13 +08:00
|
|
|
fn delete_documents_with_strange_primary_key() {
|
2022-12-19 16:47:29 +08:00
|
|
|
delete_documents_with_strange_primary_key_(DeletionStrategy::AlwaysHard);
|
|
|
|
delete_documents_with_strange_primary_key_(DeletionStrategy::AlwaysSoft);
|
2022-09-22 20:01:13 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
fn filtered_placeholder_search_should_not_return_deleted_documents_(
|
2022-12-19 16:47:29 +08:00
|
|
|
deletion_strategy: DeletionStrategy,
|
2022-09-22 20:01:13 +08:00
|
|
|
) {
|
2022-08-02 21:13:06 +08:00
|
|
|
let index = TempIndex::new();
|
2021-08-20 23:32:02 +08:00
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2022-08-02 21:13:06 +08:00
|
|
|
|
|
|
|
index
|
|
|
|
.update_settings_using_wtxn(&mut wtxn, |settings| {
|
|
|
|
settings.set_primary_key(S("docid"));
|
2022-09-22 20:01:13 +08:00
|
|
|
settings.set_filterable_fields(hashset! { S("label"), S("label2") });
|
2022-08-02 21:13:06 +08:00
|
|
|
})
|
|
|
|
.unwrap();
|
|
|
|
|
|
|
|
index
|
|
|
|
.add_documents_using_wtxn(
|
|
|
|
&mut wtxn,
|
|
|
|
documents!([
|
2022-08-25 20:51:50 +08:00
|
|
|
{ "docid": "1_4", "label": ["sign"] },
|
|
|
|
{ "docid": "1_5", "label": ["letter"] },
|
|
|
|
{ "docid": "1_7", "label": ["abstract","cartoon","design","pattern"] },
|
|
|
|
{ "docid": "1_36", "label": ["drawing","painting","pattern"] },
|
|
|
|
{ "docid": "1_37", "label": ["art","drawing","outdoor"] },
|
|
|
|
{ "docid": "1_38", "label": ["aquarium","art","drawing"] },
|
|
|
|
{ "docid": "1_39", "label": ["abstract"] },
|
|
|
|
{ "docid": "1_40", "label": ["cartoon"] },
|
|
|
|
{ "docid": "1_41", "label": ["art","drawing"] },
|
|
|
|
{ "docid": "1_42", "label": ["art","pattern"] },
|
|
|
|
{ "docid": "1_43", "label": ["abstract","art","drawing","pattern"] },
|
|
|
|
{ "docid": "1_44", "label": ["drawing"] },
|
|
|
|
{ "docid": "1_45", "label": ["art"] },
|
|
|
|
{ "docid": "1_46", "label": ["abstract","colorfulness","pattern"] },
|
|
|
|
{ "docid": "1_47", "label": ["abstract","pattern"] },
|
|
|
|
{ "docid": "1_52", "label": ["abstract","cartoon"] },
|
|
|
|
{ "docid": "1_57", "label": ["abstract","drawing","pattern"] },
|
|
|
|
{ "docid": "1_58", "label": ["abstract","art","cartoon"] },
|
|
|
|
{ "docid": "1_68", "label": ["design"] },
|
|
|
|
{ "docid": "1_69", "label": ["geometry"] },
|
|
|
|
{ "docid": "1_70", "label2": ["geometry", 1.2] },
|
|
|
|
{ "docid": "1_71", "label2": ["design", 2.2] },
|
|
|
|
{ "docid": "1_72", "label2": ["geometry", 1.2] }
|
2022-08-02 21:13:06 +08:00
|
|
|
]),
|
|
|
|
)
|
|
|
|
.unwrap();
|
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
delete_documents(&mut wtxn, &index, &["1_4", "1_70", "1_72"], deletion_strategy);
|
2021-08-20 23:32:02 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
// Placeholder search with filter
|
2021-12-09 18:13:12 +08:00
|
|
|
let filter = Filter::from_str("label = sign").unwrap().unwrap();
|
2021-08-20 23:32:02 +08:00
|
|
|
let results = index.search(&wtxn).filter(filter).execute().unwrap();
|
|
|
|
assert!(results.documents_ids.is_empty());
|
|
|
|
|
|
|
|
wtxn.commit().unwrap();
|
2022-08-25 20:51:50 +08:00
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
db_snap!(index, soft_deleted_documents_ids, deletion_strategy);
|
|
|
|
db_snap!(index, word_docids, deletion_strategy);
|
|
|
|
db_snap!(index, facet_id_f64_docids, deletion_strategy);
|
|
|
|
db_snap!(index, word_pair_proximity_docids, deletion_strategy);
|
|
|
|
db_snap!(index, facet_id_exists_docids, deletion_strategy);
|
|
|
|
db_snap!(index, facet_id_string_docids, deletion_strategy);
|
2022-08-25 20:51:50 +08:00
|
|
|
}
|
2022-09-22 20:01:13 +08:00
|
|
|
|
2022-08-25 20:51:50 +08:00
|
|
|
#[test]
|
2022-09-22 20:01:13 +08:00
|
|
|
fn filtered_placeholder_search_should_not_return_deleted_documents() {
|
2022-12-19 16:47:29 +08:00
|
|
|
filtered_placeholder_search_should_not_return_deleted_documents_(
|
|
|
|
DeletionStrategy::AlwaysHard,
|
|
|
|
);
|
|
|
|
filtered_placeholder_search_should_not_return_deleted_documents_(
|
|
|
|
DeletionStrategy::AlwaysSoft,
|
|
|
|
);
|
2022-09-22 20:01:13 +08:00
|
|
|
}
|
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
fn placeholder_search_should_not_return_deleted_documents_(
|
|
|
|
deletion_strategy: DeletionStrategy,
|
|
|
|
) {
|
2022-08-25 20:51:50 +08:00
|
|
|
let index = TempIndex::new();
|
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
|
|
|
index
|
|
|
|
.update_settings_using_wtxn(&mut wtxn, |settings| {
|
|
|
|
settings.set_primary_key(S("docid"));
|
|
|
|
})
|
|
|
|
.unwrap();
|
|
|
|
|
|
|
|
index
|
|
|
|
.add_documents_using_wtxn(
|
|
|
|
&mut wtxn,
|
|
|
|
documents!([
|
|
|
|
{ "docid": "1_4", "label": ["sign"] },
|
|
|
|
{ "docid": "1_5", "label": ["letter"] },
|
|
|
|
{ "docid": "1_7", "label": ["abstract","cartoon","design","pattern"] },
|
|
|
|
{ "docid": "1_36", "label": ["drawing","painting","pattern"] },
|
|
|
|
{ "docid": "1_37", "label": ["art","drawing","outdoor"] },
|
|
|
|
{ "docid": "1_38", "label": ["aquarium","art","drawing"] },
|
|
|
|
{ "docid": "1_39", "label": ["abstract"] },
|
|
|
|
{ "docid": "1_40", "label": ["cartoon"] },
|
|
|
|
{ "docid": "1_41", "label": ["art","drawing"] },
|
|
|
|
{ "docid": "1_42", "label": ["art","pattern"] },
|
|
|
|
{ "docid": "1_43", "label": ["abstract","art","drawing","pattern"] },
|
|
|
|
{ "docid": "1_44", "label": ["drawing"] },
|
|
|
|
{ "docid": "1_45", "label": ["art"] },
|
|
|
|
{ "docid": "1_46", "label": ["abstract","colorfulness","pattern"] },
|
|
|
|
{ "docid": "1_47", "label": ["abstract","pattern"] },
|
|
|
|
{ "docid": "1_52", "label": ["abstract","cartoon"] },
|
|
|
|
{ "docid": "1_57", "label": ["abstract","drawing","pattern"] },
|
|
|
|
{ "docid": "1_58", "label": ["abstract","art","cartoon"] },
|
|
|
|
{ "docid": "1_68", "label": ["design"] },
|
|
|
|
{ "docid": "1_69", "label": ["geometry"] },
|
|
|
|
{ "docid": "1_70", "label2": ["geometry", 1.2] },
|
|
|
|
{ "docid": "1_71", "label2": ["design", 2.2] },
|
|
|
|
{ "docid": "1_72", "label2": ["geometry", 1.2] }
|
|
|
|
]),
|
|
|
|
)
|
|
|
|
.unwrap();
|
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
let deleted_internal_ids = delete_documents(&mut wtxn, &index, &["1_4"], deletion_strategy);
|
2022-06-13 22:39:33 +08:00
|
|
|
|
|
|
|
// Placeholder search
|
|
|
|
let results = index.search(&wtxn).execute().unwrap();
|
|
|
|
assert!(!results.documents_ids.is_empty());
|
|
|
|
for id in results.documents_ids.iter() {
|
|
|
|
assert!(
|
2022-10-10 21:28:03 +08:00
|
|
|
!deleted_internal_ids.contains(id),
|
2022-06-13 22:39:33 +08:00
|
|
|
"The document {} was supposed to be deleted",
|
|
|
|
id
|
|
|
|
);
|
|
|
|
}
|
|
|
|
|
|
|
|
wtxn.commit().unwrap();
|
|
|
|
}
|
|
|
|
|
|
|
|
#[test]
|
2022-09-22 20:01:13 +08:00
|
|
|
fn placeholder_search_should_not_return_deleted_documents() {
|
2022-12-19 16:47:29 +08:00
|
|
|
placeholder_search_should_not_return_deleted_documents_(DeletionStrategy::AlwaysHard);
|
|
|
|
placeholder_search_should_not_return_deleted_documents_(DeletionStrategy::AlwaysSoft);
|
2022-09-22 20:01:13 +08:00
|
|
|
}
|
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
fn search_should_not_return_deleted_documents_(deletion_strategy: DeletionStrategy) {
|
2022-08-02 21:13:06 +08:00
|
|
|
let index = TempIndex::new();
|
2022-06-13 22:39:33 +08:00
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2022-08-02 21:13:06 +08:00
|
|
|
index
|
|
|
|
.update_settings_using_wtxn(&mut wtxn, |settings| {
|
|
|
|
settings.set_primary_key(S("docid"));
|
|
|
|
})
|
|
|
|
.unwrap();
|
|
|
|
|
|
|
|
index
|
|
|
|
.add_documents_using_wtxn(
|
|
|
|
&mut wtxn,
|
|
|
|
documents!([
|
2022-09-22 20:01:13 +08:00
|
|
|
{ "docid": "1_4", "label": ["sign"] },
|
|
|
|
{ "docid": "1_5", "label": ["letter"] },
|
|
|
|
{ "docid": "1_7", "label": ["abstract","cartoon","design","pattern"] },
|
|
|
|
{ "docid": "1_36", "label": ["drawing","painting","pattern"] },
|
|
|
|
{ "docid": "1_37", "label": ["art","drawing","outdoor"] },
|
|
|
|
{ "docid": "1_38", "label": ["aquarium","art","drawing"] },
|
|
|
|
{ "docid": "1_39", "label": ["abstract"] },
|
|
|
|
{ "docid": "1_40", "label": ["cartoon"] },
|
|
|
|
{ "docid": "1_41", "label": ["art","drawing"] },
|
|
|
|
{ "docid": "1_42", "label": ["art","pattern"] },
|
|
|
|
{ "docid": "1_43", "label": ["abstract","art","drawing","pattern"] },
|
|
|
|
{ "docid": "1_44", "label": ["drawing"] },
|
|
|
|
{ "docid": "1_45", "label": ["art"] },
|
|
|
|
{ "docid": "1_46", "label": ["abstract","colorfulness","pattern"] },
|
|
|
|
{ "docid": "1_47", "label": ["abstract","pattern"] },
|
|
|
|
{ "docid": "1_52", "label": ["abstract","cartoon"] },
|
|
|
|
{ "docid": "1_57", "label": ["abstract","drawing","pattern"] },
|
|
|
|
{ "docid": "1_58", "label": ["abstract","art","cartoon"] },
|
|
|
|
{ "docid": "1_68", "label": ["design"] },
|
|
|
|
{ "docid": "1_69", "label": ["geometry"] },
|
|
|
|
{ "docid": "1_70", "label2": ["geometry", 1.2] },
|
|
|
|
{ "docid": "1_71", "label2": ["design", 2.2] },
|
|
|
|
{ "docid": "1_72", "label2": ["geometry", 1.2] }
|
2022-08-02 21:13:06 +08:00
|
|
|
]),
|
|
|
|
)
|
|
|
|
.unwrap();
|
|
|
|
|
2022-09-22 20:01:13 +08:00
|
|
|
let deleted_internal_ids =
|
2022-12-19 16:47:29 +08:00
|
|
|
delete_documents(&mut wtxn, &index, &["1_7", "1_52"], deletion_strategy);
|
2022-06-13 22:39:33 +08:00
|
|
|
|
|
|
|
// search for abstract
|
|
|
|
let results = index.search(&wtxn).query("abstract").execute().unwrap();
|
|
|
|
assert!(!results.documents_ids.is_empty());
|
|
|
|
for id in results.documents_ids.iter() {
|
|
|
|
assert!(
|
2022-10-10 21:28:03 +08:00
|
|
|
!deleted_internal_ids.contains(id),
|
2022-06-13 22:39:33 +08:00
|
|
|
"The document {} was supposed to be deleted",
|
|
|
|
id
|
|
|
|
);
|
|
|
|
}
|
|
|
|
|
|
|
|
wtxn.commit().unwrap();
|
2022-08-25 20:51:50 +08:00
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
db_snap!(index, soft_deleted_documents_ids, deletion_strategy);
|
2022-06-13 22:39:33 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
#[test]
|
2022-09-22 20:01:13 +08:00
|
|
|
fn search_should_not_return_deleted_documents() {
|
2022-12-19 16:47:29 +08:00
|
|
|
search_should_not_return_deleted_documents_(DeletionStrategy::AlwaysHard);
|
|
|
|
search_should_not_return_deleted_documents_(DeletionStrategy::AlwaysSoft);
|
2022-09-22 20:01:13 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
fn geo_filtered_placeholder_search_should_not_return_deleted_documents_(
|
2022-12-19 16:47:29 +08:00
|
|
|
deletion_strategy: DeletionStrategy,
|
2022-09-22 20:01:13 +08:00
|
|
|
) {
|
2022-08-02 21:13:06 +08:00
|
|
|
let index = TempIndex::new();
|
2021-08-26 19:27:32 +08:00
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2022-08-02 21:13:06 +08:00
|
|
|
index
|
|
|
|
.update_settings_using_wtxn(&mut wtxn, |settings| {
|
|
|
|
settings.set_primary_key(S("id"));
|
|
|
|
settings.set_filterable_fields(hashset!(S("_geo")));
|
|
|
|
settings.set_sortable_fields(hashset!(S("_geo")));
|
|
|
|
})
|
|
|
|
.unwrap();
|
|
|
|
|
|
|
|
index.add_documents_using_wtxn(&mut wtxn, documents!([
|
2022-06-13 23:59:34 +08:00
|
|
|
{ "id": "1", "city": "Lille", "_geo": { "lat": 50.6299, "lng": 3.0569 } },
|
|
|
|
{ "id": "2", "city": "Mons-en-Barœul", "_geo": { "lat": 50.6415, "lng": 3.1106 } },
|
|
|
|
{ "id": "3", "city": "Hellemmes", "_geo": { "lat": 50.6312, "lng": 3.1106 } },
|
|
|
|
{ "id": "4", "city": "Villeneuve-d'Ascq", "_geo": { "lat": 50.6224, "lng": 3.1476 } },
|
|
|
|
{ "id": "5", "city": "Hem", "_geo": { "lat": 50.6552, "lng": 3.1897 } },
|
|
|
|
{ "id": "6", "city": "Roubaix", "_geo": { "lat": 50.6924, "lng": 3.1763 } },
|
|
|
|
{ "id": "7", "city": "Tourcoing", "_geo": { "lat": 50.7263, "lng": 3.1541 } },
|
|
|
|
{ "id": "8", "city": "Mouscron", "_geo": { "lat": 50.7453, "lng": 3.2206 } },
|
|
|
|
{ "id": "9", "city": "Tournai", "_geo": { "lat": 50.6053, "lng": 3.3758 } },
|
|
|
|
{ "id": "10", "city": "Ghent", "_geo": { "lat": 51.0537, "lng": 3.6957 } },
|
|
|
|
{ "id": "11", "city": "Brussels", "_geo": { "lat": 50.8466, "lng": 4.3370 } },
|
|
|
|
{ "id": "12", "city": "Charleroi", "_geo": { "lat": 50.4095, "lng": 4.4347 } },
|
|
|
|
{ "id": "13", "city": "Mons", "_geo": { "lat": 50.4502, "lng": 3.9623 } },
|
|
|
|
{ "id": "14", "city": "Valenciennes", "_geo": { "lat": 50.3518, "lng": 3.5326 } },
|
|
|
|
{ "id": "15", "city": "Arras", "_geo": { "lat": 50.2844, "lng": 2.7637 } },
|
|
|
|
{ "id": "16", "city": "Cambrai", "_geo": { "lat": 50.1793, "lng": 3.2189 } },
|
|
|
|
{ "id": "17", "city": "Bapaume", "_geo": { "lat": 50.1112, "lng": 2.8547 } },
|
|
|
|
{ "id": "18", "city": "Amiens", "_geo": { "lat": 49.9314, "lng": 2.2710 } },
|
|
|
|
{ "id": "19", "city": "Compiègne", "_geo": { "lat": 49.4449, "lng": 2.7913 } },
|
|
|
|
{ "id": "20", "city": "Paris", "_geo": { "lat": 48.9021, "lng": 2.3708 } }
|
2022-08-02 21:13:06 +08:00
|
|
|
])).unwrap();
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2022-08-02 21:13:06 +08:00
|
|
|
let external_ids_to_delete = ["5", "6", "7", "12", "17", "19"];
|
2022-09-22 20:01:13 +08:00
|
|
|
let deleted_internal_ids =
|
2022-12-19 16:47:29 +08:00
|
|
|
delete_documents(&mut wtxn, &index, &external_ids_to_delete, deletion_strategy);
|
2021-12-08 21:12:07 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
// Placeholder search with geo filter
|
|
|
|
let filter = Filter::from_str("_geoRadius(50.6924, 3.1763, 20000)").unwrap().unwrap();
|
|
|
|
let results = index.search(&wtxn).filter(filter).execute().unwrap();
|
|
|
|
assert!(!results.documents_ids.is_empty());
|
|
|
|
for id in results.documents_ids.iter() {
|
|
|
|
assert!(
|
2022-10-10 21:28:03 +08:00
|
|
|
!deleted_internal_ids.contains(id),
|
2022-06-13 22:39:33 +08:00
|
|
|
"The document {} was supposed to be deleted",
|
|
|
|
id
|
|
|
|
);
|
|
|
|
}
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
wtxn.commit().unwrap();
|
2022-08-25 20:51:50 +08:00
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
db_snap!(index, soft_deleted_documents_ids, deletion_strategy);
|
|
|
|
db_snap!(index, facet_id_f64_docids, deletion_strategy);
|
|
|
|
db_snap!(index, facet_id_string_docids, deletion_strategy);
|
2022-06-13 22:39:33 +08:00
|
|
|
}
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
#[test]
|
2022-09-22 20:01:13 +08:00
|
|
|
fn geo_filtered_placeholder_search_should_not_return_deleted_documents() {
|
2022-12-19 16:47:29 +08:00
|
|
|
geo_filtered_placeholder_search_should_not_return_deleted_documents_(
|
|
|
|
DeletionStrategy::AlwaysHard,
|
|
|
|
);
|
|
|
|
geo_filtered_placeholder_search_should_not_return_deleted_documents_(
|
|
|
|
DeletionStrategy::AlwaysSoft,
|
|
|
|
);
|
2022-09-22 20:01:13 +08:00
|
|
|
}
|
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
fn get_documents_should_not_return_deleted_documents_(deletion_strategy: DeletionStrategy) {
|
2022-08-02 21:13:06 +08:00
|
|
|
let index = TempIndex::new();
|
2022-06-13 22:39:33 +08:00
|
|
|
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2022-08-02 21:13:06 +08:00
|
|
|
index
|
|
|
|
.update_settings_using_wtxn(&mut wtxn, |settings| {
|
|
|
|
settings.set_primary_key(S("docid"));
|
|
|
|
})
|
|
|
|
.unwrap();
|
|
|
|
|
|
|
|
index
|
|
|
|
.add_documents_using_wtxn(
|
|
|
|
&mut wtxn,
|
|
|
|
documents!([
|
2022-09-22 20:01:13 +08:00
|
|
|
{ "docid": "1_4", "label": ["sign"] },
|
|
|
|
{ "docid": "1_5", "label": ["letter"] },
|
|
|
|
{ "docid": "1_7", "label": ["abstract","cartoon","design","pattern"] },
|
|
|
|
{ "docid": "1_36", "label": ["drawing","painting","pattern"] },
|
|
|
|
{ "docid": "1_37", "label": ["art","drawing","outdoor"] },
|
|
|
|
{ "docid": "1_38", "label": ["aquarium","art","drawing"] },
|
|
|
|
{ "docid": "1_39", "label": ["abstract"] },
|
|
|
|
{ "docid": "1_40", "label": ["cartoon"] },
|
|
|
|
{ "docid": "1_41", "label": ["art","drawing"] },
|
|
|
|
{ "docid": "1_42", "label": ["art","pattern"] },
|
|
|
|
{ "docid": "1_43", "label": ["abstract","art","drawing","pattern"] },
|
|
|
|
{ "docid": "1_44", "label": ["drawing"] },
|
|
|
|
{ "docid": "1_45", "label": ["art"] },
|
|
|
|
{ "docid": "1_46", "label": ["abstract","colorfulness","pattern"] },
|
|
|
|
{ "docid": "1_47", "label": ["abstract","pattern"] },
|
|
|
|
{ "docid": "1_52", "label": ["abstract","cartoon"] },
|
|
|
|
{ "docid": "1_57", "label": ["abstract","drawing","pattern"] },
|
|
|
|
{ "docid": "1_58", "label": ["abstract","art","cartoon"] },
|
|
|
|
{ "docid": "1_68", "label": ["design"] },
|
|
|
|
{ "docid": "1_69", "label": ["geometry"] },
|
|
|
|
{ "docid": "1_70", "label2": ["geometry", 1.2] },
|
|
|
|
{ "docid": "1_71", "label2": ["design", 2.2] },
|
|
|
|
{ "docid": "1_72", "label2": ["geometry", 1.2] }
|
2022-08-02 21:13:06 +08:00
|
|
|
]),
|
|
|
|
)
|
|
|
|
.unwrap();
|
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
let deleted_external_ids = ["1_7", "1_52"];
|
2022-09-22 20:01:13 +08:00
|
|
|
let deleted_internal_ids =
|
2022-12-19 16:47:29 +08:00
|
|
|
delete_documents(&mut wtxn, &index, &deleted_external_ids, deletion_strategy);
|
2022-06-13 22:39:33 +08:00
|
|
|
|
|
|
|
// list all documents
|
|
|
|
let results = index.all_documents(&wtxn).unwrap();
|
|
|
|
for result in results {
|
|
|
|
let (id, _) = result.unwrap();
|
|
|
|
assert!(
|
|
|
|
!deleted_internal_ids.contains(&id),
|
|
|
|
"The document {} was supposed to be deleted",
|
|
|
|
id
|
|
|
|
);
|
|
|
|
}
|
|
|
|
|
|
|
|
// list internal document ids
|
|
|
|
let results = index.documents_ids(&wtxn).unwrap();
|
|
|
|
for id in results {
|
|
|
|
assert!(
|
|
|
|
!deleted_internal_ids.contains(&id),
|
|
|
|
"The document {} was supposed to be deleted",
|
|
|
|
id
|
|
|
|
);
|
|
|
|
}
|
2022-12-20 17:37:50 +08:00
|
|
|
wtxn.commit().unwrap();
|
|
|
|
|
|
|
|
let rtxn = index.read_txn().unwrap();
|
2022-06-13 22:39:33 +08:00
|
|
|
|
|
|
|
// get internal docids from deleted external document ids
|
2022-12-20 17:37:50 +08:00
|
|
|
let results = index.external_documents_ids(&rtxn).unwrap();
|
2022-06-13 22:39:33 +08:00
|
|
|
for id in deleted_external_ids {
|
|
|
|
assert!(results.get(id).is_none(), "The document {} was supposed to be deleted", id);
|
|
|
|
}
|
2022-12-20 17:37:50 +08:00
|
|
|
drop(rtxn);
|
2022-08-25 20:51:50 +08:00
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
db_snap!(index, soft_deleted_documents_ids, deletion_strategy);
|
2022-06-13 22:39:33 +08:00
|
|
|
}
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
#[test]
|
2022-09-22 20:01:13 +08:00
|
|
|
fn get_documents_should_not_return_deleted_documents() {
|
2022-12-19 16:47:29 +08:00
|
|
|
get_documents_should_not_return_deleted_documents_(DeletionStrategy::AlwaysHard);
|
|
|
|
get_documents_should_not_return_deleted_documents_(DeletionStrategy::AlwaysSoft);
|
2022-09-22 20:01:13 +08:00
|
|
|
}
|
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
fn stats_should_not_return_deleted_documents_(deletion_strategy: DeletionStrategy) {
|
2022-08-02 21:13:06 +08:00
|
|
|
let index = TempIndex::new();
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
2021-08-26 23:49:50 +08:00
|
|
|
|
2022-08-02 21:13:06 +08:00
|
|
|
index
|
|
|
|
.update_settings_using_wtxn(&mut wtxn, |settings| {
|
|
|
|
settings.set_primary_key(S("docid"));
|
|
|
|
})
|
|
|
|
.unwrap();
|
|
|
|
|
|
|
|
index.add_documents_using_wtxn(&mut wtxn, documents!([
|
2022-09-22 20:01:13 +08:00
|
|
|
{ "docid": "1_4", "label": ["sign"]},
|
|
|
|
{ "docid": "1_5", "label": ["letter"]},
|
|
|
|
{ "docid": "1_7", "label": ["abstract","cartoon","design","pattern"], "title": "Mickey Mouse"},
|
|
|
|
{ "docid": "1_36", "label": ["drawing","painting","pattern"]},
|
|
|
|
{ "docid": "1_37", "label": ["art","drawing","outdoor"]},
|
|
|
|
{ "docid": "1_38", "label": ["aquarium","art","drawing"], "title": "Nemo"},
|
|
|
|
{ "docid": "1_39", "label": ["abstract"]},
|
|
|
|
{ "docid": "1_40", "label": ["cartoon"]},
|
|
|
|
{ "docid": "1_41", "label": ["art","drawing"]},
|
|
|
|
{ "docid": "1_42", "label": ["art","pattern"]},
|
|
|
|
{ "docid": "1_43", "label": ["abstract","art","drawing","pattern"], "number": 32i32},
|
|
|
|
{ "docid": "1_44", "label": ["drawing"], "number": 44i32},
|
|
|
|
{ "docid": "1_45", "label": ["art"]},
|
|
|
|
{ "docid": "1_46", "label": ["abstract","colorfulness","pattern"]},
|
|
|
|
{ "docid": "1_47", "label": ["abstract","pattern"]},
|
|
|
|
{ "docid": "1_52", "label": ["abstract","cartoon"]},
|
|
|
|
{ "docid": "1_57", "label": ["abstract","drawing","pattern"]},
|
|
|
|
{ "docid": "1_58", "label": ["abstract","art","cartoon"]},
|
|
|
|
{ "docid": "1_68", "label": ["design"]},
|
|
|
|
{ "docid": "1_69", "label": ["geometry"]}
|
2022-08-02 21:13:06 +08:00
|
|
|
])).unwrap();
|
2021-08-26 23:49:50 +08:00
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
delete_documents(&mut wtxn, &index, &["1_7", "1_52"], deletion_strategy);
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
// count internal documents
|
|
|
|
let results = index.number_of_documents(&wtxn).unwrap();
|
|
|
|
assert_eq!(18, results);
|
|
|
|
|
|
|
|
// count field distribution
|
|
|
|
let results = index.field_distribution(&wtxn).unwrap();
|
|
|
|
assert_eq!(Some(&18), results.get("label"));
|
|
|
|
assert_eq!(Some(&1), results.get("title"));
|
|
|
|
assert_eq!(Some(&2), results.get("number"));
|
2021-08-26 19:27:32 +08:00
|
|
|
|
2022-06-13 22:39:33 +08:00
|
|
|
wtxn.commit().unwrap();
|
2022-08-25 20:51:50 +08:00
|
|
|
|
2022-12-19 16:47:29 +08:00
|
|
|
db_snap!(index, soft_deleted_documents_ids, deletion_strategy);
|
2022-09-22 20:01:13 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
fn stats_should_not_return_deleted_documents() {
|
2022-12-19 16:47:29 +08:00
|
|
|
stats_should_not_return_deleted_documents_(DeletionStrategy::AlwaysHard);
|
|
|
|
stats_should_not_return_deleted_documents_(DeletionStrategy::AlwaysSoft);
|
2021-08-26 19:27:32 +08:00
|
|
|
}
|
2022-10-15 05:25:09 +08:00
|
|
|
|
2023-02-01 22:34:01 +08:00
|
|
|
fn stored_detected_script_and_language_should_not_return_deleted_documents_(
|
|
|
|
deletion_strategy: DeletionStrategy,
|
|
|
|
) {
|
2022-10-15 05:25:09 +08:00
|
|
|
use charabia::{Language, Script};
|
|
|
|
let index = TempIndex::new();
|
|
|
|
let mut wtxn = index.write_txn().unwrap();
|
|
|
|
index
|
|
|
|
.add_documents_using_wtxn(
|
|
|
|
&mut wtxn,
|
|
|
|
documents!([
|
|
|
|
{ "id": "0", "title": "The quick (\"brown\") fox can't jump 32.3 feet, right? Brr, it's 29.3°F!" },
|
|
|
|
{ "id": "1", "title": "人人生而自由﹐在尊嚴和權利上一律平等。他們賦有理性和良心﹐並應以兄弟關係的精神互相對待。" },
|
|
|
|
{ "id": "2", "title": "הַשּׁוּעָל הַמָּהִיר (״הַחוּם״) לֹא יָכוֹל לִקְפֹּץ 9.94 מֶטְרִים, נָכוֹן? ברר, 1.5°C- בַּחוּץ!" },
|
|
|
|
{ "id": "3", "title": "関西国際空港限定トートバッグ すもももももももものうち" },
|
|
|
|
{ "id": "4", "title": "ภาษาไทยง่ายนิดเดียว" },
|
|
|
|
{ "id": "5", "title": "The quick 在尊嚴和權利上一律平等。" },
|
|
|
|
]))
|
|
|
|
.unwrap();
|
|
|
|
|
2023-02-01 22:34:01 +08:00
|
|
|
let key_cmn = (Script::Cj, Language::Cmn);
|
|
|
|
let cj_cmn_docs =
|
|
|
|
index.script_language_documents_ids(&wtxn, &key_cmn).unwrap().unwrap_or_default();
|
|
|
|
let mut expected_cj_cmn_docids = RoaringBitmap::new();
|
|
|
|
expected_cj_cmn_docids.push(1);
|
|
|
|
expected_cj_cmn_docids.push(5);
|
|
|
|
assert_eq!(cj_cmn_docs, expected_cj_cmn_docids);
|
|
|
|
|
|
|
|
delete_documents(&mut wtxn, &index, &["1"], deletion_strategy);
|
2022-10-15 05:25:09 +08:00
|
|
|
wtxn.commit().unwrap();
|
|
|
|
|
|
|
|
let rtxn = index.read_txn().unwrap();
|
2023-02-01 22:34:01 +08:00
|
|
|
let cj_cmn_docs =
|
|
|
|
index.script_language_documents_ids(&rtxn, &key_cmn).unwrap().unwrap_or_default();
|
2022-10-15 05:25:09 +08:00
|
|
|
let mut expected_cj_cmn_docids = RoaringBitmap::new();
|
|
|
|
expected_cj_cmn_docids.push(5);
|
|
|
|
assert_eq!(cj_cmn_docs, expected_cj_cmn_docids);
|
|
|
|
}
|
2023-02-01 22:34:01 +08:00
|
|
|
|
|
|
|
#[test]
|
|
|
|
fn stored_detected_script_and_language_should_not_return_deleted_documents() {
|
|
|
|
stored_detected_script_and_language_should_not_return_deleted_documents_(
|
|
|
|
DeletionStrategy::AlwaysHard,
|
|
|
|
);
|
|
|
|
stored_detected_script_and_language_should_not_return_deleted_documents_(
|
|
|
|
DeletionStrategy::AlwaysSoft,
|
|
|
|
);
|
|
|
|
}
|
2021-02-13 21:04:23 +08:00
|
|
|
}
|