Compare commits

14 Commits

Author SHA1 Message Date
rustmailer
178b25d27d fix:After deleting an email, its attachment remains visible/active in the application #245 2026-05-20 17:40:12 +08:00
rustmailer
d160ca75f5 fix: migration link doesn't exist #244 2026-05-20 17:19:31 +08:00
rustmailer
04136a4ae2 bump to v1.1.2 2026-05-19 23:58:52 +08:00
rustmailer
1d6f5d9a22 Merge pull request #243 from tremor021/smallfix
Fix small typo in store.rs
2026-05-19 23:56:11 +08:00
rustmailer
105a6d9b15 Update dedup.rs 2026-05-19 23:50:46 +08:00
rustmailer
df440c8441 Update README.md 2026-05-19 23:33:44 +08:00
Slaviša Arežina
4116a59b79 Merge branch 'main' into smallfix 2026-05-19 17:28:15 +02:00
rustmailer
609eee1b84 show storage and index usage to everyone 2026-05-19 23:25:36 +08:00
tremor021
79b9f07888 fix small typo in store.rs 2026-05-19 17:18:10 +02:00
rustmailer
ba28369202 update 2026-05-19 13:58:20 +08:00
rustmailer
ff64b66f79 fix: Migration to v1.0 panics with index out of bounds: the len is 0 but the index is 0 #234 2026-05-19 12:12:37 +08:00
rustmailer
a4f8e674c3 Update Cargo.lock 2026-05-18 20:46:31 +08:00
rustmailer
dde6b990da bump to v1.1.0 2026-05-18 20:46:19 +08:00
rustmailer
6b1f843bd5 Merge pull request #237 from rustmailer/fix/cli-mbox-memory
fix: bichon-cli OOMs on import #233
2026-05-18 20:27:03 +08:00
10 changed files with 97 additions and 64 deletions

10
Cargo.lock generated
View File

@@ -293,7 +293,7 @@ checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6"
[[package]] [[package]]
name = "bichon-admin" name = "bichon-admin"
version = "1.0.2" version = "1.1.2"
dependencies = [ dependencies = [
"bichon-core", "bichon-core",
"console", "console",
@@ -312,7 +312,7 @@ dependencies = [
[[package]] [[package]]
name = "bichon-cli" name = "bichon-cli"
version = "1.0.2" version = "1.1.2"
dependencies = [ dependencies = [
"base64 0.22.1", "base64 0.22.1",
"bichon-core", "bichon-core",
@@ -338,7 +338,7 @@ dependencies = [
[[package]] [[package]]
name = "bichon-core" name = "bichon-core"
version = "1.0.2" version = "1.1.2"
dependencies = [ dependencies = [
"async-imap", "async-imap",
"base64 0.22.1", "base64 0.22.1",
@@ -396,7 +396,7 @@ dependencies = [
[[package]] [[package]]
name = "bichon-server" name = "bichon-server"
version = "1.0.2" version = "1.1.2"
dependencies = [ dependencies = [
"bichon-core", "bichon-core",
"bichon-smtp", "bichon-smtp",
@@ -421,7 +421,7 @@ dependencies = [
[[package]] [[package]]
name = "bichon-smtp" name = "bichon-smtp"
version = "1.0.2" version = "1.1.2"
dependencies = [ dependencies = [
"base64 0.22.1", "base64 0.22.1",
"bichon-core", "bichon-core",

View File

@@ -11,7 +11,7 @@ members = [
resolver = "2" resolver = "2"
[workspace.package] [workspace.package]
version = "1.0.2" version = "1.1.2"
edition = "2021" edition = "2021"
[workspace.dependencies] [workspace.dependencies]

View File

@@ -271,6 +271,10 @@ All settings accept both CLI flags (`--bichon-http-port`) and environment variab
> [!TIP] > [!TIP]
> Place `BICHON_INDEX_DIR` on fast SSD storage for responsive search, and `BICHON_DATA_DIR` on high-capacity HDD for cost-effective blob storage. > Place `BICHON_INDEX_DIR` on fast SSD storage for responsive search, and `BICHON_DATA_DIR` on high-capacity HDD for cost-effective blob storage.
> [!IMPORTANT]
> Bichon does NOT support writing data directly to a network file system (NFS, CIFS/SMB, etc.). All directories — `BICHON_ROOT_DIR`, `BICHON_DATA_DIR`, and `BICHON_INDEX_DIR` — must reside on a **local file system**; otherwise, data corruption may occur.
### Performance Tuning ### Performance Tuning
| Variable | Default | Description | | Variable | Default | Description |

View File

@@ -83,16 +83,11 @@ impl DashboardStats {
stat.email_count = ENVELOPE_MANAGER.total_emails(&authorized_ids)?; stat.email_count = ENVELOPE_MANAGER.total_emails(&authorized_ids)?;
stat.attachment_count = ATTACHMENT_MANAGER.total_attachments(&authorized_ids)?; stat.attachment_count = ATTACHMENT_MANAGER.total_attachments(&authorized_ids)?;
if has_all_accounts {
stat.storage_usage_bytes = get_total_size(&DATA_DIR_MANAGER.storage_dir) stat.storage_usage_bytes = get_total_size(&DATA_DIR_MANAGER.storage_dir)
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?; .map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
stat.index_usage_bytes = get_total_size(&&DATA_DIR_MANAGER.envelope_dir) stat.index_usage_bytes = get_total_size(&&DATA_DIR_MANAGER.envelope_dir)
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?; .map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
} else {
stat.storage_usage_bytes = 0;
stat.index_usage_bytes = 0;
}
stat.system_version = bichon_version!().to_string(); stat.system_version = bichon_version!().to_string();

View File

@@ -26,6 +26,6 @@ pub async fn delete_messages_impl(request: HashMap<u64, Vec<String>>) -> BichonR
.delete_envelopes_multi_account(request.clone()) .delete_envelopes_multi_account(request.clone())
.await?; .await?;
ATTACHMENT_MANAGER ATTACHMENT_MANAGER
.delete_envelopes_multi_account(request) .delete_attachments_multi_account(request)
.await .await
} }

View File

@@ -259,6 +259,12 @@ impl NewIndexWriter {
.parse(eml_bytes) .parse(eml_bytes)
.ok_or_else(|| raise_error!("failed to parse eml".into(), ErrorCode::InternalError))?; .ok_or_else(|| raise_error!("failed to parse eml".into(), ErrorCode::InternalError))?;
if message.parts.is_empty() {
return Err(raise_error!(
"Malformed or completely empty EML (no parts found)".into(),
ErrorCode::InternalError
));
}
// ── text / preview ──────────────────────────────────────────────── // ── text / preview ────────────────────────────────────────────────
let text = message let text = message
.body_text(0) .body_text(0)
@@ -436,7 +442,7 @@ impl NewIndexWriter {
.commit() .commit()
.map_err(|e| raise_error!(format!("{e:#?}"), ErrorCode::InternalError))?; .map_err(|e| raise_error!(format!("{e:#?}"), ErrorCode::InternalError))?;
} }
println!("tantivy commit elasped: {:#?}", start.elapsed()); println!("tantivy commit elapsed: {:#?}", start.elapsed());
tracing::info!(count = self.pending, "committed tantivy batch"); tracing::info!(count = self.pending, "committed tantivy batch");
self.pending = 0; self.pending = 0;
Ok(()) Ok(())

View File

@@ -260,7 +260,7 @@ impl IndexManager {
IndexRecordOption::Basic, IndexRecordOption::Basic,
); );
let envelope_id_query = TermQuery::new( let envelope_id_query = TermQuery::new(
Term::from_field_text(SchemaTools::attachment_fields().f_id, aid), Term::from_field_text(SchemaTools::attachment_fields().f_envelope_id, aid),
IndexRecordOption::Basic, IndexRecordOption::Basic,
); );
let boolean_query = BooleanQuery::new(vec![ let boolean_query = BooleanQuery::new(vec![
@@ -633,7 +633,7 @@ impl IndexManager {
Ok(()) Ok(())
} }
pub async fn delete_envelopes_multi_account( pub async fn delete_attachments_multi_account(
&self, &self,
deletes: HashMap<u64, Vec<String>>, deletes: HashMap<u64, Vec<String>>,
) -> BichonResult<()> { ) -> BichonResult<()> {

View File

@@ -158,10 +158,10 @@ fn dedup_account(
) -> BichonResult<u64> { ) -> BichonResult<u64> {
let searcher = email_reader.searcher(); let searcher = email_reader.searcher();
let fields = SchemaTools::email_fields(); let fields = SchemaTools::email_fields();
eprintln!( // eprintln!(
"DEBUG dedup_account: entry account={account_id} f_id_field={:?} f_content_hash_field={:?}", // "DEBUG dedup_account: entry account={account_id} f_id_field={:?} f_content_hash_field={:?}",
fields.f_id, fields.f_content_hash // fields.f_id, fields.f_content_hash
); // );
let mut map: DedupMap = HashMap::new(); let mut map: DedupMap = HashMap::new();
// ── Phase 1: build the dedup map via FAST column scans ────────────────── // ── Phase 1: build the dedup map via FAST column scans ──────────────────
@@ -204,7 +204,11 @@ fn dedup_account(
let ingest_at = ingest_col.values.get_val(doc_id); let ingest_at = ingest_col.values.get_val(doc_id);
// Read content_hash from the dictionary-encoded string column // Read content_hash from the dictionary-encoded string column
let hash_ord = hash_col.ords().values_for_doc(doc_id as u32).next().unwrap_or(0); let hash_ord = hash_col
.ords()
.values_for_doc(doc_id as u32)
.next()
.unwrap_or(0);
let mut hash_buf = String::new(); let mut hash_buf = String::new();
hash_col hash_col
.ord_to_str(hash_ord, &mut hash_buf) .ord_to_str(hash_ord, &mut hash_buf)
@@ -212,16 +216,20 @@ fn dedup_account(
let content_hash = hash_buf; let content_hash = hash_buf;
// Read f_id from the dictionary-encoded string column // Read f_id from the dictionary-encoded string column
let id_ord = id_col.ords().values_for_doc(doc_id as u32).next().unwrap_or(0); let id_ord = id_col
.ords()
.values_for_doc(doc_id as u32)
.next()
.unwrap_or(0);
let mut id_buf = String::new(); let mut id_buf = String::new();
id_col id_col
.ord_to_str(id_ord, &mut id_buf) .ord_to_str(id_ord, &mut id_buf)
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?; .map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
let email_id = id_buf; let email_id = id_buf;
eprintln!( // eprintln!(
"DEBUG dedup_account: account={account_id} doc_id={doc_id} mailbox={mailbox_id} hash={content_hash:?} id={email_id:?} ingest_at={ingest_at}" // "DEBUG dedup_account: account={account_id} doc_id={doc_id} mailbox={mailbox_id} hash={content_hash:?} id={email_id:?} ingest_at={ingest_at}"
); // );
map.entry((mailbox_id, content_hash)) map.entry((mailbox_id, content_hash))
.or_default() .or_default()
@@ -247,7 +255,11 @@ fn dedup_account(
// uidvalidity, which is required for correct incremental sync. // uidvalidity, which is required for correct incremental sync.
entries.sort_by_key(|e| std::cmp::Reverse(e.ingest_at)); entries.sort_by_key(|e| std::cmp::Reverse(e.ingest_at));
eprintln!("DEBUG Phase2: key={_key:?} kept={} deleting={}", entries[0].email_id, entries.len() - 1); eprintln!(
"DEBUG Phase2: key={_key:?} kept={} deleting={}",
entries[0].email_id,
entries.len() - 1
);
// Keep entries[0], soft-delete everything else via term query on f_id // Keep entries[0], soft-delete everything else via term query on f_id
for entry in &entries[1..] { for entry in &entries[1..] {
eprintln!( eprintln!(
@@ -315,13 +327,14 @@ mod tests {
/// Collect non-deleted f_id values from the email index. /// Collect non-deleted f_id values from the email index.
fn surviving_email_ids(reader: &IndexReader) -> HashSet<String> { fn surviving_email_ids(reader: &IndexReader) -> HashSet<String> {
reader reader.reload().expect("reader reload failed");
.reload()
.expect("reader reload failed");
let searcher = reader.searcher(); let searcher = reader.searcher();
let mut ids = HashSet::new(); let mut ids = HashSet::new();
let segments = searcher.segment_readers(); let segments = searcher.segment_readers();
eprintln!("DEBUG surviving_email_ids: segment_count={}", segments.len()); eprintln!(
"DEBUG surviving_email_ids: segment_count={}",
segments.len()
);
for (seg_idx, seg) in segments.iter().enumerate() { for (seg_idx, seg) in segments.iter().enumerate() {
let id_col = seg let id_col = seg
.fast_fields() .fast_fields()
@@ -332,7 +345,11 @@ mod tests {
eprintln!("DEBUG surviving_email_ids: seg={seg_idx} max_doc={max_doc}"); eprintln!("DEBUG surviving_email_ids: seg={seg_idx} max_doc={max_doc}");
for doc_id in 0..max_doc { for doc_id in 0..max_doc {
let is_del = seg.is_deleted(doc_id); let is_del = seg.is_deleted(doc_id);
let ord = id_col.ords().values_for_doc(doc_id as u32).next().unwrap_or(0); let ord = id_col
.ords()
.values_for_doc(doc_id as u32)
.next()
.unwrap_or(0);
let mut buf = String::new(); let mut buf = String::new();
id_col.ord_to_str(ord, &mut buf).unwrap(); id_col.ord_to_str(ord, &mut buf).unwrap();
eprintln!("DEBUG surviving_email_ids: seg={seg_idx} doc_id={doc_id} is_deleted={is_del} ord={ord} buf={buf:?}"); eprintln!("DEBUG surviving_email_ids: seg={seg_idx} doc_id={doc_id} is_deleted={is_del} ord={ord} buf={buf:?}");
@@ -359,7 +376,11 @@ mod tests {
if seg.is_deleted(doc_id) { if seg.is_deleted(doc_id) {
continue; continue;
} }
let ord = id_col.ords().values_for_doc(doc_id as u32).next().unwrap_or(0); let ord = id_col
.ords()
.values_for_doc(doc_id as u32)
.next()
.unwrap_or(0);
let mut buf = String::new(); let mut buf = String::new();
id_col.ord_to_str(ord, &mut buf).unwrap(); id_col.ord_to_str(ord, &mut buf).unwrap();
ids.insert(buf); ids.insert(buf);
@@ -454,15 +475,17 @@ mod tests {
let email_r = email_idx.reader().unwrap(); let email_r = email_idx.reader().unwrap();
let survivors = surviving_email_ids(&email_r); let survivors = surviving_email_ids(&email_r);
let expected: HashSet<String> = let expected: HashSet<String> = expected_emails.iter().map(|s| s.to_string()).collect();
expected_emails.iter().map(|s| s.to_string()).collect();
assert_eq!(survivors, expected, "[{case}] email survivors mismatch"); assert_eq!(survivors, expected, "[{case}] email survivors mismatch");
let attach_r = attach_idx.reader().unwrap(); let attach_r = attach_idx.reader().unwrap();
let att_survivors = surviving_attachment_ids(&attach_r); let att_survivors = surviving_attachment_ids(&attach_r);
let att_expected: HashSet<String> = let att_expected: HashSet<String> =
expected_attachments.iter().map(|s| s.to_string()).collect(); expected_attachments.iter().map(|s| s.to_string()).collect();
assert_eq!(att_survivors, att_expected, "[{case}] attachment survivors mismatch"); assert_eq!(
att_survivors, att_expected,
"[{case}] attachment survivors mismatch"
);
} }
} }
@@ -591,7 +614,7 @@ mod tests {
/// This test is read-only — it does not modify the index. /// This test is read-only — it does not modify the index.
#[test] #[test]
fn inspect_production_duplicates() { fn inspect_production_duplicates() {
let index_path = r"E:\db\data\bichon-indices\mail_metadata"; let index_path = r"E:\bichon-data\bichon-indices\mail_metadata";
let report_path = std::path::PathBuf::from(r"E:\bichon\dedup_report.txt"); let report_path = std::path::PathBuf::from(r"E:\bichon\dedup_report.txt");
let mut report = String::new(); let mut report = String::new();
@@ -622,26 +645,21 @@ mod tests {
let searcher = reader.searcher(); let searcher = reader.searcher();
let mut total_docs = 0u64; let mut total_docs = 0u64;
let mut groups: std::collections::HashMap<u64, std::collections::HashMap<(u64, String), u64>> = let mut groups: std::collections::HashMap<
std::collections::HashMap::new(); u64,
std::collections::HashMap<(u64, String), u64>,
> = std::collections::HashMap::new();
for segment_reader in searcher.segment_readers() { for segment_reader in searcher.segment_readers() {
let account_col = segment_reader let account_col = segment_reader.fast_fields().u64(F_ACCOUNT_ID).unwrap();
.fast_fields() let mailbox_col = segment_reader.fast_fields().u64(F_MAILBOX_ID).unwrap();
.u64(F_ACCOUNT_ID) let hash_col = match segment_reader.fast_fields().str(F_CONTENT_HASH).unwrap() {
.unwrap();
let mailbox_col = segment_reader
.fast_fields()
.u64(F_MAILBOX_ID)
.unwrap();
let hash_col = match segment_reader
.fast_fields()
.str(F_CONTENT_HASH)
.unwrap()
{
Some(c) => c, Some(c) => c,
None => { None => {
let _ = writeln!(report, "Segment has no FAST str column for content_hash, skipping"); let _ = writeln!(
report,
"Segment has no FAST str column for content_hash, skipping"
);
continue; continue;
} }
}; };
@@ -655,7 +673,11 @@ mod tests {
let account_id = account_col.values.get_val(doc_id); let account_id = account_col.values.get_val(doc_id);
let mailbox_id = mailbox_col.values.get_val(doc_id); let mailbox_id = mailbox_col.values.get_val(doc_id);
let hash_ord = hash_col.ords().values_for_doc(doc_id as u32).next().unwrap_or(0); let hash_ord = hash_col
.ords()
.values_for_doc(doc_id as u32)
.next()
.unwrap_or(0);
let mut hash_buf = String::new(); let mut hash_buf = String::new();
hash_col.ord_to_str(hash_ord, &mut hash_buf).unwrap(); hash_col.ord_to_str(hash_ord, &mut hash_buf).unwrap();
let content_hash = hash_buf; let content_hash = hash_buf;

View File

@@ -26,10 +26,10 @@ use std::sync::LazyLock;
use crate::error::code::ErrorCode; use crate::error::code::ErrorCode;
use crate::error::BichonResult; use crate::error::BichonResult;
use crate::settings::cli::SETTINGS;
use crate::raise_error; use crate::raise_error;
use crate::settings::cli::SETTINGS;
static ENCRYPT_PASSWORD: LazyLock<String> = LazyLock::new(|| { pub static ENCRYPT_PASSWORD: LazyLock<String> = LazyLock::new(|| {
if let Some(file_path) = &SETTINGS.bichon_encrypt_password_file { if let Some(file_path) = &SETTINGS.bichon_encrypt_password_file {
return fs::read_to_string(file_path) return fs::read_to_string(file_path)
.expect("failed to read the file with the encrypt password") .expect("failed to read the file with the encrypt password")
@@ -102,7 +102,10 @@ pub fn internal_encrypt_string(
Ok(general_purpose::URL_SAFE.encode(&result)) Ok(general_purpose::URL_SAFE.encode(&result))
} }
pub fn internal_decrypt_string(password: &str, data: &str) -> Result<String, ring::error::Unspecified> { pub fn internal_decrypt_string(
password: &str,
data: &str,
) -> Result<String, ring::error::Unspecified> {
let data = general_purpose::URL_SAFE let data = general_purpose::URL_SAFE
.decode(data) .decode(data)
.map_err(|_| ring::error::Unspecified)?; .map_err(|_| ring::error::Unspecified)?;
@@ -146,8 +149,7 @@ mod tests {
#[test] #[test]
fn test_wrong_password_fails() { fn test_wrong_password_fails() {
let encrypted = let encrypted = internal_encrypt_string("correct_password", "secret").unwrap();
internal_encrypt_string("correct_password", "secret").unwrap();
assert!(internal_decrypt_string("wrong_password", &encrypted).is_err()); assert!(internal_decrypt_string("wrong_password", &encrypted).is_err());
} }

View File

@@ -73,8 +73,9 @@ async fn main() -> BichonResult<()> {
Ok(false) => { Ok(false) => {
error!("Incompatible data format detected."); error!("Incompatible data format detected.");
error!("Your data was created by an older version of Bichon and must be migrated before use."); error!("Your data was created by an older version of Bichon and must be migrated before use.");
error!("Please stop the Bichon v0.3.7 service before migration.");
error!("Please run: bichon-admin"); error!("Please run: bichon-admin");
error!("Documentation: https://github.com/rustmailer/bichon/wiki/migration"); error!("Documentation: https://github.com/rustmailer/bichon/wiki/Bichon-Data-Migration:-v0.3.7-%E2%86%92-v1.0");
return Err(raise_error!( return Err(raise_error!(
"Legacy data layout detected".into(), "Legacy data layout detected".into(),
ErrorCode::InternalError ErrorCode::InternalError
@@ -202,7 +203,10 @@ mod api_tests {
.map(|v| v.object().get("name").string()) .map(|v| v.object().get("name").string())
.collect(); .collect();
assert!(tag_names.contains(&"AccessToken"), "missing AccessToken tag"); assert!(
tag_names.contains(&"AccessToken"),
"missing AccessToken tag"
);
assert!(tag_names.contains(&"Attachment"), "missing Attachment tag"); assert!(tag_names.contains(&"Attachment"), "missing Attachment tag");
assert!(tag_names.contains(&"AutoConfig"), "missing AutoConfig tag"); assert!(tag_names.contains(&"AutoConfig"), "missing AutoConfig tag");
assert!(tag_names.contains(&"Account"), "missing Account tag"); assert!(tag_names.contains(&"Account"), "missing Account tag");