mirror of
https://github.com/rustmailer/bichon.git
synced 2026-08-31 01:52:30 +00:00
Compare commits
234 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5fe45795b9 | ||
|
|
4996dac1ea | ||
|
|
42ce36a439 | ||
|
|
40ae2d49d4 | ||
|
|
5034760517 | ||
|
|
875eb2e309 | ||
|
|
f0d0f202ba | ||
|
|
0d37825625 | ||
|
|
cfb8172607 | ||
|
|
8641bb4b56 | ||
|
|
cdf27f2dd4 | ||
|
|
db36272bae | ||
|
|
cad447275f | ||
|
|
8542ba0d28 | ||
|
|
af089958e9 | ||
|
|
bba1ea5cc7 | ||
|
|
9e55026f12 | ||
|
|
b0f229618c | ||
|
|
ff59d47a0d | ||
|
|
b6957c8ceb | ||
|
|
cd98c050b6 | ||
|
|
11e40750fe | ||
|
|
b0374517e8 | ||
|
|
4e12ae5457 | ||
|
|
495c91b7ae | ||
|
|
e442db4a2d | ||
|
|
cc54063669 | ||
|
|
04745458bc | ||
|
|
6fcc9e0e8e | ||
|
|
220aa268c1 | ||
|
|
e3fd9d2f29 | ||
|
|
41c3b84e65 | ||
|
|
89557700ae | ||
|
|
2f5de48c6a | ||
|
|
debb119d3d | ||
|
|
ce3f8944a3 | ||
|
|
6f572b15ae | ||
|
|
3bfc080258 | ||
|
|
c3a725770c | ||
|
|
c736afffb0 | ||
|
|
62cb5264fd | ||
|
|
aebb94ee4e | ||
|
|
769630f9d7 | ||
|
|
368b18c45f | ||
|
|
42861f6cc9 | ||
|
|
327a3f39d9 | ||
|
|
427f7248d2 | ||
|
|
a2a51a2037 | ||
|
|
4d783d5301 | ||
|
|
4172f11f00 | ||
|
|
257736a47b | ||
|
|
e8469da3bc | ||
|
|
1346dd216a | ||
|
|
d40ba90b54 | ||
|
|
048d5f361c | ||
|
|
4597df515a | ||
|
|
f4be4a2e8c | ||
|
|
4a3c42c1eb | ||
|
|
8fcb55320f | ||
|
|
86869ac848 | ||
|
|
38453accc6 | ||
|
|
a60b2c7dc4 | ||
|
|
8e46c7a162 | ||
|
|
1a615e1c45 | ||
|
|
83dd9cdd6b | ||
|
|
f30cd66e00 | ||
|
|
4bd714a670 | ||
|
|
c0a63a1e3c | ||
|
|
04a022e850 | ||
|
|
f9c2fc77ff | ||
|
|
ec3e842bbb | ||
|
|
7311529908 | ||
|
|
0792bb546d | ||
|
|
0be2670600 | ||
|
|
575f851cfb | ||
|
|
61430b72b0 | ||
|
|
15c0cfc1d9 | ||
|
|
ea8c493374 | ||
|
|
005b1c2116 | ||
|
|
d8b78b8010 | ||
|
|
36f3f19cdc | ||
|
|
9f3097df32 | ||
|
|
c075b9ef12 | ||
|
|
21a7f7e9d5 | ||
|
|
26c14fcaaf | ||
|
|
6873841ba4 | ||
|
|
0dd81f599d | ||
|
|
b3afc52a82 | ||
|
|
171a40d70f | ||
|
|
895ea543a9 | ||
|
|
6eca351994 | ||
|
|
1f477eca65 | ||
|
|
a3cdc094e8 | ||
|
|
95147a7824 | ||
|
|
b22811f78c | ||
|
|
fd61d013a2 | ||
|
|
3a950e7591 | ||
|
|
f17820bfa8 | ||
|
|
178b25d27d | ||
|
|
d160ca75f5 | ||
|
|
04136a4ae2 | ||
|
|
1d6f5d9a22 | ||
|
|
105a6d9b15 | ||
|
|
df440c8441 | ||
|
|
4116a59b79 | ||
|
|
609eee1b84 | ||
|
|
79b9f07888 | ||
|
|
ba28369202 | ||
|
|
ff64b66f79 | ||
|
|
a4f8e674c3 | ||
|
|
dde6b990da | ||
|
|
6b1f843bd5 | ||
|
|
66fd50bc23 | ||
|
|
9daab241b0 | ||
|
|
85d5490834 | ||
|
|
6e984f376c | ||
|
|
7470125a23 | ||
|
|
469d254e2b | ||
|
|
7fde7ee19a | ||
|
|
d543508a23 | ||
|
|
f440069912 | ||
|
|
9f9fc71d16 | ||
|
|
b2a75643da | ||
|
|
37a38a2910 | ||
|
|
8817ed96f6 | ||
|
|
1ee2eade3a | ||
|
|
7afb1e29aa | ||
|
|
715858183d | ||
|
|
407d865a9a | ||
|
|
e708dc524b | ||
|
|
55996f8f9b | ||
|
|
8bb39b4095 | ||
|
|
0a7a22fa7f | ||
|
|
3fa4fe453c | ||
|
|
907e59027d | ||
|
|
3ea330c884 | ||
|
|
d52c5eef3e | ||
|
|
6b6f11e4c1 | ||
|
|
1682f21cf7 | ||
|
|
5406c4322c | ||
|
|
0abaa66a40 | ||
|
|
123260b69d | ||
|
|
c086a3caf5 | ||
|
|
c07e39495a | ||
|
|
41322c8357 | ||
|
|
2f8ba5ad40 | ||
|
|
213c6452a8 | ||
|
|
9c91025c53 | ||
|
|
be934e2b3f | ||
|
|
12e7bb6ba3 | ||
|
|
f66a86392d | ||
|
|
66b595908c | ||
|
|
174d56e7b4 | ||
|
|
7eacfbfb20 | ||
|
|
2802c7ea07 | ||
|
|
f583c3413c | ||
|
|
9bacf3fb7a | ||
|
|
e2fb0ee39e | ||
|
|
1be5fea51a | ||
|
|
2d29a8111b | ||
|
|
0c2e540834 | ||
|
|
e83a00fe39 | ||
|
|
fa70437a62 | ||
|
|
0b866c81ff | ||
|
|
5b884125f7 | ||
|
|
c3a12eafb2 | ||
|
|
c14834abe8 | ||
|
|
51329fb2e1 | ||
|
|
5ad940269f | ||
|
|
f9ceb83293 | ||
|
|
c185ea102c | ||
|
|
da79916396 | ||
|
|
b29c6ea9ff | ||
|
|
7dd722f9b8 | ||
|
|
19b5168960 | ||
|
|
d90943bbfe | ||
|
|
18a0d52c57 | ||
|
|
15dd26228c | ||
|
|
39d8168de5 | ||
|
|
82915e76ba | ||
|
|
6d5953c73b | ||
|
|
61161b5f3b | ||
|
|
286ae16057 | ||
|
|
8b4dc44c07 | ||
|
|
64a66b4f98 | ||
|
|
5a50f5e327 | ||
|
|
bd10e15c65 | ||
|
|
c123ecb24a | ||
|
|
01182dc90d | ||
|
|
7626634863 | ||
|
|
0914bf710e | ||
|
|
3f11c5dbbf | ||
|
|
5d0039cb74 | ||
|
|
a41b5417e3 | ||
|
|
c19f3977ba | ||
|
|
884fdeba10 | ||
|
|
2228e98410 | ||
|
|
d690f57290 | ||
|
|
8a42fcdb4a | ||
|
|
5028061f20 | ||
|
|
a8b3b24d59 | ||
|
|
af0f47c0e3 | ||
|
|
2b10d201ee | ||
|
|
638a93f184 | ||
|
|
40eca89a75 | ||
|
|
f3c46f97b9 | ||
|
|
396383aa97 | ||
|
|
8f331080bf | ||
|
|
4b0d571cf2 | ||
|
|
16f0fad91e | ||
|
|
dda6d77046 | ||
|
|
a273b7f5e1 | ||
|
|
2cba001431 | ||
|
|
54ed2c3c0a | ||
|
|
5e3d0f1c06 | ||
|
|
a9a9b4a85f | ||
|
|
a5ea98f731 | ||
|
|
21c3d2b795 | ||
|
|
c15fe2a503 | ||
|
|
5f013ea173 | ||
|
|
393b7361e8 | ||
|
|
fd35f4be8e | ||
|
|
d2936ed4a7 | ||
|
|
ef4ab3496e | ||
|
|
2295585deb | ||
|
|
57afa30b5b | ||
|
|
ba8ecdd899 | ||
|
|
673e593c4f | ||
|
|
579822762f | ||
|
|
d63b1e0d7c | ||
|
|
a0d8d069c0 | ||
|
|
bdbbc04832 | ||
|
|
fed28c3eca | ||
|
|
01dba4f71b |
@@ -8,6 +8,12 @@ assignees: ""
|
||||
|
||||
> **Please write and communicate in English.**
|
||||
|
||||
|
||||
---
|
||||
**Help us build a more stable Bichon!** 🛠️
|
||||
While we look into this bug, consider sharing your usage patterns in our [2026 Roadmap Survey](https://docs.google.com/forms/d/e/1FAIpQLScOlwsiUMfyQPBCLW2MLkygdRmAutEgvXDYPzzvEGPz0HFPXQ/viewform) to help us prioritize stability and features.
|
||||
---
|
||||
|
||||
### Version
|
||||
Which version are you using?
|
||||
|
||||
7
.github/ISSUE_TEMPLATE/feature.md
vendored
7
.github/ISSUE_TEMPLATE/feature.md
vendored
@@ -8,6 +8,13 @@ assignees: ""
|
||||
|
||||
> **Please write and communicate in English.**
|
||||
|
||||
---
|
||||
### 📢 Shape the Future of Bichon
|
||||
**Want your requested feature to be prioritized?** Help us shape the **2026 Roadmap** by filling out our 1-minute survey:
|
||||
👉 **[Bichon User Survey](https://docs.google.com/forms/d/e/1FAIpQLScOlwsiUMfyQPBCLW2MLkygdRmAutEgvXDYPzzvEGPz0HFPXQ/viewform)**
|
||||
---
|
||||
|
||||
|
||||
### Description
|
||||
What feature would you like to see?
|
||||
|
||||
|
||||
14
.github/workflows/release.yml
vendored
14
.github/workflows/release.yml
vendored
@@ -5,8 +5,8 @@ on:
|
||||
tags:
|
||||
- '[0-9]+.[0-9]+.[0-9]+'
|
||||
env:
|
||||
BINARY_NAME: bichon
|
||||
BINARY_CTL: bichonctl
|
||||
BINARY_NAME: bichon-server
|
||||
BINARY_CLI: bichon-cli
|
||||
BINARY_ADMIN: bichon-admin
|
||||
|
||||
permissions:
|
||||
@@ -82,18 +82,18 @@ jobs:
|
||||
if: matrix.target == 'aarch64-unknown-linux-gnu'
|
||||
run: |
|
||||
cargo install cross --force
|
||||
cross build --release --features vendored-openssl --target=${{ matrix.target }}
|
||||
cross build --release --target=${{ matrix.target }}
|
||||
|
||||
- name: Build Rust backend
|
||||
if: matrix.target != 'aarch64-unknown-linux-gnu'
|
||||
run: |
|
||||
cargo build --release --features vendored-openssl --target=${{ matrix.target }}
|
||||
cargo build --release --target=${{ matrix.target }}
|
||||
|
||||
- name: Strip binary (Linux and macOS)
|
||||
if: matrix.os != 'windows-latest' && matrix.target != 'aarch64-unknown-linux-gnu'
|
||||
run: |
|
||||
strip target/${{ matrix.target }}/release/${{ env.BINARY_NAME }}
|
||||
strip target/${{ matrix.target }}/release/${{ env.BINARY_CTL }}
|
||||
strip target/${{ matrix.target }}/release/${{ env.BINARY_CLI }}
|
||||
strip target/${{ matrix.target }}/release/${{ env.BINARY_ADMIN }}
|
||||
|
||||
- name: Pack artifact (Linux/macOS)
|
||||
@@ -104,7 +104,7 @@ jobs:
|
||||
BINARY="target/${{ matrix.target }}/release/${{ env.BINARY_NAME }}"
|
||||
cp README.md LICENSE release/
|
||||
cp target/${{ matrix.target }}/release/${{ env.BINARY_NAME }} release/
|
||||
cp target/${{ matrix.target }}/release/${{ env.BINARY_CTL }} release/
|
||||
cp target/${{ matrix.target }}/release/${{ env.BINARY_CLI }} release/
|
||||
cp target/${{ matrix.target }}/release/${{ env.BINARY_ADMIN }} release/
|
||||
tar -czvf "${{ env.BINARY_NAME }}-${{ github.ref_name }}-${{ matrix.target }}.tar.gz" -C release .
|
||||
mv "${{ env.BINARY_NAME }}-${{ github.ref_name }}-${{ matrix.target }}.tar.gz" release/
|
||||
@@ -123,7 +123,7 @@ jobs:
|
||||
mkdir -p release
|
||||
|
||||
Copy-Item "target/${{ matrix.target }}/release/${{ env.BINARY_NAME }}.exe" release/
|
||||
Copy-Item "target/${{ matrix.target }}/release/${{ env.BINARY_CTL }}.exe" release/
|
||||
Copy-Item "target/${{ matrix.target }}/release/${{ env.BINARY_CLI }}.exe" release/
|
||||
Copy-Item "target/${{ matrix.target }}/release/${{ env.BINARY_ADMIN }}.exe" release/
|
||||
|
||||
Copy-Item -Path README.md -Destination release/
|
||||
|
||||
4
.gitignore
vendored
4
.gitignore
vendored
@@ -1,3 +1,7 @@
|
||||
/target
|
||||
.vscode
|
||||
.idea
|
||||
config.toml
|
||||
node_modules
|
||||
dedup_report.txt
|
||||
crates/*/target
|
||||
3830
Cargo.lock
generated
3830
Cargo.lock
generated
File diff suppressed because it is too large
Load Diff
151
Cargo.toml
151
Cargo.toml
@@ -1,57 +1,36 @@
|
||||
[package]
|
||||
name = "bichon"
|
||||
version = "0.3.6"
|
||||
[workspace]
|
||||
|
||||
members = [
|
||||
"crates/memdb",
|
||||
"crates/core",
|
||||
"crates/blob",
|
||||
"crates/server",
|
||||
"crates/cli",
|
||||
"crates/admin",
|
||||
"crates/smtp",
|
||||
]
|
||||
|
||||
resolver = "2"
|
||||
|
||||
[workspace.package]
|
||||
version = "1.6.1"
|
||||
edition = "2021"
|
||||
|
||||
[[bin]]
|
||||
name = "bichon"
|
||||
path = "src/main.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "bichonctl"
|
||||
path = "src/bin/bichonctl.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "bichon-admin"
|
||||
path = "src/bin/bichon_admin.rs"
|
||||
|
||||
|
||||
[features]
|
||||
default = []
|
||||
vendored-openssl = ["openssl-sys"]
|
||||
|
||||
[profile.release]
|
||||
strip = true
|
||||
lto = true
|
||||
opt-level = 3
|
||||
codegen-units = 1
|
||||
|
||||
[dependencies]
|
||||
chrono = "0.4.43"
|
||||
clap = { version = "4.5.54", features = ["derive", "env"] }
|
||||
mimalloc = "0.1.48"
|
||||
native_db = "0.8.2"
|
||||
itertools = "0.14.0"
|
||||
native_model = "0.4.20"
|
||||
poem = { version = "3.1.12", features = ["embed", "compression", "rustls"] }
|
||||
poem-derive = "3.1.12"
|
||||
poem-openapi = { version = "5.1.16", features = [
|
||||
"openapi-explorer",
|
||||
"rapidoc",
|
||||
"scalar",
|
||||
"redoc",
|
||||
"swagger-ui",
|
||||
"email",
|
||||
] }
|
||||
[workspace.dependencies]
|
||||
chrono = "0.4.45"
|
||||
clap = { version = "4.6.1", features = ["derive", "env"] }
|
||||
bichon-memdb = { path = "crates/memdb" }
|
||||
bichon-blob = { path = "crates/blob" }
|
||||
itertools = "0.15.0"
|
||||
ring = { version = "0.17.14", features = ["std"] }
|
||||
serde = { version = "1.0.228", features = ["derive"] }
|
||||
serde_json = "1.0.149"
|
||||
tokio = { version = "1.49.0", features = ["full"] }
|
||||
serde_json = "1.0.150"
|
||||
tokio = { version = "1.52.3", features = ["full"] }
|
||||
tracing = "0.1.44"
|
||||
tracing-appender = "0.2.3"
|
||||
tracing-subscriber = { version = "0.3.22", features = ["env-filter", "json"] }
|
||||
tracing-subscriber = { version = "0.3.23", features = ["env-filter", "json"] }
|
||||
base64 = "0.22.1"
|
||||
snafu = "0.8.9"
|
||||
snafu = "0.9.1"
|
||||
reqwest = { version = "0.12.24", default-features = false, features = [
|
||||
"json",
|
||||
"stream",
|
||||
@@ -60,71 +39,61 @@ reqwest = { version = "0.12.24", default-features = false, features = [
|
||||
"blocking",
|
||||
"socks",
|
||||
] }
|
||||
tokio-socks = "0.5.2"
|
||||
http = "1.4.0"
|
||||
regex = "1.12.2"
|
||||
tokio-socks = "0.5.3"
|
||||
http = "1.4.2"
|
||||
regex = "1.12.4"
|
||||
email_address = "0.2.9"
|
||||
futures = "0.3.31"
|
||||
futures = "0.3.32"
|
||||
utf7-imap = "0.3.2"
|
||||
imap-proto = "0.16.6"
|
||||
mail-parser = { version = '0.11.1', features = ["serde"] }
|
||||
mail-parser = { version = '0.11.4', features = ["serde"] }
|
||||
# mail-send = "0.5.2"
|
||||
tokio-rustls = { version = "0.26.4", default-features = false, features = [
|
||||
"ring",
|
||||
"tls12",
|
||||
] }
|
||||
timeago = "0.5.0"
|
||||
ahash = "0.8.12"
|
||||
timeago = "0.6.0"
|
||||
oauth2 = { version = "5.0.0", features = ["reqwest-blocking"] }
|
||||
url = { version = "2.5.8", features = ["serde"] }
|
||||
sysinfo = "0.37.2"
|
||||
sysinfo = "0.39.4"
|
||||
num_cpus = "1.17.0"
|
||||
cacache = { version = "13.1.0", default-features = false, features = [
|
||||
"tokio-runtime",
|
||||
"mmap",
|
||||
] }
|
||||
rand = "0.9.2"
|
||||
rand = "0.10.1"
|
||||
encoding_rs = "0.8.35"
|
||||
async-imap = { version = "0.11.1", default-features = false, features = [
|
||||
"runtime-tokio",
|
||||
"compress",
|
||||
] }
|
||||
webpki-roots = "1.0.5"
|
||||
rustls = { version = "0.23.36", default-features = false, features = ["ring"] }
|
||||
rustls-pki-types = "1.14.0"
|
||||
webpki-roots = "1.0.8"
|
||||
rustls = { version = "0.23.41", default-features = false, features = ["ring"] }
|
||||
rustls-pki-types = "1.14.1"
|
||||
tokio-io-timeout = "1.2.1"
|
||||
bb8 = "0.9.1"
|
||||
semver = "1.0.27"
|
||||
semver = "1.0.28"
|
||||
governor = "0.10.4"
|
||||
lru = "0.16.3"
|
||||
lru = "0.18.0"
|
||||
mime_guess = "2.0.5"
|
||||
hex = "0.4.3"
|
||||
time = { version = "0.3.45", features = [
|
||||
time = { version = "0.3.51", features = [
|
||||
"formatting",
|
||||
"parsing",
|
||||
"local-offset",
|
||||
] }
|
||||
rust-embed = "8.11.0"
|
||||
murmur3 = "0.5.2"
|
||||
autoconfig = "0.4.0"
|
||||
urlencoding = "2.1.3"
|
||||
dashmap = "6.1.0"
|
||||
# Statically links OpenSSL by compiling from source, avoiding system library dependencies
|
||||
openssl-sys = { version = "0.9.111", optional = true, features = ["vendored"] }
|
||||
dashmap = "6.2.1"
|
||||
gethostname = "1.1.0"
|
||||
tantivy = { version = "0.25.0", features = ["quickwit", "zstd-compression"] }
|
||||
itoa = "1.0.17"
|
||||
html2text = "0.16.6"
|
||||
bytes = "1.11.0"
|
||||
itoa = "1.0.18"
|
||||
html2text = "0.17.1"
|
||||
bytes = "1.11.1"
|
||||
dialoguer = "0.12.0"
|
||||
console = "0.16.2"
|
||||
toml = "0.9.8"
|
||||
memmap2 = "0.9.9"
|
||||
outlook-pst = { git = "https://github.com/rustmailer/outlook-pst-rs.git", branch = "main" }
|
||||
compressed-rtf = "1.0.0"
|
||||
codepage-strings = "1.0.2"
|
||||
mail-send = "0.5.2"
|
||||
[dev-dependencies]
|
||||
#bincode = "1.3.3"
|
||||
#secret-lib = "1.0.0"
|
||||
tempfile = "3.24.0"
|
||||
console = "0.16.3"
|
||||
mail-send = "0.6.0"
|
||||
rcgen = "0.14.8"
|
||||
rustls-pemfile = "2.2.0"
|
||||
blake3 = "1.8.5"
|
||||
uuid = { version = "1.23.4", features = ["v4", "serde"] }
|
||||
fjall = { version = "3.1.5", features = ["lz4", "metrics", "bytes_1"] }
|
||||
tracing-log = "0.2.0"
|
||||
tokio-util = "0.7.18"
|
||||
indicatif = "0.18.4"
|
||||
|
||||
[profile.release]
|
||||
strip = true
|
||||
lto = true
|
||||
opt-level = 3
|
||||
codegen-units = 1
|
||||
|
||||
@@ -1,2 +1,2 @@
|
||||
base_url = "http://localhost:15630"
|
||||
api_token = "lZHmfpH1CRr9XsRiOGd1RnOr"
|
||||
api_token = "eErI7WN3PtKeLwWAbIfSXCP6"
|
||||
|
||||
20
crates/admin/Cargo.toml
Normal file
20
crates/admin/Cargo.toml
Normal file
@@ -0,0 +1,20 @@
|
||||
[package]
|
||||
name = "bichon-admin"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
|
||||
|
||||
[dependencies]
|
||||
bichon-core = { path = "../core" }
|
||||
tokio.workspace = true
|
||||
dialoguer.workspace = true
|
||||
console.workspace = true
|
||||
indicatif.workspace = true
|
||||
native_db = "0.8.2"
|
||||
native_model = "0.4.20"
|
||||
serde.workspace = true
|
||||
serde_json.workspace = true
|
||||
|
||||
itertools.workspace = true
|
||||
snafu.workspace = true
|
||||
bichon-memdb.workspace = true
|
||||
61
crates/admin/src/main.rs
Normal file
61
crates/admin/src/main.rs
Normal file
@@ -0,0 +1,61 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use console::style;
|
||||
use dialoguer::{theme::ColorfulTheme, Select};
|
||||
|
||||
use crate::{migrate::handle_migration, reset::handle_reset_password};
|
||||
|
||||
pub mod meta;
|
||||
pub mod migrate;
|
||||
pub mod reset;
|
||||
|
||||
|
||||
fn main() {
|
||||
run_interactive();
|
||||
}
|
||||
|
||||
#[tokio::main]
|
||||
async fn run_interactive() {
|
||||
let theme = ColorfulTheme::default();
|
||||
println!(
|
||||
"\n{}\n",
|
||||
style("BICHON ADMINISTRATIVE TOOL").bold().bright().cyan()
|
||||
);
|
||||
|
||||
let main_options = vec![
|
||||
"Reset Admin Password",
|
||||
"Migrate Legacy v0.3.7 Storage to v1.x",
|
||||
"Exit",
|
||||
];
|
||||
|
||||
let selection = Select::with_theme(&theme)
|
||||
.with_prompt("Select an operation")
|
||||
.default(0)
|
||||
.items(&main_options)
|
||||
.interact()
|
||||
.unwrap();
|
||||
|
||||
match selection {
|
||||
0 => handle_reset_password(&theme),
|
||||
1 => handle_migration(&theme),
|
||||
_ => {
|
||||
println!("{}", style("Exiting...").dim());
|
||||
}
|
||||
}
|
||||
}
|
||||
917
crates/admin/src/meta.rs
Normal file
917
crates/admin/src/meta.rs
Normal file
@@ -0,0 +1,917 @@
|
||||
use std::{
|
||||
collections::{BTreeMap, BTreeSet},
|
||||
path::PathBuf,
|
||||
sync::{Arc, LazyLock},
|
||||
};
|
||||
|
||||
use bichon_core::{
|
||||
account::{
|
||||
entity::ImapConfig,
|
||||
migration::{AccountModel, AccountType},
|
||||
since::{DateSince, RelativeDate},
|
||||
},
|
||||
autoconfig::entity::MailServerConfig,
|
||||
cache::imap::mailbox::Attribute,
|
||||
database::batch_insert_impl,
|
||||
error::{code::ErrorCode, BichonError, BichonResult},
|
||||
raise_error,
|
||||
token::TokenType,
|
||||
users::{acl::AccessControl, role::RoleType},
|
||||
};
|
||||
use bichon_memdb::{Durability, MemDb};
|
||||
use console::style;
|
||||
use itertools::Itertools;
|
||||
use native_db::*;
|
||||
use native_model::{native_model, Model};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
pub const DEFAULT_ADMIN_USER_ID: u64 = 100000000000000;
|
||||
|
||||
#[derive(Clone, Debug, Default, PartialEq, Eq, Deserialize, Serialize)]
|
||||
#[native_model(id = 3, version = 1)]
|
||||
#[native_db]
|
||||
pub struct CachedMailSettings {
|
||||
#[primary_key]
|
||||
pub domain: String,
|
||||
pub config: MailServerConfig,
|
||||
pub created_at: i64,
|
||||
}
|
||||
|
||||
impl From<CachedMailSettings> for bichon_core::autoconfig::CachedMailSettings {
|
||||
fn from(value: CachedMailSettings) -> Self {
|
||||
Self {
|
||||
domain: value.domain,
|
||||
config: value.config,
|
||||
created_at: value.created_at,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[native_model(id = 4, version = 1)]
|
||||
#[native_db(primary_key(pk -> String))]
|
||||
pub struct AccountV1 {
|
||||
#[secondary_key(unique)]
|
||||
pub id: u64,
|
||||
pub imap: Option<ImapConfig>,
|
||||
pub enabled: bool,
|
||||
pub email: String,
|
||||
pub name: Option<String>,
|
||||
pub capabilities: Option<Vec<String>>,
|
||||
pub date_since: Option<DateSince>,
|
||||
pub folder_limit: Option<u32>,
|
||||
pub sync_folders: Option<Vec<String>>,
|
||||
pub account_type: AccountType,
|
||||
pub sync_interval_min: Option<i64>,
|
||||
pub known_folders: Option<BTreeSet<String>>,
|
||||
pub created_at: i64,
|
||||
pub updated_at: i64,
|
||||
pub use_proxy: Option<u64>,
|
||||
}
|
||||
impl AccountV1 {
|
||||
fn pk(&self) -> String {
|
||||
format!("{}_{}", self.created_at, self.id)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[native_model(id = 4, version = 2, from = AccountV1)]
|
||||
#[native_db(primary_key(pk -> String))]
|
||||
pub struct AccountV2 {
|
||||
#[secondary_key(unique)]
|
||||
pub id: u64,
|
||||
pub imap: Option<ImapConfig>,
|
||||
pub enabled: bool,
|
||||
pub email: String,
|
||||
pub name: Option<String>,
|
||||
pub capabilities: Option<Vec<String>>,
|
||||
pub date_since: Option<DateSince>,
|
||||
pub folder_limit: Option<u32>,
|
||||
pub sync_folders: Option<Vec<String>>,
|
||||
pub account_type: AccountType,
|
||||
pub sync_interval_min: Option<i64>,
|
||||
pub known_folders: Option<BTreeSet<String>>,
|
||||
pub created_at: i64,
|
||||
pub updated_at: i64,
|
||||
pub use_proxy: Option<u64>,
|
||||
pub use_dangerous: bool,
|
||||
pub pgp_key: Option<String>,
|
||||
}
|
||||
|
||||
impl AccountV2 {
|
||||
fn pk(&self) -> String {
|
||||
format!("{}_{}", self.created_at, self.id)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[native_model(id = 4, version = 3, from = AccountV2)]
|
||||
#[native_db(primary_key(pk -> String))]
|
||||
pub struct AccountV3 {
|
||||
#[secondary_key(unique)]
|
||||
pub id: u64,
|
||||
pub imap: Option<ImapConfig>,
|
||||
pub enabled: bool,
|
||||
pub email: String,
|
||||
pub name: Option<String>,
|
||||
pub capabilities: Option<Vec<String>>,
|
||||
pub date_since: Option<DateSince>,
|
||||
pub date_before: Option<RelativeDate>,
|
||||
pub folder_limit: Option<u32>,
|
||||
pub sync_folders: Option<Vec<String>>,
|
||||
pub account_type: AccountType,
|
||||
pub sync_interval_min: Option<i64>,
|
||||
pub sync_batch_size: Option<u32>,
|
||||
pub known_folders: Option<BTreeSet<String>>,
|
||||
pub created_at: i64,
|
||||
pub updated_at: i64,
|
||||
pub created_by: u64, //user id
|
||||
pub use_proxy: Option<u64>,
|
||||
pub use_dangerous: bool,
|
||||
pub pgp_key: Option<String>,
|
||||
}
|
||||
|
||||
impl AccountV3 {
|
||||
fn pk(&self) -> String {
|
||||
format!("{}_{}", self.created_at, self.id)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<AccountV1> for AccountV2 {
|
||||
fn from(value: AccountV1) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
imap: value.imap,
|
||||
enabled: value.enabled,
|
||||
email: value.email,
|
||||
name: value.name,
|
||||
capabilities: value.capabilities,
|
||||
date_since: value.date_since,
|
||||
folder_limit: value.folder_limit,
|
||||
sync_folders: value.sync_folders,
|
||||
account_type: value.account_type,
|
||||
sync_interval_min: value.sync_interval_min,
|
||||
known_folders: value.known_folders,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
use_proxy: value.use_proxy,
|
||||
use_dangerous: false,
|
||||
pgp_key: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<AccountV2> for AccountV1 {
|
||||
fn from(value: AccountV2) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
imap: value.imap,
|
||||
enabled: value.enabled,
|
||||
email: value.email,
|
||||
name: value.name,
|
||||
capabilities: value.capabilities,
|
||||
date_since: value.date_since,
|
||||
folder_limit: value.folder_limit,
|
||||
sync_folders: value.sync_folders,
|
||||
account_type: value.account_type,
|
||||
sync_interval_min: value.sync_interval_min,
|
||||
known_folders: value.known_folders,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
use_proxy: value.use_proxy,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<AccountV3> for AccountV2 {
|
||||
fn from(value: AccountV3) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
imap: value.imap,
|
||||
enabled: value.enabled,
|
||||
email: value.email,
|
||||
name: value.name,
|
||||
capabilities: value.capabilities,
|
||||
date_since: value.date_since,
|
||||
folder_limit: value.folder_limit,
|
||||
sync_folders: value.sync_folders,
|
||||
account_type: value.account_type,
|
||||
sync_interval_min: value.sync_interval_min,
|
||||
known_folders: value.known_folders,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
use_proxy: value.use_proxy,
|
||||
use_dangerous: value.use_dangerous,
|
||||
pgp_key: value.pgp_key,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<AccountV2> for AccountV3 {
|
||||
fn from(value: AccountV2) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
imap: value.imap,
|
||||
enabled: value.enabled,
|
||||
email: value.email,
|
||||
name: value.name,
|
||||
capabilities: value.capabilities,
|
||||
date_since: value.date_since,
|
||||
folder_limit: value.folder_limit,
|
||||
sync_folders: value.sync_folders,
|
||||
account_type: value.account_type,
|
||||
sync_interval_min: value.sync_interval_min,
|
||||
known_folders: value.known_folders,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
created_by: DEFAULT_ADMIN_USER_ID,
|
||||
use_proxy: value.use_proxy,
|
||||
use_dangerous: value.use_dangerous,
|
||||
pgp_key: value.pgp_key,
|
||||
sync_batch_size: None,
|
||||
date_before: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<AccountV3> for AccountModel {
|
||||
fn from(value: AccountV3) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
imap: value.imap,
|
||||
enabled: value.enabled,
|
||||
email: value.email,
|
||||
account_name: None,
|
||||
login_name: value.name,
|
||||
capabilities: value.capabilities,
|
||||
date_since: value.date_since,
|
||||
date_before: value.date_before,
|
||||
download_folders: value.sync_folders,
|
||||
account_type: value.account_type,
|
||||
download_interval_min: value.sync_interval_min,
|
||||
download_batch_size: value.sync_batch_size,
|
||||
max_email_size_bytes: None,
|
||||
known_folders: value.known_folders,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
created_by: value.created_by,
|
||||
use_proxy: value.use_proxy,
|
||||
use_dangerous: value.use_dangerous,
|
||||
pgp_key: value.pgp_key,
|
||||
imap_quota_window: None,
|
||||
imap_quota_bytes: None,
|
||||
auto_download_new_mailboxes: None,
|
||||
download_schedule: None,
|
||||
deleting: false,
|
||||
archive_rules: None,
|
||||
extraction_rules: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[native_model(id = 5, version = 1)]
|
||||
#[native_db(primary_key(pk -> String))]
|
||||
pub struct OAuth2 {
|
||||
/// A unique identifier for the OAuth2 configuration.
|
||||
#[secondary_key(unique)]
|
||||
pub id: u64,
|
||||
/// A description of what this configuration is used for.
|
||||
pub description: Option<String>,
|
||||
/// The client ID used for authenticating the application with the OAuth2 provider.
|
||||
pub client_id: String,
|
||||
/// The client secret used in conjunction with the client ID.
|
||||
///
|
||||
/// Users should provide a plaintext secret.
|
||||
/// The server will encrypt it using AES-256-GCM and securely store it.
|
||||
/// The plaintext secret is never stored, so users must ensure it is valid for OAuth2 authentication.
|
||||
pub client_secret: String,
|
||||
/// The URL to redirect users to for OAuth2 authorization.
|
||||
pub auth_url: String,
|
||||
/// The URL to exchange authorization codes for access tokens.
|
||||
pub token_url: String,
|
||||
/// The URI where the OAuth2 provider will redirect to after authorization.
|
||||
pub redirect_uri: String,
|
||||
/// The scopes of access that are being requested (e.g., email, profile).
|
||||
pub scopes: Option<Vec<String>>,
|
||||
/// Any additional parameters to include in the OAuth2 requests (e.g., access_type, prompt).
|
||||
pub extra_params: Option<BTreeMap<String, String>>,
|
||||
/// Indicates whether this configuration is enabled or disabled.
|
||||
pub enabled: bool,
|
||||
/// route OAuth through proxy (when direct access is blocked)
|
||||
pub use_proxy: Option<u64>,
|
||||
/// The timestamp when the configuration was created, in milliseconds since the Unix epoch.
|
||||
pub created_at: i64,
|
||||
/// The timestamp when the configuration was last updated, in milliseconds since the Unix epoch.
|
||||
pub updated_at: i64,
|
||||
}
|
||||
|
||||
impl OAuth2 {
|
||||
fn pk(&self) -> String {
|
||||
format!("{}_{}", &self.created_at, &self.id)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<OAuth2> for bichon_core::oauth2::entity::OAuth2 {
|
||||
fn from(value: OAuth2) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
description: value.description,
|
||||
client_id: value.client_id,
|
||||
client_secret: value.client_secret,
|
||||
auth_url: value.auth_url,
|
||||
token_url: value.token_url,
|
||||
redirect_uri: value.redirect_uri,
|
||||
scopes: value.scopes,
|
||||
extra_params: value.extra_params,
|
||||
enabled: value.enabled,
|
||||
use_proxy: value.use_proxy,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[native_model(id = 6, version = 1)]
|
||||
#[native_db]
|
||||
pub struct OAuth2PendingEntity {
|
||||
/// Unique identifier for the OAuth2 request record
|
||||
pub oauth2_id: u64,
|
||||
|
||||
pub account_id: u64,
|
||||
/// CSRF protection state parameter used to verify the integrity of the authorization request
|
||||
#[primary_key]
|
||||
pub state: String,
|
||||
|
||||
/// PKCE code verifier used in the authorization code exchange process to ensure security
|
||||
pub code_verifier: String,
|
||||
|
||||
/// Timestamp when the OAuth2 request was created, used to determine request expiration
|
||||
pub created_at: i64,
|
||||
}
|
||||
|
||||
impl From<OAuth2PendingEntity> for bichon_core::oauth2::pending::OAuth2PendingEntity {
|
||||
fn from(value: OAuth2PendingEntity) -> Self {
|
||||
Self {
|
||||
oauth2_id: value.oauth2_id,
|
||||
account_id: value.account_id,
|
||||
state: value.state,
|
||||
code_verifier: value.code_verifier,
|
||||
created_at: value.created_at,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[native_model(id = 7, version = 1)]
|
||||
#[native_db]
|
||||
pub struct OAuth2AccessToken {
|
||||
/// The ID of the account associated with this access token.
|
||||
#[primary_key]
|
||||
pub account_id: u64,
|
||||
/// The id of the OAuth2 configuration associated with this access token.
|
||||
#[secondary_key]
|
||||
pub oauth2_id: u64,
|
||||
/// The OAuth2 access token used to authenticate requests to the provider.
|
||||
pub access_token: Option<String>,
|
||||
/// The OAuth2 refresh token used to obtain new access tokens.
|
||||
pub refresh_token: Option<String>,
|
||||
/// The timestamp when the token record was created, in milliseconds since the Unix epoch.
|
||||
pub created_at: i64,
|
||||
/// The timestamp when the token record was last updated, in milliseconds since the Unix epoch.
|
||||
pub updated_at: i64,
|
||||
}
|
||||
|
||||
impl From<OAuth2AccessToken> for bichon_core::oauth2::token::OAuth2AccessToken {
|
||||
fn from(value: OAuth2AccessToken) -> Self {
|
||||
Self {
|
||||
account_id: value.account_id,
|
||||
oauth2_id: value.oauth2_id,
|
||||
access_token: value.access_token,
|
||||
refresh_token: value.refresh_token,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[native_model(id = 8, version = 1)]
|
||||
#[native_db]
|
||||
pub struct Proxy {
|
||||
/// The unique identifier for this proxy configuration.
|
||||
#[primary_key]
|
||||
pub id: u64,
|
||||
|
||||
/// The proxy URL (e.g., socks5://127.0.0.1:1080) used to route network requests.
|
||||
pub url: String,
|
||||
|
||||
/// The creation timestamp of this record, represented as milliseconds since the Unix epoch.
|
||||
pub created_at: i64,
|
||||
|
||||
/// The last update timestamp of this record, represented as milliseconds since the Unix epoch.
|
||||
pub updated_at: i64,
|
||||
}
|
||||
|
||||
impl From<Proxy> for bichon_core::settings::proxy::Proxy {
|
||||
fn from(value: Proxy) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
url: value.url,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[native_model(id = 9, version = 1)]
|
||||
#[native_db]
|
||||
pub struct UserRole {
|
||||
#[primary_key]
|
||||
pub id: u64,
|
||||
pub name: String,
|
||||
pub description: Option<String>,
|
||||
pub permissions: BTreeSet<String>,
|
||||
pub is_builtin: bool,
|
||||
pub created_at: i64,
|
||||
pub role_type: RoleType,
|
||||
pub updated_at: i64,
|
||||
}
|
||||
|
||||
impl From<UserRole> for bichon_core::users::role::UserRole {
|
||||
fn from(value: UserRole) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
name: value.name,
|
||||
description: value.description,
|
||||
permissions: value.permissions,
|
||||
is_builtin: value.is_builtin,
|
||||
created_at: value.created_at,
|
||||
role_type: value.role_type,
|
||||
updated_at: value.updated_at,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[native_model(id = 10, version = 1)]
|
||||
#[native_db]
|
||||
pub struct BichonUser {
|
||||
#[primary_key]
|
||||
pub id: u64,
|
||||
#[secondary_key(unique)]
|
||||
pub username: String,
|
||||
#[secondary_key(unique)]
|
||||
pub email: String,
|
||||
|
||||
pub password: Option<String>,
|
||||
|
||||
/// Scoped Access: Defines per-account permissions.
|
||||
/// Example:
|
||||
/// { account_id: 1, role_id: role_manager_id } -> Manager on Account 1
|
||||
/// { account_id: 2, role_id: role_viewer_id } -> Viewer on Account 2
|
||||
pub account_access_map: BTreeMap<u64, u64>,
|
||||
|
||||
pub description: Option<String>,
|
||||
|
||||
/// System Roles: Permissions that apply to the whole system
|
||||
/// (e.g., system settings, creating new users).
|
||||
pub global_roles: Vec<u64>,
|
||||
|
||||
pub avatar: Option<String>,
|
||||
pub created_at: i64,
|
||||
pub updated_at: i64,
|
||||
/// Optional access control settings
|
||||
pub acl: Option<AccessControl>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[native_model(id = 10, version = 2, from = BichonUser)]
|
||||
#[native_db]
|
||||
pub struct BichonUserV2 {
|
||||
#[primary_key]
|
||||
pub id: u64,
|
||||
#[secondary_key(unique)]
|
||||
pub username: String,
|
||||
#[secondary_key(unique)]
|
||||
pub email: String,
|
||||
|
||||
pub password: Option<String>,
|
||||
|
||||
/// Scoped Access: Defines per-account permissions.
|
||||
/// Example:
|
||||
/// { account_id: 1, role_id: role_manager_id } -> Manager on Account 1
|
||||
/// { account_id: 2, role_id: role_viewer_id } -> Viewer on Account 2
|
||||
pub account_access_map: BTreeMap<u64, u64>,
|
||||
|
||||
pub description: Option<String>,
|
||||
|
||||
/// System Roles: Permissions that apply to the whole system
|
||||
/// (e.g., system settings, creating new users).
|
||||
pub global_roles: Vec<u64>,
|
||||
|
||||
pub avatar: Option<String>,
|
||||
pub created_at: i64,
|
||||
pub updated_at: i64,
|
||||
/// Optional access control settings
|
||||
pub acl: Option<AccessControl>,
|
||||
|
||||
pub theme: Option<String>,
|
||||
pub language: Option<String>,
|
||||
}
|
||||
|
||||
impl From<BichonUserV2> for BichonUser {
|
||||
fn from(value: BichonUserV2) -> Self {
|
||||
BichonUser {
|
||||
id: value.id,
|
||||
username: value.username,
|
||||
email: value.email,
|
||||
password: value.password,
|
||||
account_access_map: value.account_access_map,
|
||||
description: value.description,
|
||||
global_roles: value.global_roles,
|
||||
avatar: value.avatar,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
acl: value.acl,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<BichonUser> for BichonUserV2 {
|
||||
fn from(value: BichonUser) -> Self {
|
||||
BichonUserV2 {
|
||||
id: value.id,
|
||||
username: value.username,
|
||||
email: value.email,
|
||||
password: value.password,
|
||||
account_access_map: value.account_access_map,
|
||||
description: value.description,
|
||||
global_roles: value.global_roles,
|
||||
avatar: value.avatar,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
acl: value.acl,
|
||||
theme: None,
|
||||
language: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<BichonUserV2> for bichon_core::users::BichonUserV2 {
|
||||
fn from(value: BichonUserV2) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
username: value.username,
|
||||
email: value.email,
|
||||
password: value.password,
|
||||
account_access_map: value.account_access_map,
|
||||
description: value.description,
|
||||
global_roles: value.global_roles,
|
||||
avatar: value.avatar,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
acl: value.acl,
|
||||
theme: value.theme,
|
||||
language: value.language,
|
||||
sso_id: None,
|
||||
sso_provider: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq, Deserialize, Serialize)]
|
||||
#[native_model(id = 11, version = 1)]
|
||||
#[native_db]
|
||||
pub struct AccessTokenModel {
|
||||
/// The ID of the user who owns this token
|
||||
#[secondary_key]
|
||||
pub user_id: u64,
|
||||
/// The unique token string used for authentication
|
||||
#[primary_key]
|
||||
pub token: String,
|
||||
/// An optional name of the token.
|
||||
pub name: Option<String>,
|
||||
/// Token type: WebUI or API
|
||||
pub token_type: TokenType,
|
||||
/// The timestamp (in milliseconds since epoch) when the token was created.
|
||||
pub created_at: i64,
|
||||
/// The timestamp (in milliseconds since epoch) when the token was last updated.
|
||||
pub updated_at: i64,
|
||||
/// The timestamp (in milliseconds since epoch) when the token expires.
|
||||
/// None means the token does not expire (this applies only to API tokens).
|
||||
pub expire_at: Option<i64>,
|
||||
/// The timestamp (in milliseconds since epoch) when the token was last used.
|
||||
pub last_access_at: i64,
|
||||
}
|
||||
|
||||
impl From<AccessTokenModel> for bichon_core::token::AccessTokenModel {
|
||||
fn from(value: AccessTokenModel) -> Self {
|
||||
Self {
|
||||
user_id: value.user_id,
|
||||
token: value.token,
|
||||
name: value.name,
|
||||
token_type: value.token_type,
|
||||
created_at: value.created_at,
|
||||
updated_at: value.updated_at,
|
||||
expire_at: value.expire_at,
|
||||
last_access_at: value.last_access_at,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[native_model(id = 1, version = 1)]
|
||||
#[native_db]
|
||||
pub struct MailBox {
|
||||
/// The unique identifier for the mailbox
|
||||
#[primary_key]
|
||||
pub id: u64,
|
||||
/// The ID of the account associated with the mailbox
|
||||
#[secondary_key]
|
||||
pub account_id: u64,
|
||||
/// The unique, decoded, human-readable name of the mailbox (e.g., "INBOX", "Sent Items").
|
||||
/// This is the decoded name as presented to users, derived from the IMAP server's mailbox name
|
||||
/// (e.g., after decoding UTF-7 or other encodings per RFC 3501).
|
||||
pub name: String,
|
||||
/// Optional delimiter used to separate mailbox names in a hierarchy (e.g., "/" or ".").
|
||||
/// Used in IMAP to structure nested mailboxes (e.g., "INBOX/Archive").
|
||||
pub delimiter: Option<String>,
|
||||
/// List of attributes associated with the mailbox (e.g., `\NoSelect`, `\Deleted`).
|
||||
/// These indicate special properties, such as whether the mailbox can hold messages.
|
||||
pub attributes: Vec<Attribute>,
|
||||
/// The number of messages that currently exist in the mailbox.
|
||||
pub exists: u32,
|
||||
/// Optional number of unseen messages in the mailbox (i.e., messages without the `\Seen` flag).
|
||||
pub unseen: Option<u32>,
|
||||
/// The next unique identifier (UID) that will be assigned to a new message in the mailbox.
|
||||
/// If `None`, the IMAP server has not provided this information.
|
||||
pub uid_next: Option<u32>,
|
||||
/// The validity identifier for UIDs in this mailbox, used to ensure UID consistency across sessions.
|
||||
/// If `None`, the IMAP server has not provided this information.
|
||||
pub uid_validity: Option<u32>,
|
||||
}
|
||||
|
||||
impl From<MailBox> for bichon_core::cache::imap::mailbox::MailBox {
|
||||
fn from(value: MailBox) -> Self {
|
||||
Self {
|
||||
id: value.id,
|
||||
account_id: value.account_id,
|
||||
name: value.name,
|
||||
delimiter: value.delimiter,
|
||||
attributes: value.attributes,
|
||||
exists: value.exists,
|
||||
unseen: value.unseen,
|
||||
uid_next: value.uid_next,
|
||||
uid_validity: value.uid_validity,
|
||||
highest_uid: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub static META_MODELS: LazyLock<Models> = LazyLock::new(|| {
|
||||
let mut adapter = ModelsAdapter::new();
|
||||
adapter.register_metadata_models();
|
||||
adapter.models
|
||||
});
|
||||
|
||||
pub static MAILBOX_MODELS: LazyLock<Models> = LazyLock::new(|| {
|
||||
let mut adapter = ModelsAdapter::new();
|
||||
adapter.register_model::<MailBox>();
|
||||
adapter.models
|
||||
});
|
||||
|
||||
pub struct ModelsAdapter {
|
||||
pub models: Models,
|
||||
}
|
||||
|
||||
impl ModelsAdapter {
|
||||
pub fn new() -> Self {
|
||||
ModelsAdapter {
|
||||
models: Models::new(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn register_model<T: ToInput>(&mut self) {
|
||||
self.models.define::<T>().expect("failed to define model ");
|
||||
}
|
||||
|
||||
pub fn register_metadata_models(&mut self) {
|
||||
self.register_model::<CachedMailSettings>();
|
||||
self.register_model::<AccountV1>();
|
||||
self.register_model::<AccountV2>();
|
||||
self.register_model::<AccountV3>();
|
||||
self.register_model::<OAuth2>();
|
||||
self.register_model::<OAuth2PendingEntity>();
|
||||
self.register_model::<OAuth2AccessToken>();
|
||||
self.register_model::<Proxy>();
|
||||
self.register_model::<UserRole>();
|
||||
self.register_model::<BichonUser>();
|
||||
self.register_model::<BichonUserV2>();
|
||||
self.register_model::<AccessTokenModel>();
|
||||
}
|
||||
}
|
||||
|
||||
fn init_meta_database(root_path: &PathBuf) -> BichonResult<Arc<Database<'static>>> {
|
||||
let mut database = Builder::new()
|
||||
.set_cache_size(134217728)
|
||||
.create(&META_MODELS, root_path.join("meta.db"))
|
||||
.map_err(handle_database_error)?;
|
||||
|
||||
let rw = database
|
||||
.rw_transaction()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
rw.migrate::<AccountV3>()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
rw.migrate::<BichonUserV2>()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
rw.commit()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
database
|
||||
.compact()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
Ok(Arc::new(database))
|
||||
}
|
||||
|
||||
fn init_evenlope_database(root_path: &PathBuf) -> BichonResult<Arc<Database<'static>>> {
|
||||
let mut database = Builder::new()
|
||||
.set_cache_size(1073741824)
|
||||
.create(&MAILBOX_MODELS, root_path.join("mailbox.db"))
|
||||
.map_err(handle_database_error)?;
|
||||
|
||||
let rw = database
|
||||
.rw_transaction()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
rw.commit()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
database
|
||||
.compact()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
Ok(Arc::new(database))
|
||||
}
|
||||
|
||||
fn handle_database_error(error: native_db::db_type::Error) -> BichonError {
|
||||
raise_error!(
|
||||
format!("Failed to create database: {:?}", error),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
}
|
||||
|
||||
pub fn list_all_impl<T: ToInput + Clone + Send + 'static>(
|
||||
database: &Arc<Database<'static>>,
|
||||
) -> BichonResult<Vec<T>> {
|
||||
let r_transaction = database
|
||||
.r_transaction()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
let entities: Vec<T> = r_transaction
|
||||
.scan()
|
||||
.primary()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
.all()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
.try_collect()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
Ok(entities)
|
||||
}
|
||||
|
||||
pub fn migrate_metadata(root_path: &PathBuf) -> Result<(), Box<dyn std::error::Error>> {
|
||||
// Pre-flight: verify old metadata databases exist
|
||||
let meta_db_path = root_path.join("meta.db");
|
||||
if !meta_db_path.exists() {
|
||||
return Err(format!(
|
||||
"Legacy metadata database not found at '{}'. \
|
||||
Make sure the root directory points to a valid v0.3.7 installation.",
|
||||
meta_db_path.display()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
let mailbox_db_path = root_path.join("mailbox.db");
|
||||
if !mailbox_db_path.exists() {
|
||||
return Err(format!(
|
||||
"Legacy mailbox database not found at '{}'. \
|
||||
Make sure the root directory points to a valid v0.3.7 installation.",
|
||||
mailbox_db_path.display()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
|
||||
// Initialize legacy database connections
|
||||
let meta_db = init_meta_database(root_path)
|
||||
.map_err(|e| format!("Failed to initialize legacy metadata database: {}", e))?;
|
||||
let envelope_db = init_evenlope_database(root_path)
|
||||
.map_err(|e| format!("Failed to initialize legacy envelope database: {}", e))?;
|
||||
|
||||
// Prepare new database directory
|
||||
let db_path = root_path.join("memdb");
|
||||
if !db_path.exists() {
|
||||
std::fs::create_dir_all(&db_path)?;
|
||||
}
|
||||
|
||||
// Open new database (disable full durability for faster bulk writes)
|
||||
let db = MemDb::open_with(&db_path, Durability::Off)
|
||||
.map_err(|e| format!("Failed to open new memdb database: {}", e))?;
|
||||
|
||||
println!(
|
||||
"{}",
|
||||
style("Step 1: Migrating Metadata Entities...")
|
||||
.bold()
|
||||
.cyan()
|
||||
);
|
||||
|
||||
// Migration helper macro to reduce boilerplate
|
||||
macro_rules! migrate_collection {
|
||||
($name:expr, $old_type:ty, $new_type:ty, $source_db:expr) => {
|
||||
print!(" > {:<25} ", $name);
|
||||
let items = list_all_impl::<$old_type>($source_db)?;
|
||||
let count = items.len();
|
||||
let converted: Vec<$new_type> = items.into_iter().map(|a| a.into()).collect();
|
||||
batch_insert_impl(&db, converted)?;
|
||||
println!("{} ({} items)", style("done").green(), count);
|
||||
};
|
||||
}
|
||||
|
||||
// --- Migrate each entity type ---
|
||||
|
||||
migrate_collection!(
|
||||
"Mail Settings",
|
||||
CachedMailSettings,
|
||||
bichon_core::autoconfig::CachedMailSettings,
|
||||
&meta_db
|
||||
);
|
||||
|
||||
migrate_collection!("Accounts", AccountV3, AccountModel, &meta_db);
|
||||
|
||||
migrate_collection!(
|
||||
"OAuth2 Entities",
|
||||
OAuth2,
|
||||
bichon_core::oauth2::entity::OAuth2,
|
||||
&meta_db
|
||||
);
|
||||
|
||||
migrate_collection!(
|
||||
"OAuth2 Pending",
|
||||
OAuth2PendingEntity,
|
||||
bichon_core::oauth2::pending::OAuth2PendingEntity,
|
||||
&meta_db
|
||||
);
|
||||
|
||||
migrate_collection!(
|
||||
"OAuth2 Access Tokens",
|
||||
OAuth2AccessToken,
|
||||
bichon_core::oauth2::token::OAuth2AccessToken,
|
||||
&meta_db
|
||||
);
|
||||
|
||||
migrate_collection!(
|
||||
"Proxy Settings",
|
||||
Proxy,
|
||||
bichon_core::settings::proxy::Proxy,
|
||||
&meta_db
|
||||
);
|
||||
|
||||
migrate_collection!(
|
||||
"User Roles",
|
||||
UserRole,
|
||||
bichon_core::users::role::UserRole,
|
||||
&meta_db
|
||||
);
|
||||
|
||||
migrate_collection!(
|
||||
"Users",
|
||||
BichonUserV2,
|
||||
bichon_core::users::BichonUserV2,
|
||||
&meta_db
|
||||
);
|
||||
|
||||
migrate_collection!(
|
||||
"Access Tokens",
|
||||
AccessTokenModel,
|
||||
bichon_core::token::AccessTokenModel,
|
||||
&meta_db
|
||||
);
|
||||
|
||||
// Mailboxes (from envelope_db)
|
||||
migrate_collection!(
|
||||
"Mailboxes",
|
||||
MailBox,
|
||||
bichon_core::cache::imap::mailbox::MailBox,
|
||||
&envelope_db
|
||||
);
|
||||
|
||||
// Persist and finish
|
||||
db.snapshot()
|
||||
.map_err(|e| format!("Snapshot save failed: {}", e))?;
|
||||
println!(
|
||||
"{}",
|
||||
style("Metadata migration completed successfully.")
|
||||
.green()
|
||||
.bold()
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
449
crates/admin/src/migrate.rs
Normal file
449
crates/admin/src/migrate.rs
Normal file
@@ -0,0 +1,449 @@
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use bichon_core::migrate::{
|
||||
count_eml_segments, do_migrate_segment, is_tantivy_index_dir,
|
||||
store::{LegacyDirs, NewDirs, NewIndexWriter},
|
||||
};
|
||||
use console::style;
|
||||
use dialoguer::{theme::ColorfulTheme, Confirm, Input};
|
||||
use indicatif::{ProgressBar, ProgressStyle};
|
||||
|
||||
pub fn handle_migration(theme: &ColorfulTheme) {
|
||||
println!(
|
||||
"\n{}",
|
||||
style("MIGRATION: Bichon v0.3.7 Storage Architecture → v1.x")
|
||||
.bold()
|
||||
.yellow()
|
||||
);
|
||||
|
||||
println!(
|
||||
"{}",
|
||||
style(
|
||||
"This tool migrates data from the legacy v0.3.7 Tantivy-based storage \
|
||||
architecture to the new v1.x \
|
||||
separated index and Fjall-backed storage format."
|
||||
)
|
||||
.dim()
|
||||
);
|
||||
|
||||
println!(
|
||||
"{}",
|
||||
style(
|
||||
"Legacy v0.3.7 architecture:\n\
|
||||
• envelope metadata stored in Tantivy\n\
|
||||
• message data stored in Tantivy\n\n\
|
||||
New v1.x architecture:\n\
|
||||
• mail indexes stored in Tantivy\n\
|
||||
• attachment indexes stored in Tantivy\n\
|
||||
• raw message data stored in Fjall\n\
|
||||
• attachment blobs stored in Fjall"
|
||||
)
|
||||
.dim()
|
||||
);
|
||||
|
||||
println!(
|
||||
"\n{} {}",
|
||||
style("IMPORTANT:").yellow().bold(),
|
||||
style(
|
||||
"The paths below must exactly match what your old bichon server was configured with."
|
||||
)
|
||||
.yellow()
|
||||
);
|
||||
|
||||
// --- bichon-root-dir ---
|
||||
let root_dir_str: String = Input::with_theme(theme)
|
||||
.with_prompt("Enter --bichon-root-dir (same value used by the old server)")
|
||||
.validate_with(|input: &String| -> Result<(), &str> {
|
||||
let path = Path::new(input);
|
||||
if !path.is_absolute() {
|
||||
return Err("Path must be absolute.");
|
||||
}
|
||||
if !path.exists() {
|
||||
return Err("Directory does not exist.");
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
let root_path = PathBuf::from(&root_dir_str);
|
||||
|
||||
// --- bichon-index-dir ---
|
||||
let default_index = root_path.join("envelope");
|
||||
let default_new_index = root_path.join("bichon-indices");
|
||||
let index_dir_str: String = Input::with_theme(theme)
|
||||
.with_prompt(format!(
|
||||
"Enter --bichon-index-dir (leave blank to use default: {})",
|
||||
style(default_index.display()).cyan()
|
||||
))
|
||||
.allow_empty(true)
|
||||
.validate_with(|input: &String| -> Result<(), &str> {
|
||||
if input.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
let path = Path::new(input);
|
||||
if !path.is_absolute() {
|
||||
return Err("Path must be absolute.");
|
||||
}
|
||||
|
||||
if !path.exists() {
|
||||
return Err("Directory does not exist.");
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
let index_path = if index_dir_str.is_empty() {
|
||||
default_index
|
||||
} else {
|
||||
PathBuf::from(&index_dir_str)
|
||||
};
|
||||
|
||||
let new_index_path = if index_dir_str.is_empty() {
|
||||
default_new_index
|
||||
} else {
|
||||
PathBuf::from(&index_dir_str).join("bichon-indices")
|
||||
};
|
||||
|
||||
// --- bichon-data-dir ---
|
||||
let default_data = root_path.join("eml");
|
||||
let default_new_data = root_path.join("bichon-storage");
|
||||
let data_dir_str: String = Input::with_theme(theme)
|
||||
.with_prompt(format!(
|
||||
"Enter --bichon-data-dir (leave blank to use default: {})",
|
||||
style(default_data.display()).cyan()
|
||||
))
|
||||
.allow_empty(true)
|
||||
.validate_with(|input: &String| -> Result<(), &str> {
|
||||
if input.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
let path = Path::new(input);
|
||||
if !path.is_absolute() {
|
||||
return Err("Path must be absolute.");
|
||||
}
|
||||
if !path.exists() {
|
||||
return Err("Directory does not exist.");
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
let data_path = if data_dir_str.is_empty() {
|
||||
default_data
|
||||
} else {
|
||||
PathBuf::from(&data_dir_str)
|
||||
};
|
||||
|
||||
let new_data_path = if data_dir_str.is_empty() {
|
||||
default_new_data
|
||||
} else {
|
||||
PathBuf::from(&data_dir_str).join("bichon-storage")
|
||||
};
|
||||
|
||||
println!("\n{}", style("Paths to be migrated:").bold());
|
||||
println!("----------------------------------------");
|
||||
println!(
|
||||
"{:<20} : {}",
|
||||
"bichon-root-dir",
|
||||
style(root_path.display()).cyan()
|
||||
);
|
||||
println!(
|
||||
"{:<20} : {}",
|
||||
"bichon-index-dir",
|
||||
style(index_path.display()).cyan()
|
||||
);
|
||||
println!(
|
||||
"{:<20} : {}",
|
||||
"bichon-data-dir",
|
||||
style(data_path.display()).cyan()
|
||||
);
|
||||
println!("----------------------------------------");
|
||||
|
||||
println!(
|
||||
"\n{} Checking legacy v0.3.7 storage layout...",
|
||||
style("⌛").yellow()
|
||||
);
|
||||
|
||||
match is_legacy_data_layout_with_paths(&index_path, &data_path) {
|
||||
Ok(true) => {
|
||||
println!(
|
||||
"{} {}",
|
||||
style("✔").green(),
|
||||
style("Legacy v0.3.7 Tantivy-based storage detected. Migration to v1.x is required.")
|
||||
.yellow()
|
||||
);
|
||||
}
|
||||
Ok(false) => {
|
||||
println!(
|
||||
"{} {}",
|
||||
style("✔").green(),
|
||||
style("No legacy v0.3.7 storage layout was detected at the specified paths.").green()
|
||||
);
|
||||
|
||||
println!(
|
||||
"{}",
|
||||
style(
|
||||
"The selected directories may already be using the v1.x storage architecture."
|
||||
)
|
||||
.dim()
|
||||
);
|
||||
|
||||
return;
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
"{} Failed to verify legacy storage layout: {:?}",
|
||||
style("ERROR:").red().bold(),
|
||||
e
|
||||
);
|
||||
|
||||
std::process::exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
println!(
|
||||
"\n{} {}",
|
||||
style("⚠").yellow(),
|
||||
style(
|
||||
"This migration is non-destructive. Existing v0.x storage files will remain unchanged."
|
||||
)
|
||||
.yellow()
|
||||
);
|
||||
|
||||
if !Confirm::with_theme(theme)
|
||||
.with_prompt("Ready to migrate?")
|
||||
.default(true)
|
||||
.interact()
|
||||
.unwrap()
|
||||
{
|
||||
println!("{}", style("Migration cancelled.").dim());
|
||||
return;
|
||||
}
|
||||
|
||||
// Step 1: Migrate metadata (meta.db + mailbox.db → memdb)
|
||||
match crate::meta::migrate_metadata(&root_path) {
|
||||
Ok(()) => {}
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
"\n{} Metadata migration failed:\n{}",
|
||||
style("✘").red().bold(),
|
||||
style(e).red()
|
||||
);
|
||||
eprintln!(
|
||||
"{}",
|
||||
style("Aborting migration. No changes have been made to Tantivy data.").yellow()
|
||||
);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
println!(
|
||||
"\n{} {}",
|
||||
style("⌛").yellow(),
|
||||
style("Step 2: Migrating email index and blob data...").cyan()
|
||||
);
|
||||
|
||||
println!(
|
||||
"\n{} {}",
|
||||
style("ℹ").blue(),
|
||||
style("Batch size controls memory usage during migration:").dim()
|
||||
);
|
||||
println!(
|
||||
" {} 1000 — ~500MB RAM (slower, low memory)",
|
||||
style("•").dim()
|
||||
);
|
||||
println!(" {} 3000 — ~1GB RAM (recommended)", style("•").dim());
|
||||
println!(
|
||||
" {} 5000 — ~2GB RAM (faster, high memory)",
|
||||
style("•").dim()
|
||||
);
|
||||
println!(
|
||||
" {} Note: actual memory usage depends on your average email size.",
|
||||
style("•").yellow()
|
||||
);
|
||||
println!(
|
||||
" {} If your mailbox contains many large attachments, use a smaller batch size.\n",
|
||||
style(" ").dim()
|
||||
);
|
||||
|
||||
let batch_size: u32 = {
|
||||
let input: String = Input::with_theme(&ColorfulTheme::default())
|
||||
.with_prompt("Enter batch size (affects memory usage, see notes above)")
|
||||
.default("3000".to_string())
|
||||
.validate_with(|s: &String| match s.trim().parse::<usize>() {
|
||||
Ok(n) if n > 0 => Ok(()),
|
||||
_ => Err("Please enter a valid positive number"),
|
||||
})
|
||||
.interact_text()
|
||||
.unwrap_or("3000".to_string());
|
||||
input.trim().parse::<u32>().unwrap_or(3000)
|
||||
};
|
||||
|
||||
println!(
|
||||
"{} Using batch size: {}\n",
|
||||
style("✓").green(),
|
||||
style(batch_size).cyan().bold()
|
||||
);
|
||||
|
||||
let legacy = LegacyDirs::new(index_path.clone(), data_path.clone());
|
||||
let total_segments = match count_eml_segments(&legacy) {
|
||||
Ok(n) => n,
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
"\n{} Failed to count EML segments:\n{:?}",
|
||||
style("✘").red().bold(),
|
||||
e
|
||||
);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
if total_segments == 0 {
|
||||
println!(
|
||||
"{} {}",
|
||||
style("✔").green(),
|
||||
style("No EML segments found. Nothing to migrate.").bold()
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
println!(
|
||||
"{} EML segments to migrate: {}",
|
||||
style("⌛").yellow(),
|
||||
style(total_segments).cyan()
|
||||
);
|
||||
|
||||
let pb = ProgressBar::new(total_segments as u64);
|
||||
pb.set_style(
|
||||
ProgressStyle::default_bar()
|
||||
.template(
|
||||
"{spinner:.green} [{elapsed_precise}] [{bar:40.cyan/blue}] {pos}/{len} ({eta}) {msg}",
|
||||
)
|
||||
.unwrap()
|
||||
.progress_chars("#>-"),
|
||||
);
|
||||
|
||||
let mut writer = match NewIndexWriter::open(NewDirs::new(
|
||||
new_index_path.clone(),
|
||||
new_data_path.clone(),
|
||||
)) {
|
||||
Ok(w) => w,
|
||||
Err(e) => {
|
||||
pb.finish_with_message(format!("{}", style("Migration failed.").red()));
|
||||
eprintln!("\n{} {:?}", style("✘").red().bold(), e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let mut grand_total_migrated: usize = 0;
|
||||
let mut grand_total_skipped: usize = 0;
|
||||
|
||||
for seg_idx in 0..total_segments {
|
||||
let seg_total: std::cell::Cell<usize> = std::cell::Cell::new(0);
|
||||
|
||||
pb.set_message(format!("Segment {}/{}", seg_idx + 1, total_segments));
|
||||
let legacy = LegacyDirs::new(index_path.clone(), data_path.clone());
|
||||
match do_migrate_segment(
|
||||
batch_size,
|
||||
legacy,
|
||||
&mut writer,
|
||||
seg_idx,
|
||||
|msg| {
|
||||
if let Some(data) = msg.strip_prefix("TOTAL:") {
|
||||
seg_total.set(data.parse().unwrap_or(0));
|
||||
} else if let Some(data) = msg.strip_prefix("PHASE1:") {
|
||||
let parts: Vec<&str> = data.split('/').collect();
|
||||
let scanned: usize = parts.get(0).and_then(|s| s.parse().ok()).unwrap_or(0);
|
||||
let total: usize = parts
|
||||
.get(1)
|
||||
.and_then(|s| s.split_once(" skipped:").map(|(n, _)| n))
|
||||
.and_then(|s| s.parse().ok())
|
||||
.unwrap_or(0);
|
||||
let skipped: usize = data
|
||||
.split_once("skipped:")
|
||||
.and_then(|(_, s)| s.parse().ok())
|
||||
.unwrap_or(0);
|
||||
let pct = if total > 0 {
|
||||
(scanned * 100) / total
|
||||
} else {
|
||||
0
|
||||
};
|
||||
pb.set_message(format!(
|
||||
"Segment {}/{} [scanning {}/{} skipped:{} {}%]",
|
||||
seg_idx + 1,
|
||||
total_segments,
|
||||
scanned,
|
||||
total,
|
||||
skipped,
|
||||
pct,
|
||||
));
|
||||
} else if let Some(data) = msg.strip_prefix("PROGRESS:") {
|
||||
let parts: Vec<&str> = data.split(':').collect();
|
||||
let migrated: usize = parts.get(0).and_then(|s| s.parse().ok()).unwrap_or(0);
|
||||
let total = seg_total.get();
|
||||
let pct = if total > 0 {
|
||||
(migrated * 100) / total
|
||||
} else {
|
||||
0
|
||||
};
|
||||
pb.set_message(format!(
|
||||
"Segment {}/{} [migrating {}/{} {}%]",
|
||||
seg_idx + 1,
|
||||
total_segments,
|
||||
migrated,
|
||||
total,
|
||||
pct,
|
||||
));
|
||||
} else if let Some(warn) = msg.strip_prefix("WARN:") {
|
||||
pb.println(format!("{} {}", style("⚠").yellow(), warn));
|
||||
} else if let Some(done_data) = msg.strip_prefix("DONE:") {
|
||||
let parts: Vec<&str> = done_data.split(':').collect();
|
||||
let migrated: usize = parts.get(0).and_then(|s| s.parse().ok()).unwrap_or(0);
|
||||
let skipped: usize = parts.get(1).and_then(|s| s.parse().ok()).unwrap_or(0);
|
||||
grand_total_migrated += migrated;
|
||||
grand_total_skipped += skipped;
|
||||
}
|
||||
},
|
||||
) {
|
||||
Ok(()) => {}
|
||||
Err(e) => {
|
||||
pb.finish_with_message(format!("{}", style("Migration failed.").red()));
|
||||
eprintln!("\n{} {:?}", style("✘").red().bold(), e);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
pb.set_position((seg_idx + 1) as u64);
|
||||
}
|
||||
|
||||
pb.set_message(style("Finalizing indexes...").dim().to_string());
|
||||
if let Err(e) = writer.finish_writers() {
|
||||
pb.finish_with_message(format!("{}", style("Migration failed.").red()));
|
||||
eprintln!("\n{} {:?}", style("✘").red().bold(), e);
|
||||
return;
|
||||
}
|
||||
|
||||
pb.finish_with_message(format!(
|
||||
"Migration finished. Total: {}, Skipped: {}",
|
||||
grand_total_migrated, grand_total_skipped
|
||||
));
|
||||
|
||||
println!(
|
||||
"{} {}",
|
||||
style("✔").green(),
|
||||
style("Migration completed successfully!").bold()
|
||||
);
|
||||
}
|
||||
|
||||
pub fn is_legacy_data_layout_with_paths(
|
||||
envelope_dir: &PathBuf,
|
||||
eml_dir: &PathBuf,
|
||||
) -> std::io::Result<bool> {
|
||||
let envelope_result = is_tantivy_index_dir(envelope_dir)?;
|
||||
let eml_result = is_tantivy_index_dir(eml_dir)?;
|
||||
|
||||
Ok(envelope_result || eml_result)
|
||||
}
|
||||
@@ -1,41 +1,14 @@
|
||||
use std::{
|
||||
fs,
|
||||
path::{Path, PathBuf},
|
||||
rc::Rc,
|
||||
};
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use bichon::modules::{
|
||||
cli::admin::meta::{find_admin, init_meta_database, update_admin_password},
|
||||
error::BichonError,
|
||||
use bichon_core::{
|
||||
admin::meta::{find_admin, open_database, update_admin_password},
|
||||
utils::encrypt::internal_decrypt_string,
|
||||
};
|
||||
use console::{style, Emoji};
|
||||
use dialoguer::Confirm;
|
||||
use dialoguer::{theme::ColorfulTheme, Input, Password, Select};
|
||||
use native_db::Database;
|
||||
use dialoguer::{theme::ColorfulTheme, Confirm, Input, Password, Select};
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() {
|
||||
let theme = ColorfulTheme::default();
|
||||
println!(
|
||||
"\n{}\n",
|
||||
style("BICHON ADMINISTRATIVE TOOL").bold().bright().cyan()
|
||||
);
|
||||
|
||||
let main_options = vec!["Reset Admin Password", "Exit"];
|
||||
let selection = Select::with_theme(&theme)
|
||||
.with_prompt("Select an operation")
|
||||
.default(0)
|
||||
.items(&main_options)
|
||||
.interact()
|
||||
.unwrap();
|
||||
|
||||
if selection == 1 {
|
||||
println!("{}", style("Exiting...").dim());
|
||||
return;
|
||||
}
|
||||
|
||||
let root_dir_str: String = Input::with_theme(&theme)
|
||||
pub fn handle_reset_password(theme: &ColorfulTheme) {
|
||||
let root_dir_str: String = Input::with_theme(theme)
|
||||
.with_prompt("Enter the absolute path for 'bichon_root_dir'")
|
||||
.validate_with(|input: &String| -> Result<(), &str> {
|
||||
let path = Path::new(input);
|
||||
@@ -45,9 +18,9 @@ async fn main() {
|
||||
if !path.exists() {
|
||||
return Err("Directory does not exist.");
|
||||
}
|
||||
let has_metadata = path.join("meta.db").exists();
|
||||
if !has_metadata {
|
||||
return Err("Invalid directory: 'meta.db' not found.");
|
||||
let memdb_dir = path.join("memdb");
|
||||
if !memdb_dir.exists() || !memdb_dir.is_dir() {
|
||||
return Err("Invalid directory: 'memdb' data directory not found.");
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
@@ -55,40 +28,14 @@ async fn main() {
|
||||
.unwrap();
|
||||
|
||||
let root_path = PathBuf::from(&root_dir_str);
|
||||
|
||||
let database: Rc<Database<'static>> = match init_meta_database(&root_path.join("meta.db")) {
|
||||
Ok(database) => database,
|
||||
Err(e) => match e {
|
||||
BichonError::Generic {
|
||||
message,
|
||||
location,
|
||||
code,
|
||||
} => {
|
||||
if message.contains("RedbDatabaseError(DatabaseAlreadyOpen") {
|
||||
println!("\n{}", style("ERROR: Database is locked.").red().bold());
|
||||
println!(
|
||||
"{}",
|
||||
style("The Bichon service is likely still running.").yellow()
|
||||
);
|
||||
println!(
|
||||
"Since the database cannot be shared between multiple instances, \n\
|
||||
you must {} the Bichon service before proceeding.",
|
||||
style("STOP").underlined().bold()
|
||||
);
|
||||
std::process::exit(1);
|
||||
} else {
|
||||
eprintln!(
|
||||
"\n{} (Code: {:#?})\nLocation: {}\nMessage: {}",
|
||||
style("A database error occurred:").red().bold(),
|
||||
code,
|
||||
location,
|
||||
message
|
||||
);
|
||||
std::process::exit(1);
|
||||
}
|
||||
}
|
||||
},
|
||||
};
|
||||
let database = open_database(&root_path.join("memdb")).unwrap_or_else(|e| {
|
||||
eprintln!(
|
||||
"\n{} Failed to open database.",
|
||||
style("ERROR:").red().bold()
|
||||
);
|
||||
eprintln!("Details: {:?}", e);
|
||||
std::process::exit(1);
|
||||
});
|
||||
|
||||
let admin = find_admin(&database);
|
||||
|
||||
@@ -124,24 +71,24 @@ async fn main() {
|
||||
"Enter encryption password manually",
|
||||
"Read from password file",
|
||||
];
|
||||
let method = Select::with_theme(&theme)
|
||||
let method = Select::with_theme(theme)
|
||||
.with_prompt("How would you like to provide the database encryption key?")
|
||||
.items(&auth_methods)
|
||||
.interact()
|
||||
.unwrap();
|
||||
|
||||
let raw_key = if method == 0 {
|
||||
Password::with_theme(&theme)
|
||||
Input::with_theme(theme)
|
||||
.with_prompt("Enter Encryption Password")
|
||||
.interact()
|
||||
.unwrap()
|
||||
} else {
|
||||
let file_path: String = Input::with_theme(&theme)
|
||||
let file_path: String = Input::with_theme(theme)
|
||||
.with_prompt("Enter path to encryption password file")
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
match fs::read_to_string(&file_path) {
|
||||
match std::fs::read_to_string(&file_path) {
|
||||
Ok(content) => content.trim().to_string(),
|
||||
Err(e) => {
|
||||
println!("{}: {}", style("Failed to read file").red(), e);
|
||||
@@ -170,7 +117,7 @@ async fn main() {
|
||||
style("BICHON_ENCRYPT_PASSWORD_FILE").green()
|
||||
);
|
||||
|
||||
if Confirm::with_theme(&theme)
|
||||
if Confirm::with_theme(theme)
|
||||
.with_prompt(prompt_message)
|
||||
.default(true)
|
||||
.interact()
|
||||
@@ -216,7 +163,7 @@ async fn main() {
|
||||
println!("{:<12} : {}", "Password", pwd_display);
|
||||
println!("----------------------------------------");
|
||||
|
||||
if !dialoguer::Confirm::with_theme(&theme)
|
||||
if !dialoguer::Confirm::with_theme(theme)
|
||||
.with_prompt(format!(
|
||||
"Do you want to reset the password for '{}'?",
|
||||
user.username
|
||||
@@ -255,13 +202,13 @@ async fn main() {
|
||||
.bold()
|
||||
);
|
||||
|
||||
let new_login_password = Password::with_theme(&theme)
|
||||
let new_login_password = Password::with_theme(theme)
|
||||
.with_prompt("Enter new Admin Login Password")
|
||||
.with_confirmation("Repeat password to confirm", "Passwords do not match!")
|
||||
.interact()
|
||||
.unwrap();
|
||||
|
||||
if !Confirm::with_theme(&theme)
|
||||
if !Confirm::with_theme(theme)
|
||||
.with_prompt("Proceed with database update?")
|
||||
.interact()
|
||||
.unwrap()
|
||||
1105
crates/blob/Cargo.lock
generated
Normal file
1105
crates/blob/Cargo.lock
generated
Normal file
File diff suppressed because it is too large
Load Diff
24
crates/blob/Cargo.toml
Normal file
24
crates/blob/Cargo.toml
Normal file
@@ -0,0 +1,24 @@
|
||||
[package]
|
||||
name = "bichon-blob"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
description = "Embedded KV storage engine for email"
|
||||
|
||||
[dependencies]
|
||||
crc32fast = "1.4"
|
||||
zstd = "0.13"
|
||||
lz4_flex = "0.13.1"
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
bincode = "1"
|
||||
tracing = "0.1"
|
||||
thiserror = "2"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
rand = "0.10.1"
|
||||
criterion = { version = "0.6", features = ["html_reports"] }
|
||||
|
||||
[[bench]]
|
||||
name = "benchmark"
|
||||
harness = false
|
||||
316
crates/blob/benches/benchmark.rs
Normal file
316
crates/blob/benches/benchmark.rs
Normal file
@@ -0,0 +1,316 @@
|
||||
use criterion::{criterion_group, criterion_main, BatchSize, Criterion, Throughput};
|
||||
use std::time::Duration;
|
||||
use tempfile::TempDir;
|
||||
|
||||
use bichon_blob::{Codec, Config, Engine};
|
||||
|
||||
fn make_key(seed: u64) -> [u8; 32] {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..8].copy_from_slice(&seed.to_le_bytes());
|
||||
key
|
||||
}
|
||||
|
||||
fn make_value(size: usize) -> Vec<u8> {
|
||||
let mut v = Vec::with_capacity(size);
|
||||
// Fill with somewhat realistic text-like data so compression works
|
||||
let pattern = b"The quick brown fox jumps over the lazy dog. ";
|
||||
while v.len() < size {
|
||||
let rem = size - v.len();
|
||||
let n = rem.min(pattern.len());
|
||||
v.extend_from_slice(&pattern[..n]);
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
pub fn bench_write_small(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("write");
|
||||
group.throughput(Throughput::Elements(1));
|
||||
group.measurement_time(Duration::from_secs(10));
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
let value = make_value(1024); // 1 KB
|
||||
let mut counter = 0u64;
|
||||
|
||||
group.bench_function("1KB", |b| {
|
||||
b.iter_batched(
|
||||
|| {
|
||||
counter += 1;
|
||||
(make_key(counter), value.clone())
|
||||
},
|
||||
|(key, val)| {
|
||||
engine
|
||||
.write("bench", key, &val, Codec::Zstd)
|
||||
.unwrap()
|
||||
},
|
||||
BatchSize::SmallInput,
|
||||
)
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
pub fn bench_write_medium(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("write");
|
||||
group.throughput(Throughput::Bytes(64 * 1024));
|
||||
group.measurement_time(Duration::from_secs(10));
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
let value = make_value(64 * 1024); // 64 KB
|
||||
let mut counter = 0u64;
|
||||
|
||||
group.bench_function("64KB", |b| {
|
||||
b.iter_batched(
|
||||
|| {
|
||||
counter += 1;
|
||||
(make_key(counter), value.clone())
|
||||
},
|
||||
|(key, val)| {
|
||||
engine
|
||||
.write("bench", key, &val, Codec::Zstd)
|
||||
.unwrap()
|
||||
},
|
||||
BatchSize::SmallInput,
|
||||
)
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
pub fn bench_write_large(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("write");
|
||||
group.throughput(Throughput::Bytes(1024 * 1024));
|
||||
group.measurement_time(Duration::from_secs(15));
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
let value = make_value(1024 * 1024); // 1 MB
|
||||
let mut counter = 0u64;
|
||||
|
||||
group.bench_function("1MB", |b| {
|
||||
b.iter_batched(
|
||||
|| {
|
||||
counter += 1;
|
||||
(make_key(counter), value.clone())
|
||||
},
|
||||
|(key, val)| {
|
||||
engine
|
||||
.write("bench", key, &val, Codec::Zstd)
|
||||
.unwrap()
|
||||
},
|
||||
BatchSize::SmallInput,
|
||||
)
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
pub fn bench_read_cache_hit(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("read");
|
||||
group.throughput(Throughput::Elements(1));
|
||||
group.measurement_time(Duration::from_secs(10));
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
// Pre-populate: 10 keys, all in same bucket → cache hit after first read
|
||||
let value = make_value(4096);
|
||||
for i in 0..10u64 {
|
||||
engine
|
||||
.write("bench", make_key(i), &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let mut counter = 0u64;
|
||||
group.bench_function("cache_hit", |b| {
|
||||
b.iter(|| {
|
||||
let key = make_key(counter % 10);
|
||||
counter += 1;
|
||||
std::hint::black_box(engine.read("bench", &key).unwrap());
|
||||
})
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
pub fn bench_read_cache_miss(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("read");
|
||||
group.throughput(Throughput::Elements(1));
|
||||
group.measurement_time(Duration::from_secs(10));
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut config = Config::default();
|
||||
config.lru_bucket_count = 8; // Small cache to force misses
|
||||
let engine = Engine::open(dir.path(), config).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
let value = make_value(4096);
|
||||
// Write 1000 keys spread across all 16 buckets — small LRU will thrash
|
||||
for i in 0..1000u64 {
|
||||
engine
|
||||
.write("bench", make_key(i), &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let mut counter = 0u64;
|
||||
group.bench_function("cache_miss", |b| {
|
||||
b.iter(|| {
|
||||
let key = make_key(counter % 1000);
|
||||
counter += 1;
|
||||
std::hint::black_box(engine.read("bench", &key).unwrap());
|
||||
})
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
pub fn bench_read_large_value(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("read");
|
||||
group.throughput(Throughput::Bytes(1024 * 1024));
|
||||
group.measurement_time(Duration::from_secs(10));
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
let value = make_value(1024 * 1024); // 1 MB
|
||||
for i in 0..5u64 {
|
||||
engine
|
||||
.write("bench", make_key(i), &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let mut counter = 0u64;
|
||||
group.bench_function("1MB_cache_hit", |b| {
|
||||
b.iter(|| {
|
||||
let key = make_key(counter % 5);
|
||||
counter += 1;
|
||||
std::hint::black_box(engine.read("bench", &key).unwrap());
|
||||
})
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
pub fn bench_delete(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("delete");
|
||||
group.throughput(Throughput::Elements(1));
|
||||
group.measurement_time(Duration::from_secs(10));
|
||||
|
||||
group.bench_function("delete", |b| {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
let value = make_value(4096);
|
||||
let mut counter = 0u64;
|
||||
|
||||
b.iter_batched(
|
||||
|| {
|
||||
counter += 1;
|
||||
let key = make_key(counter);
|
||||
engine
|
||||
.write("bench", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
key
|
||||
},
|
||||
|key| {
|
||||
engine.delete("bench", &key).unwrap();
|
||||
},
|
||||
BatchSize::SmallInput,
|
||||
)
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
pub fn bench_mixed_workload(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("mixed");
|
||||
group.throughput(Throughput::Elements(1));
|
||||
group.measurement_time(Duration::from_secs(15));
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
// Pre-populate with 500 entries
|
||||
let value = make_value(8192);
|
||||
for i in 0..500u64 {
|
||||
engine
|
||||
.write("bench", make_key(i), &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let mut counter: u64 = 500;
|
||||
group.bench_function("80w_15r_5d", |b| {
|
||||
b.iter(|| {
|
||||
counter += 1;
|
||||
let op = counter % 100;
|
||||
match op {
|
||||
0..=79 => {
|
||||
// 80% writes
|
||||
let key = make_key(counter);
|
||||
let val = make_value(4096);
|
||||
engine.write("bench", key, &val, Codec::Zstd).unwrap();
|
||||
}
|
||||
80..=94 => {
|
||||
// 15% reads
|
||||
std::hint::black_box(engine.read("bench", &make_key(counter % 500)).unwrap());
|
||||
}
|
||||
_ => {
|
||||
// 5% deletes
|
||||
if counter % 2 == 0 {
|
||||
let key = make_key(counter % 500);
|
||||
let _ = engine.delete("bench", &key);
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
pub fn bench_gc(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("gc");
|
||||
group.measurement_time(Duration::from_secs(30));
|
||||
group.sample_size(10);
|
||||
|
||||
group.bench_function("gc_30pct_deleted", |b| {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("bench").unwrap();
|
||||
|
||||
// Fill a segment with ~1000 entries, then delete 30%
|
||||
let value = make_value(200_000); // 200KB each → ~1000 entries to fill 256MB
|
||||
let n = 1200u64;
|
||||
for i in 0..n {
|
||||
engine
|
||||
.write("bench", make_key(i), &value, Codec::None)
|
||||
.unwrap();
|
||||
}
|
||||
// Delete ~30%
|
||||
for i in (0..n).step_by(3) {
|
||||
engine.delete("bench", &make_key(i)).unwrap();
|
||||
}
|
||||
|
||||
b.iter(|| {
|
||||
engine.gc("bench").unwrap();
|
||||
})
|
||||
});
|
||||
group.finish();
|
||||
}
|
||||
|
||||
criterion_group!(
|
||||
benches,
|
||||
bench_write_small,
|
||||
bench_write_medium,
|
||||
bench_write_large,
|
||||
bench_read_cache_hit,
|
||||
bench_read_cache_miss,
|
||||
bench_read_large_value,
|
||||
bench_delete,
|
||||
bench_mixed_workload,
|
||||
bench_gc,
|
||||
);
|
||||
criterion_main!(benches);
|
||||
284
crates/blob/src/account.rs
Normal file
284
crates/blob/src/account.rs
Normal file
@@ -0,0 +1,284 @@
|
||||
use std::collections::HashMap;
|
||||
use std::fs;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::{Arc, Mutex, RwLock};
|
||||
|
||||
use crate::bucket::{self, BucketFile, IndexRecord};
|
||||
use crate::error::{Error, Result};
|
||||
use crate::file_pool::FilePool;
|
||||
use crate::meta::{AccountMeta, SegmentStats};
|
||||
use crate::segment::{self, SegmentReader, SegmentWriter};
|
||||
use crate::types::Codec;
|
||||
|
||||
// ── AccountHandle ──────────────────────────────────────────────────────────
|
||||
|
||||
pub struct AccountHandle {
|
||||
id: String,
|
||||
dir: PathBuf,
|
||||
inner: RwLock<AccountInner>,
|
||||
pub(crate) write_mutex: Mutex<()>,
|
||||
file_pool: FilePool,
|
||||
}
|
||||
|
||||
impl AccountHandle {
|
||||
pub fn id(&self) -> &str {
|
||||
&self.id
|
||||
}
|
||||
|
||||
pub fn dir(&self) -> &Path {
|
||||
&self.dir
|
||||
}
|
||||
|
||||
/// Open an existing account.
|
||||
pub fn open(store_root: &Path, account_id: &str) -> Result<Arc<Self>> {
|
||||
let dir = store_root.join("accounts").join(account_id);
|
||||
if !dir.exists() {
|
||||
return Err(Error::AccountNotFound(account_id.to_string()));
|
||||
}
|
||||
let inner = AccountInner::open(&dir)?;
|
||||
Ok(Arc::new(Self {
|
||||
id: account_id.to_string(),
|
||||
dir,
|
||||
inner: RwLock::new(inner),
|
||||
write_mutex: Mutex::new(()),
|
||||
file_pool: FilePool::new(8),
|
||||
}))
|
||||
}
|
||||
|
||||
/// Create a new account.
|
||||
pub fn create(store_root: &Path, account_id: &str) -> Result<Arc<Self>> {
|
||||
let dir = store_root.join("accounts").join(account_id);
|
||||
if dir.exists() {
|
||||
return Err(Error::AccountAlreadyExists(account_id.to_string()));
|
||||
}
|
||||
let inner = AccountInner::create(&dir, account_id)?;
|
||||
Ok(Arc::new(Self {
|
||||
id: account_id.to_string(),
|
||||
dir,
|
||||
inner: RwLock::new(inner),
|
||||
write_mutex: Mutex::new(()),
|
||||
file_pool: FilePool::new(8),
|
||||
}))
|
||||
}
|
||||
|
||||
/// Lock the inner state for reading.
|
||||
pub fn read(&self) -> std::sync::RwLockReadGuard<'_, AccountInner> {
|
||||
self.inner.read().unwrap()
|
||||
}
|
||||
|
||||
/// Lock the inner state for writing.
|
||||
pub fn write(&self) -> std::sync::RwLockWriteGuard<'_, AccountInner> {
|
||||
self.inner.write().unwrap()
|
||||
}
|
||||
|
||||
/// Get a cached file handle for a segment.
|
||||
pub fn get_segment_file(&self, seg_id: u32, path: &Path) -> Result<Arc<Mutex<std::fs::File>>> {
|
||||
self.file_pool.get(seg_id, path)
|
||||
}
|
||||
|
||||
/// Invalidate cached file handles for a segment (after GC).
|
||||
pub fn invalidate_file_cache(&self, seg_id: u32) {
|
||||
self.file_pool.invalidate(seg_id);
|
||||
}
|
||||
}
|
||||
|
||||
// ── AccountInner ───────────────────────────────────────────────────────────
|
||||
|
||||
pub struct AccountInner {
|
||||
dir: PathBuf,
|
||||
meta: AccountMeta,
|
||||
active_writer: SegmentWriter,
|
||||
readers: HashMap<u32, SegmentReader>,
|
||||
}
|
||||
|
||||
impl AccountInner {
|
||||
fn open(dir: &Path) -> Result<Self> {
|
||||
let meta = AccountMeta::load(dir)?;
|
||||
|
||||
let seg_path = dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(meta.active_segment_id));
|
||||
let active_writer = if seg_path.exists() {
|
||||
SegmentWriter::open_append(seg_path, meta.active_segment_id)?
|
||||
} else {
|
||||
fs::create_dir_all(dir.join("segments"))?;
|
||||
SegmentWriter::create(seg_path, meta.active_segment_id)?
|
||||
};
|
||||
|
||||
let mut readers = HashMap::new();
|
||||
for (&seg_id, stats) in &meta.segments {
|
||||
if stats.sealed {
|
||||
let seg_path = dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(seg_id));
|
||||
if seg_path.exists() {
|
||||
readers.insert(seg_id, SegmentReader::open(seg_path, seg_id)?);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(Self {
|
||||
dir: dir.to_path_buf(),
|
||||
meta,
|
||||
active_writer,
|
||||
readers,
|
||||
})
|
||||
}
|
||||
|
||||
fn create(dir: &Path, account_id: &str) -> Result<Self> {
|
||||
fs::create_dir_all(dir.join("segments"))?;
|
||||
BucketFile::ensure_dir(dir)?;
|
||||
|
||||
let meta = AccountMeta::new(account_id.to_string(), 1);
|
||||
|
||||
let seg_path = dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(1));
|
||||
let active_writer = SegmentWriter::create(seg_path, 1)?;
|
||||
|
||||
meta.save(dir)?;
|
||||
|
||||
Ok(Self {
|
||||
dir: dir.to_path_buf(),
|
||||
meta,
|
||||
active_writer,
|
||||
readers: HashMap::new(),
|
||||
})
|
||||
}
|
||||
|
||||
pub fn meta(&self) -> &AccountMeta {
|
||||
&self.meta
|
||||
}
|
||||
|
||||
/// Mark the segment as indexed up to the given offset and persist meta.
|
||||
pub fn mark_indexed(&mut self, segment_id: u32, indexed_up_to_offset: u64) -> Result<()> {
|
||||
if let Some(stats) = self.meta.segments.get_mut(&segment_id) {
|
||||
if indexed_up_to_offset > stats.indexed_up_to_offset {
|
||||
stats.indexed_up_to_offset = indexed_up_to_offset;
|
||||
}
|
||||
}
|
||||
self.meta.save(&self.dir)
|
||||
}
|
||||
|
||||
/// Append an entry without fsync.
|
||||
pub fn append_entry(
|
||||
&mut self,
|
||||
key: [u8; 32],
|
||||
data: &[u8],
|
||||
flags: u8,
|
||||
codec: Codec,
|
||||
) -> Result<(u32, u64, u32)> {
|
||||
if self.active_writer.is_full() {
|
||||
self.seal_active()?;
|
||||
}
|
||||
|
||||
use crate::segment::Entry;
|
||||
let entry = if flags == 1 {
|
||||
Entry::tombstone(key)
|
||||
} else {
|
||||
Entry::new(key, data, flags, codec)
|
||||
};
|
||||
|
||||
let data_size = entry.data.len() as u32;
|
||||
let segment_id = self.active_writer.id();
|
||||
let offset = self.active_writer.append(&entry)?;
|
||||
|
||||
let stats = self
|
||||
.meta
|
||||
.segments
|
||||
.entry(segment_id)
|
||||
.or_insert_with(|| SegmentStats::new(segment_id));
|
||||
stats.total_bytes += data_size as u64;
|
||||
if flags == 1 {
|
||||
stats.deleted_bytes += entry.raw_size as u64;
|
||||
}
|
||||
stats.recompute_ratio();
|
||||
|
||||
Ok((segment_id, offset, data_size))
|
||||
}
|
||||
|
||||
/// Fsync the active segment and persist meta.
|
||||
pub fn flush_active(&mut self) -> Result<()> {
|
||||
self.active_writer.fsync()?;
|
||||
self.meta.save(&self.dir)
|
||||
}
|
||||
|
||||
/// Write an entry with fsync.
|
||||
pub fn write_entry(
|
||||
&mut self,
|
||||
key: [u8; 32],
|
||||
data: &[u8],
|
||||
flags: u8,
|
||||
codec: Codec,
|
||||
) -> Result<(u32, u64, u32)> {
|
||||
let result = self.append_entry(key, data, flags, codec)?;
|
||||
self.flush_active()?;
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn seal_active(&mut self) -> Result<()> {
|
||||
let old_id = self.active_writer.id();
|
||||
let old_stats = self
|
||||
.meta
|
||||
.segments
|
||||
.entry(old_id)
|
||||
.or_insert_with(|| SegmentStats::new(old_id));
|
||||
old_stats.sealed = true;
|
||||
|
||||
let seg_path = self
|
||||
.dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(old_id));
|
||||
self.readers
|
||||
.insert(old_id, SegmentReader::open(seg_path, old_id)?);
|
||||
|
||||
let new_id = old_id + 1;
|
||||
self.meta.active_segment_id = new_id;
|
||||
let new_path = self
|
||||
.dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(new_id));
|
||||
self.active_writer = SegmentWriter::create(new_path, new_id)?;
|
||||
self.meta.save(&self.dir)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Get the on-disk path for a segment.
|
||||
pub fn segment_path(&self, segment_id: u32) -> Result<PathBuf> {
|
||||
let filename = segment::segment_filename(segment_id);
|
||||
let path = self.dir.join("segments").join(&filename);
|
||||
if path.exists() {
|
||||
Ok(path)
|
||||
} else {
|
||||
Err(Error::SegmentNotFound(segment_id))
|
||||
}
|
||||
}
|
||||
|
||||
/// Append index record to the appropriate bucket file.
|
||||
pub fn append_index(&self, record: &IndexRecord) -> Result<()> {
|
||||
let bucket_id = bucket::bucket_id(&record.key);
|
||||
let bf = BucketFile::open(&self.dir, bucket_id);
|
||||
bf.append(record)
|
||||
}
|
||||
|
||||
/// Return list of sealed segment IDs.
|
||||
pub fn sealed_segments(&self) -> Vec<u32> {
|
||||
self.meta
|
||||
.segments
|
||||
.iter()
|
||||
.filter(|(_, s)| s.sealed)
|
||||
.map(|(id, _)| *id)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// All segment IDs (including active).
|
||||
pub fn all_segment_ids(&self) -> Vec<u32> {
|
||||
let mut ids: Vec<u32> = self.meta.segments.keys().copied().collect();
|
||||
if !ids.contains(&self.meta.active_segment_id) {
|
||||
ids.push(self.meta.active_segment_id);
|
||||
}
|
||||
ids.sort_unstable();
|
||||
ids
|
||||
}
|
||||
}
|
||||
329
crates/blob/src/bucket.rs
Normal file
329
crates/blob/src/bucket.rs
Normal file
@@ -0,0 +1,329 @@
|
||||
use std::fs::OpenOptions;
|
||||
use std::io::Write;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::error::Result;
|
||||
use crate::types::{BUCKET_COUNT, INDEX_RECORD_SIZE};
|
||||
|
||||
/// On-disk format: 52 bytes per record.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct IndexRecord {
|
||||
pub key: [u8; 32],
|
||||
pub segment_id: u32,
|
||||
pub offset: u64,
|
||||
pub data_size: u32,
|
||||
pub flags: u8,
|
||||
}
|
||||
|
||||
impl IndexRecord {
|
||||
pub fn new(key: [u8; 32], segment_id: u32, offset: u64, data_size: u32, flags: u8) -> Self {
|
||||
Self {
|
||||
key,
|
||||
segment_id,
|
||||
offset,
|
||||
data_size,
|
||||
flags,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_tombstone(&self) -> bool {
|
||||
self.flags == 1
|
||||
}
|
||||
|
||||
pub fn encode(&self) -> [u8; INDEX_RECORD_SIZE] {
|
||||
let mut buf = [0u8; INDEX_RECORD_SIZE];
|
||||
buf[0..32].copy_from_slice(&self.key);
|
||||
buf[32..36].copy_from_slice(&self.segment_id.to_le_bytes());
|
||||
buf[36..44].copy_from_slice(&self.offset.to_le_bytes());
|
||||
buf[44..48].copy_from_slice(&self.data_size.to_le_bytes());
|
||||
buf[48] = self.flags;
|
||||
// bytes 49..52 are padding (keep zero)
|
||||
buf
|
||||
}
|
||||
|
||||
pub fn decode(buf: &[u8; INDEX_RECORD_SIZE]) -> Self {
|
||||
let mut key = [0u8; 32];
|
||||
key.copy_from_slice(&buf[0..32]);
|
||||
let segment_id = u32::from_le_bytes(buf[32..36].try_into().unwrap());
|
||||
let offset = u64::from_le_bytes(buf[36..44].try_into().unwrap());
|
||||
let data_size = u32::from_le_bytes(buf[44..48].try_into().unwrap());
|
||||
let flags = buf[48];
|
||||
Self {
|
||||
key,
|
||||
segment_id,
|
||||
offset,
|
||||
data_size,
|
||||
flags,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Represents a loaded and deduplicated bucket in memory.
|
||||
pub struct BucketIndex {
|
||||
pub bucket_id: u16,
|
||||
/// Records sorted by key, deduplicated (one record per key, latest wins).
|
||||
pub records: Vec<IndexRecord>,
|
||||
}
|
||||
|
||||
impl BucketIndex {
|
||||
/// Build from raw records: sort by key, dedup keeping the one with max offset.
|
||||
pub fn from_records(mut records: Vec<IndexRecord>, bucket_id: u16) -> Self {
|
||||
records.sort_by_key(|a| a.key);
|
||||
// Dedup: keep last (max offset) for each key
|
||||
let mut deduped = Vec::with_capacity(records.len());
|
||||
let mut i = 0;
|
||||
while i < records.len() {
|
||||
let mut best = i;
|
||||
let mut j = i + 1;
|
||||
while j < records.len() && records[j].key == records[i].key {
|
||||
if records[j].offset > records[best].offset {
|
||||
best = j;
|
||||
}
|
||||
j += 1;
|
||||
}
|
||||
deduped.push(records[best].clone());
|
||||
i = j;
|
||||
}
|
||||
Self {
|
||||
bucket_id,
|
||||
records: deduped,
|
||||
}
|
||||
}
|
||||
|
||||
/// Binary search for a key. Returns the record if found.
|
||||
pub fn find(&self, key: &[u8; 32]) -> Option<&IndexRecord> {
|
||||
match self.records.binary_search_by(|r| r.key.cmp(key)) {
|
||||
Ok(idx) => Some(&self.records[idx]),
|
||||
Err(_) => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Append a new record and maintain sorted order.
|
||||
pub fn insert(&mut self, record: IndexRecord) {
|
||||
match self.records.binary_search_by(|r| r.key.cmp(&record.key)) {
|
||||
Ok(idx) => {
|
||||
// Replace if newer (larger offset)
|
||||
if record.offset > self.records[idx].offset {
|
||||
self.records[idx] = record;
|
||||
}
|
||||
}
|
||||
Err(idx) => {
|
||||
self.records.insert(idx, record);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
self.records.len()
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.records.is_empty()
|
||||
}
|
||||
}
|
||||
|
||||
/// Manages a bucket index file on disk.
|
||||
pub struct BucketFile {
|
||||
path: PathBuf,
|
||||
bucket_id: u16,
|
||||
}
|
||||
|
||||
impl BucketFile {
|
||||
pub fn path_for(account_dir: &Path, bucket_id: u16) -> PathBuf {
|
||||
account_dir.join("buckets").join(format!("{:02x}.idx", bucket_id))
|
||||
}
|
||||
|
||||
pub fn open(account_dir: &Path, bucket_id: u16) -> Self {
|
||||
Self {
|
||||
path: Self::path_for(account_dir, bucket_id),
|
||||
bucket_id,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn path(&self) -> &Path {
|
||||
&self.path
|
||||
}
|
||||
|
||||
pub fn bucket_id(&self) -> u16 {
|
||||
self.bucket_id
|
||||
}
|
||||
|
||||
/// Ensure the buckets directory exists.
|
||||
pub fn ensure_dir(account_dir: &Path) -> Result<()> {
|
||||
let dir = account_dir.join("buckets");
|
||||
std::fs::create_dir_all(&dir)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Append a single record to the bucket file.
|
||||
pub fn append(&self, record: &IndexRecord) -> Result<()> {
|
||||
let mut file = OpenOptions::new()
|
||||
.create(true)
|
||||
.append(true)
|
||||
.open(&self.path)?;
|
||||
file.write_all(&record.encode())?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Append multiple records at once.
|
||||
pub fn append_batch(&self, records: &[IndexRecord]) -> Result<()> {
|
||||
if records.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
let mut file = OpenOptions::new()
|
||||
.create(true)
|
||||
.append(true)
|
||||
.open(&self.path)?;
|
||||
for r in records {
|
||||
file.write_all(&r.encode())?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Load all records from the bucket file.
|
||||
/// If the file size is not a multiple of INDEX_RECORD_SIZE (partial write),
|
||||
/// the trailing bytes are silently ignored.
|
||||
pub fn load_all(&self) -> Result<Vec<IndexRecord>> {
|
||||
if !self.path.exists() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let data = std::fs::read(&self.path)?;
|
||||
let remainder = data.len() % INDEX_RECORD_SIZE;
|
||||
let count = data.len() / INDEX_RECORD_SIZE;
|
||||
let mut records = Vec::with_capacity(count);
|
||||
for i in 0..count {
|
||||
let start = i * INDEX_RECORD_SIZE;
|
||||
let end = start + INDEX_RECORD_SIZE;
|
||||
let buf: &[u8; INDEX_RECORD_SIZE] = data[start..end]
|
||||
.try_into()
|
||||
.map_err(|_| crate::error::Error::BucketIndexCorrupt {
|
||||
path: self.path.clone(),
|
||||
reason: format!("unexpected file size {}, not a multiple of {}", data.len(), INDEX_RECORD_SIZE),
|
||||
})?;
|
||||
records.push(IndexRecord::decode(buf));
|
||||
}
|
||||
if remainder > 0 {
|
||||
tracing::warn!(
|
||||
"Bucket file {:?} has {} trailing bytes (expected multiple of {}), ignoring",
|
||||
self.path, remainder, INDEX_RECORD_SIZE
|
||||
);
|
||||
}
|
||||
Ok(records)
|
||||
}
|
||||
|
||||
/// Load all records, sort, and deduplicate into a BucketIndex.
|
||||
pub fn load_index(&self) -> Result<BucketIndex> {
|
||||
let records = self.load_all()?;
|
||||
Ok(BucketIndex::from_records(records, self.bucket_id))
|
||||
}
|
||||
|
||||
/// Rewrite the bucket file with a sorted, deduplicated set of records.
|
||||
/// Uses atomic temp+rename to be safe on NFS.
|
||||
pub fn rewrite(&self, records: &[IndexRecord]) -> Result<()> {
|
||||
let mut buf = Vec::with_capacity(records.len() * INDEX_RECORD_SIZE);
|
||||
for r in records {
|
||||
buf.extend_from_slice(&r.encode());
|
||||
}
|
||||
crate::fs::create_atomic(&self.path, &buf)
|
||||
}
|
||||
|
||||
/// Delete the bucket file.
|
||||
pub fn delete(&self) -> Result<()> {
|
||||
if self.path.exists() {
|
||||
std::fs::remove_file(&self.path)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Compute bucket_id from a key's first 2 bytes.
|
||||
pub fn bucket_id(key: &[u8; 32]) -> u16 {
|
||||
u16::from_be_bytes([key[0], key[1]]) % BUCKET_COUNT
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
fn test_index_record_encode_decode() {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&[1, 2, 3, 4]);
|
||||
let rec = IndexRecord::new(key, 5, 12345, 500, 0);
|
||||
let encoded = rec.encode();
|
||||
let decoded = IndexRecord::decode(&encoded);
|
||||
assert_eq!(rec, decoded);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bucket_id_deterministic() {
|
||||
let mut key = [0u8; 32];
|
||||
key[0] = 0x00;
|
||||
key[1] = 0x0F;
|
||||
assert_eq!(bucket_id(&key), 15);
|
||||
key[0] = 0x00;
|
||||
key[1] = 0x10;
|
||||
assert_eq!(bucket_id(&key), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bucket_append_and_load() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let bucket = BucketFile::open(dir.path(), 0);
|
||||
BucketFile::ensure_dir(dir.path()).unwrap();
|
||||
|
||||
let r1 = IndexRecord::new([1u8; 32], 1, 100, 50, 0);
|
||||
let r2 = IndexRecord::new([2u8; 32], 1, 200, 60, 0);
|
||||
|
||||
bucket.append(&r1).unwrap();
|
||||
bucket.append(&r2).unwrap();
|
||||
|
||||
let loaded = bucket.load_all().unwrap();
|
||||
assert_eq!(loaded.len(), 2);
|
||||
assert_eq!(loaded[0].key, [1u8; 32]);
|
||||
assert_eq!(loaded[1].key, [2u8; 32]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bucket_index_dedup() {
|
||||
let recs = vec![
|
||||
IndexRecord::new([1u8; 32], 1, 100, 50, 0),
|
||||
IndexRecord::new([1u8; 32], 2, 200, 50, 0), // newer offset wins
|
||||
IndexRecord::new([2u8; 32], 1, 300, 60, 0),
|
||||
];
|
||||
let idx = BucketIndex::from_records(recs, 0);
|
||||
assert_eq!(idx.len(), 2);
|
||||
let found = idx.find(&[1u8; 32]).unwrap();
|
||||
assert_eq!(found.segment_id, 2);
|
||||
assert_eq!(found.offset, 200);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bucket_index_find_missing() {
|
||||
let recs = vec![IndexRecord::new([1u8; 32], 1, 100, 50, 0)];
|
||||
let idx = BucketIndex::from_records(recs, 0);
|
||||
assert!(idx.find(&[99u8; 32]).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bucket_rewrite() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let bucket = BucketFile::open(dir.path(), 0);
|
||||
BucketFile::ensure_dir(dir.path()).unwrap();
|
||||
|
||||
let r1 = IndexRecord::new([3u8; 32], 1, 300, 70, 0);
|
||||
let r2 = IndexRecord::new([1u8; 32], 1, 100, 50, 0);
|
||||
bucket.append(&r1).unwrap();
|
||||
bucket.append(&r2).unwrap();
|
||||
|
||||
// Rewrite sorted
|
||||
let sorted = vec![r2.clone(), r1.clone()];
|
||||
bucket.rewrite(&sorted).unwrap();
|
||||
|
||||
let loaded = bucket.load_all().unwrap();
|
||||
assert_eq!(loaded.len(), 2);
|
||||
assert_eq!(loaded[0].key, [1u8; 32]);
|
||||
assert_eq!(loaded[1].key, [3u8; 32]);
|
||||
}
|
||||
}
|
||||
219
crates/blob/src/cache.rs
Normal file
219
crates/blob/src/cache.rs
Normal file
@@ -0,0 +1,219 @@
|
||||
use std::collections::HashMap;
|
||||
use std::path::Path;
|
||||
use std::sync::Mutex;
|
||||
|
||||
use crate::bucket::{BucketFile, BucketIndex, IndexRecord};
|
||||
use crate::error::Result;
|
||||
|
||||
type CacheKey = (String, u16);
|
||||
|
||||
/// Thread-safe LRU bucket cache with single-lock interior.
|
||||
/// Eliminates the TOCTOU race in the old two-Mutex design.
|
||||
pub struct BucketCache {
|
||||
inner: Mutex<CacheInner>,
|
||||
}
|
||||
|
||||
struct CacheInner {
|
||||
max_entries: usize,
|
||||
entries: Vec<CacheEntry>,
|
||||
index: HashMap<CacheKey, usize>,
|
||||
}
|
||||
|
||||
struct CacheEntry {
|
||||
key: CacheKey,
|
||||
index: BucketIndex,
|
||||
}
|
||||
|
||||
impl BucketCache {
|
||||
pub fn new(max_entries: usize) -> Self {
|
||||
Self {
|
||||
inner: Mutex::new(CacheInner {
|
||||
max_entries: max_entries.max(1),
|
||||
entries: Vec::new(),
|
||||
index: HashMap::new(),
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
/// Get or load a bucket index. Eliminates TOCTOU via double-checked locking.
|
||||
pub fn get_or_load(
|
||||
&self,
|
||||
account: &str,
|
||||
bucket_id: u16,
|
||||
account_dir: &Path,
|
||||
) -> Result<Vec<IndexRecord>> {
|
||||
let key: CacheKey = (account.to_string(), bucket_id);
|
||||
|
||||
// Check cache
|
||||
{
|
||||
let inner = self.inner.lock().unwrap();
|
||||
if let Some(&pos) = inner.index.get(&key) {
|
||||
return Ok(inner.entries[pos].index.records.clone());
|
||||
}
|
||||
}
|
||||
|
||||
// Load from disk
|
||||
let bucket_file = BucketFile::open(account_dir, bucket_id);
|
||||
let bucket_index = bucket_file.load_index()?;
|
||||
let records = bucket_index.records.clone();
|
||||
|
||||
// Insert with double-check (another thread might have beaten us)
|
||||
{
|
||||
let mut inner = self.inner.lock().unwrap();
|
||||
if let Some(&pos) = inner.index.get(&key) {
|
||||
return Ok(inner.entries[pos].index.records.clone());
|
||||
}
|
||||
// Evict if full
|
||||
if inner.entries.len() >= inner.max_entries {
|
||||
if let Some(evicted) = inner.entries.pop() {
|
||||
inner.index.remove(&evicted.key);
|
||||
}
|
||||
}
|
||||
// Insert at front
|
||||
inner.entries.insert(0, CacheEntry {
|
||||
key: key.clone(),
|
||||
index: bucket_index,
|
||||
});
|
||||
// Rebuild index
|
||||
inner.index.clear();
|
||||
for i in 0..inner.entries.len() {
|
||||
let key = inner.entries[i].key.clone();
|
||||
inner.index.insert(key, i);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(records)
|
||||
}
|
||||
|
||||
/// Insert or update a single record in a cached bucket.
|
||||
pub fn update_record(&self, account: &str, bucket_id: u16, record: IndexRecord) {
|
||||
let key: CacheKey = (account.to_string(), bucket_id);
|
||||
let mut inner = self.inner.lock().unwrap();
|
||||
|
||||
if let Some(&pos) = inner.index.get(&key) {
|
||||
inner.entries[pos].index.insert(record);
|
||||
// Move to front
|
||||
let entry = inner.entries.remove(pos);
|
||||
inner.entries.insert(0, entry);
|
||||
// Rebuild index
|
||||
inner.index.clear();
|
||||
for i in 0..inner.entries.len() {
|
||||
let entry_key = inner.entries[i].key.clone();
|
||||
inner.index.insert(entry_key, i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Invalidate a cached bucket (after GC rewrites bucket files).
|
||||
pub fn invalidate(&self, account: &str, bucket_id: u16) {
|
||||
let key: CacheKey = (account.to_string(), bucket_id);
|
||||
let mut inner = self.inner.lock().unwrap();
|
||||
|
||||
if let Some(&pos) = inner.index.get(&key) {
|
||||
inner.entries.remove(pos);
|
||||
inner.index.clear();
|
||||
for i in 0..inner.entries.len() {
|
||||
let entry_key = inner.entries[i].key.clone();
|
||||
inner.index.insert(entry_key, i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
self.inner.lock().unwrap().entries.len()
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.inner.lock().unwrap().entries.is_empty()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::bucket::IndexRecord;
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
fn test_cache_miss_loads_from_disk() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
crate::bucket::BucketFile::ensure_dir(dir.path()).unwrap();
|
||||
let bf = BucketFile::open(dir.path(), 0);
|
||||
bf.append(&IndexRecord::new([1u8; 32], 1, 100, 50, 0))
|
||||
.unwrap();
|
||||
|
||||
let cache = BucketCache::new(10);
|
||||
let records = cache
|
||||
.get_or_load("test", 0, dir.path())
|
||||
.unwrap();
|
||||
assert_eq!(records.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_cache_hit() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
crate::bucket::BucketFile::ensure_dir(dir.path()).unwrap();
|
||||
let bf = BucketFile::open(dir.path(), 0);
|
||||
bf.append(&IndexRecord::new([2u8; 32], 1, 200, 60, 0))
|
||||
.unwrap();
|
||||
|
||||
let cache = BucketCache::new(10);
|
||||
let _ = cache.get_or_load("test", 0, dir.path()).unwrap();
|
||||
let records = cache
|
||||
.get_or_load("test", 0, dir.path())
|
||||
.unwrap();
|
||||
assert_eq!(records.len(), 1);
|
||||
assert_eq!(cache.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_cache_eviction() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
crate::bucket::BucketFile::ensure_dir(dir.path()).unwrap();
|
||||
let cache = BucketCache::new(2);
|
||||
|
||||
for b in 0..4 {
|
||||
let bf = BucketFile::open(dir.path(), b);
|
||||
bf.append(&IndexRecord::new([b as u8; 32], 1, 100, 50, 0))
|
||||
.unwrap();
|
||||
let _ = cache.get_or_load("test", b, dir.path()).unwrap();
|
||||
}
|
||||
|
||||
assert!(cache.len() <= 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_concurrent_get_or_load_no_deadlock() {
|
||||
use std::sync::Arc;
|
||||
use std::thread;
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
crate::bucket::BucketFile::ensure_dir(dir.path()).unwrap();
|
||||
let bf = BucketFile::open(dir.path(), 0);
|
||||
for i in 0..10u8 {
|
||||
bf.append(&IndexRecord::new([i; 32], 1, i as u64 * 100, 50, 0))
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let cache = Arc::new(BucketCache::new(10));
|
||||
let dir_path = dir.path().to_path_buf();
|
||||
|
||||
let mut handles = vec![];
|
||||
for _ in 0..4 {
|
||||
let cache = cache.clone();
|
||||
let dir_path = dir_path.clone();
|
||||
handles.push(thread::spawn(move || {
|
||||
for _ in 0..100 {
|
||||
let records = cache
|
||||
.get_or_load("test", 0, &dir_path)
|
||||
.unwrap();
|
||||
assert_eq!(records.len(), 10);
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
}
|
||||
}
|
||||
65
crates/blob/src/checksum.rs
Normal file
65
crates/blob/src/checksum.rs
Normal file
@@ -0,0 +1,65 @@
|
||||
use crc32fast::Hasher;
|
||||
|
||||
pub fn crc32(data: &[u8]) -> u32 {
|
||||
let mut h = Hasher::new();
|
||||
h.update(data);
|
||||
h.finalize()
|
||||
}
|
||||
|
||||
pub struct CrcWriter {
|
||||
hasher: Hasher,
|
||||
}
|
||||
|
||||
impl Default for CrcWriter {
|
||||
fn default() -> Self {
|
||||
Self::new()
|
||||
}
|
||||
}
|
||||
|
||||
impl CrcWriter {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
hasher: Hasher::new(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn update(&mut self, data: &[u8]) {
|
||||
self.hasher.update(data);
|
||||
}
|
||||
|
||||
pub fn finalize(self) -> u32 {
|
||||
self.hasher.finalize()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_crc32_deterministic() {
|
||||
let a = crc32(b"hello");
|
||||
let b = crc32(b"hello");
|
||||
assert_eq!(a, b);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_crc32_different() {
|
||||
let a = crc32(b"hello");
|
||||
let b = crc32(b"world");
|
||||
assert!(a != b);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_crc_writer_matches_crc32() {
|
||||
let mut w = CrcWriter::new();
|
||||
w.update(b"hello");
|
||||
w.update(b" world");
|
||||
assert_eq!(w.finalize(), crc32(b"hello world"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_crc32_empty() {
|
||||
assert_eq!(crc32(b""), 0);
|
||||
}
|
||||
}
|
||||
97
crates/blob/src/compress.rs
Normal file
97
crates/blob/src/compress.rs
Normal file
@@ -0,0 +1,97 @@
|
||||
use crate::types::Codec;
|
||||
|
||||
pub fn compress(data: &[u8], codec: Codec, threshold: usize, level: i32) -> (Vec<u8>, Codec) {
|
||||
if data.len() < threshold {
|
||||
return (data.to_vec(), Codec::None);
|
||||
}
|
||||
let (compressed, actual_codec) = match codec {
|
||||
Codec::Zstd => {
|
||||
match zstd::encode_all(data, level) {
|
||||
Ok(out) => (out, Codec::Zstd),
|
||||
Err(e) => {
|
||||
tracing::warn!("zstd compression failed, storing uncompressed: {}", e);
|
||||
(data.to_vec(), Codec::None)
|
||||
}
|
||||
}
|
||||
}
|
||||
Codec::Lz4 => {
|
||||
let out = lz4_flex::compress(data);
|
||||
(out, Codec::Lz4)
|
||||
}
|
||||
Codec::None => (data.to_vec(), Codec::None),
|
||||
};
|
||||
// If compression made it larger, store uncompressed
|
||||
if compressed.len() >= data.len() {
|
||||
(data.to_vec(), Codec::None)
|
||||
} else {
|
||||
(compressed, actual_codec)
|
||||
}
|
||||
}
|
||||
|
||||
pub fn decompress(data: &[u8], codec: Codec, raw_size: usize) -> crate::error::Result<Vec<u8>> {
|
||||
match codec {
|
||||
Codec::None => Ok(data.to_vec()),
|
||||
Codec::Zstd => {
|
||||
zstd::decode_all(data)
|
||||
.map_err(|e| crate::error::Error::Compression(format!("zstd decompress: {}", e)))
|
||||
}
|
||||
Codec::Lz4 => {
|
||||
lz4_flex::decompress(data, raw_size)
|
||||
.map_err(|e| crate::error::Error::Compression(format!("lz4 decompress: {}", e)))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_small_data_not_compressed() {
|
||||
let data = b"hi";
|
||||
let (out, codec) = compress(data, Codec::Zstd, 4096, 0);
|
||||
assert_eq!(out, b"hi");
|
||||
assert_eq!(codec, Codec::None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_large_data_compressed_zstd() {
|
||||
let data = vec![b'A'; 5000];
|
||||
let (out, codec) = compress(&data, Codec::Zstd, 4096, 0);
|
||||
assert_eq!(codec, Codec::Zstd);
|
||||
assert!(out.len() < data.len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_roundtrip_zstd() {
|
||||
let data = vec![b'B'; 10000];
|
||||
let (compressed, codec) = compress(&data, Codec::Zstd, 4096, 0);
|
||||
let decompressed = decompress(&compressed, codec, data.len()).unwrap();
|
||||
assert_eq!(decompressed, data);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_roundtrip_lz4() {
|
||||
let data = vec![b'C'; 10000];
|
||||
let (compressed, codec) = compress(&data, Codec::Lz4, 4096, 0);
|
||||
let decompressed = decompress(&compressed, codec, data.len()).unwrap();
|
||||
assert_eq!(decompressed, data);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_roundtrip_none() {
|
||||
let data = vec![b'D'; 100];
|
||||
let (compressed, codec) = compress(&data, Codec::None, 4096, 0);
|
||||
assert_eq!(codec, Codec::None);
|
||||
let decompressed = decompress(&compressed, codec, data.len()).unwrap();
|
||||
assert_eq!(decompressed, data);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_threshold_zero_always_compresses() {
|
||||
let data = vec![b'E'; 100];
|
||||
let (out, codec) = compress(&data, Codec::Zstd, 0, 0);
|
||||
assert_eq!(codec, Codec::Zstd);
|
||||
assert!(out.len() < data.len());
|
||||
}
|
||||
}
|
||||
398
crates/blob/src/engine.rs
Normal file
398
crates/blob/src/engine.rs
Normal file
@@ -0,0 +1,398 @@
|
||||
use std::collections::HashMap;
|
||||
use std::fs;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::{Arc, RwLock};
|
||||
|
||||
use crate::account::AccountHandle;
|
||||
use crate::bucket::{self, IndexRecord};
|
||||
use crate::cache::BucketCache;
|
||||
use crate::compress;
|
||||
use crate::error::{Error, Result};
|
||||
use crate::gc::{self, GcStats};
|
||||
use crate::meta::GlobalMeta;
|
||||
use crate::segment::SegmentReader;
|
||||
use crate::types::{Codec, Config, ENTRY_HEADER_SIZE};
|
||||
|
||||
pub struct Engine {
|
||||
root: PathBuf,
|
||||
config: Config,
|
||||
cache: BucketCache,
|
||||
accounts: RwLock<HashMap<String, Arc<AccountHandle>>>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct AccountStats {
|
||||
pub account_id: String,
|
||||
pub total_keys: u64,
|
||||
pub total_bytes: u64,
|
||||
pub deleted_bytes: u64,
|
||||
pub segment_count: usize,
|
||||
}
|
||||
|
||||
impl Engine {
|
||||
pub fn open(path: &Path, config: Config) -> Result<Self> {
|
||||
config.validate()?;
|
||||
fs::create_dir_all(path)?;
|
||||
fs::create_dir_all(path.join("accounts"))?;
|
||||
|
||||
let mut global = GlobalMeta::load(path)?;
|
||||
global.save(path)?;
|
||||
|
||||
let cache = BucketCache::new(config.lru_bucket_count);
|
||||
|
||||
let accounts_dir = path.join("accounts");
|
||||
let mut accounts = HashMap::new();
|
||||
|
||||
if accounts_dir.exists() {
|
||||
for entry in fs::read_dir(&accounts_dir)? {
|
||||
let entry = entry?;
|
||||
if entry.file_type()?.is_dir() {
|
||||
let account_name = entry.file_name().to_string_lossy().into_owned();
|
||||
|
||||
let _ = crate::recovery::cleanup_temp_files(&entry.path());
|
||||
|
||||
match crate::recovery::recover_account(&entry.path()) {
|
||||
Ok(_meta) => {
|
||||
match AccountHandle::open(path, &account_name) {
|
||||
Ok(handle) => {
|
||||
accounts.insert(account_name, handle);
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!(
|
||||
"Failed to open account {}: {}",
|
||||
account_name,
|
||||
e
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!(
|
||||
"Failed to recover account {}: {}",
|
||||
account_name,
|
||||
e
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
global.accounts = accounts.keys().cloned().collect();
|
||||
global.save(path)?;
|
||||
|
||||
Ok(Self {
|
||||
root: path.to_path_buf(),
|
||||
config,
|
||||
cache,
|
||||
accounts: RwLock::new(accounts),
|
||||
})
|
||||
}
|
||||
|
||||
// ── Account management ──────────────────────────────────────────────
|
||||
|
||||
pub fn create_account(&self, account_id: &str) -> Result<()> {
|
||||
let mut accounts = self.accounts.write().unwrap();
|
||||
if accounts.contains_key(account_id) {
|
||||
return Err(Error::AccountAlreadyExists(account_id.to_string()));
|
||||
}
|
||||
let handle = AccountHandle::create(&self.root, account_id)?;
|
||||
accounts.insert(account_id.to_string(), handle);
|
||||
|
||||
let mut global = GlobalMeta::load(&self.root)?;
|
||||
global.accounts = accounts.keys().cloned().collect();
|
||||
global.save(&self.root)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn delete_account(&self, account_id: &str) -> Result<()> {
|
||||
let mut accounts = self.accounts.write().unwrap();
|
||||
let handle = accounts
|
||||
.remove(account_id)
|
||||
.ok_or_else(|| Error::AccountNotFound(account_id.to_string()))?;
|
||||
|
||||
let account_dir = handle.dir().to_path_buf();
|
||||
drop(handle);
|
||||
fs::remove_dir_all(&account_dir)?;
|
||||
|
||||
let mut global = GlobalMeta::load(&self.root)?;
|
||||
global.accounts = accounts.keys().cloned().collect();
|
||||
global.save(&self.root)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn list_accounts(&self) -> Vec<String> {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
accounts.keys().cloned().collect()
|
||||
}
|
||||
|
||||
// ── Read / Write / Delete ───────────────────────────────────────────
|
||||
|
||||
pub fn write(
|
||||
&self,
|
||||
account_id: &str,
|
||||
key: [u8; 32],
|
||||
value: &[u8],
|
||||
codec: Codec,
|
||||
) -> Result<()> {
|
||||
if value.len() > crate::types::MAX_VALUE_SIZE {
|
||||
return Err(Error::ValueTooLarge { size: value.len() });
|
||||
}
|
||||
|
||||
let handle = {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
accounts
|
||||
.get(account_id)
|
||||
.ok_or_else(|| Error::AccountNotFound(account_id.to_string()))?
|
||||
.clone()
|
||||
};
|
||||
|
||||
let _write_lock = handle.write_mutex.lock().unwrap();
|
||||
let mut inner = handle.write();
|
||||
|
||||
let (data, actual_codec) =
|
||||
compress::compress(value, codec, self.config.compress_threshold, self.config.compression_level);
|
||||
|
||||
let (segment_id, offset, data_size) =
|
||||
inner.write_entry(key, &data, 0, actual_codec)?;
|
||||
|
||||
let record = IndexRecord::new(key, segment_id, offset, data_size, 0);
|
||||
inner.append_index(&record)?;
|
||||
|
||||
let entry_end = offset + ENTRY_HEADER_SIZE as u64 + data_size as u64;
|
||||
inner.mark_indexed(segment_id, entry_end)?;
|
||||
|
||||
let bucket_id = bucket::bucket_id(&key);
|
||||
self.cache.update_record(account_id, bucket_id, record);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn read(&self, account_id: &str, key: &[u8; 32]) -> Result<Option<Vec<u8>>> {
|
||||
let bucket_id = bucket::bucket_id(key);
|
||||
|
||||
let handle = {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
accounts
|
||||
.get(account_id)
|
||||
.ok_or_else(|| Error::AccountNotFound(account_id.to_string()))?
|
||||
.clone()
|
||||
};
|
||||
|
||||
let (record, seg_path): (IndexRecord, PathBuf) = {
|
||||
let inner = handle.read();
|
||||
let records = self
|
||||
.cache
|
||||
.get_or_load(account_id, bucket_id, handle.dir())?;
|
||||
match records.binary_search_by(|r| r.key.cmp(key)) {
|
||||
Ok(idx) => {
|
||||
let r = records[idx].clone();
|
||||
if r.is_tombstone() {
|
||||
return Ok(None);
|
||||
}
|
||||
let seg_path = inner.segment_path(r.segment_id)?;
|
||||
(r, seg_path)
|
||||
}
|
||||
Err(_) => return Ok(None),
|
||||
}
|
||||
};
|
||||
|
||||
if !seg_path.exists() {
|
||||
return Err(Error::SegmentNotFound(record.segment_id));
|
||||
}
|
||||
|
||||
let reader = SegmentReader::open(seg_path.clone(), record.segment_id)?;
|
||||
let file = handle.get_segment_file(record.segment_id, &seg_path)?;
|
||||
let (entry, _) = reader.read_entry_at_file(record.offset, &file)?;
|
||||
|
||||
let value = compress::decompress(&entry.data, entry.codec, entry.raw_size as usize)?;
|
||||
|
||||
Ok(Some(value))
|
||||
}
|
||||
|
||||
pub fn delete(&self, account_id: &str, key: &[u8; 32]) -> Result<()> {
|
||||
let handle = {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
accounts
|
||||
.get(account_id)
|
||||
.ok_or_else(|| Error::AccountNotFound(account_id.to_string()))?
|
||||
.clone()
|
||||
};
|
||||
|
||||
let _write_lock = handle.write_mutex.lock().unwrap();
|
||||
let mut inner = handle.write();
|
||||
|
||||
let (segment_id, offset, data_size) =
|
||||
inner.write_entry(*key, &[], 1, Codec::None)?;
|
||||
|
||||
let record = IndexRecord::new(*key, segment_id, offset, data_size, 1);
|
||||
inner.append_index(&record)?;
|
||||
|
||||
let entry_end = offset + ENTRY_HEADER_SIZE as u64 + data_size as u64;
|
||||
inner.mark_indexed(segment_id, entry_end)?;
|
||||
|
||||
let bucket_id = bucket::bucket_id(key);
|
||||
self.cache.update_record(account_id, bucket_id, record);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ── Batch write ─────────────────────────────────────────────────────
|
||||
|
||||
pub fn write_batch(&self, account_id: &str, entries: &[([u8; 32], Vec<u8>, Codec)]) -> Result<()> {
|
||||
if entries.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let handle = {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
accounts
|
||||
.get(account_id)
|
||||
.ok_or_else(|| Error::AccountNotFound(account_id.to_string()))?
|
||||
.clone()
|
||||
};
|
||||
|
||||
let _write_lock = handle.write_mutex.lock().unwrap();
|
||||
let mut inner = handle.write();
|
||||
|
||||
let mut pending: Vec<(IndexRecord, u64)> = Vec::with_capacity(entries.len());
|
||||
for (key, value, codec) in entries {
|
||||
if value.len() > crate::types::MAX_VALUE_SIZE {
|
||||
return Err(Error::ValueTooLarge { size: value.len() });
|
||||
}
|
||||
let (data, actual_codec) =
|
||||
compress::compress(value, *codec, self.config.compress_threshold, self.config.compression_level);
|
||||
|
||||
let (segment_id, offset, data_size) =
|
||||
inner.append_entry(*key, &data, 0, actual_codec)?;
|
||||
|
||||
let entry_end = offset + ENTRY_HEADER_SIZE as u64 + data_size as u64;
|
||||
let record = IndexRecord::new(*key, segment_id, offset, data_size, 0);
|
||||
pending.push((record, entry_end));
|
||||
}
|
||||
|
||||
inner.flush_active()?;
|
||||
|
||||
for (record, entry_end) in &pending {
|
||||
inner.append_index(record)?;
|
||||
inner.mark_indexed(record.segment_id, *entry_end)?;
|
||||
|
||||
let bucket_id = bucket::bucket_id(&record.key);
|
||||
self.cache.update_record(account_id, bucket_id, record.clone());
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ── GC ──────────────────────────────────────────────────────────────
|
||||
|
||||
pub fn gc(&self, account_id: &str) -> Result<Option<GcStats>> {
|
||||
let handle = {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
accounts
|
||||
.get(account_id)
|
||||
.ok_or_else(|| Error::AccountNotFound(account_id.to_string()))?
|
||||
.clone()
|
||||
};
|
||||
|
||||
// Hold write_mutex to prevent concurrent writes from racing
|
||||
// with GC's bucket rebuild phase.
|
||||
let _write_lock = handle.write_mutex.lock().unwrap();
|
||||
|
||||
let result = gc::gc_account(handle.dir(), self.config.gc_deleted_ratio)?;
|
||||
|
||||
// Invalidate FilePool for GC'd segments (they were rewritten via rename)
|
||||
if let Some(ref stats) = result {
|
||||
handle.invalidate_file_cache(stats.segment_id);
|
||||
}
|
||||
|
||||
for bid in 0..crate::types::BUCKET_COUNT {
|
||||
self.cache.invalidate(account_id, bid);
|
||||
}
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
pub fn compact_buckets(&self, account_id: &str) -> Result<()> {
|
||||
let handle = {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
accounts
|
||||
.get(account_id)
|
||||
.ok_or_else(|| Error::AccountNotFound(account_id.to_string()))?
|
||||
.clone()
|
||||
};
|
||||
|
||||
// Hold write_mutex — compact rewrites all bucket files.
|
||||
let _write_lock = handle.write_mutex.lock().unwrap();
|
||||
|
||||
gc::compact_buckets(handle.dir())?;
|
||||
|
||||
for bid in 0..crate::types::BUCKET_COUNT {
|
||||
self.cache.invalidate(account_id, bid);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ── Stats / Shutdown ────────────────────────────────────────────────
|
||||
|
||||
pub fn stats(&self, account_id: &str) -> Result<AccountStats> {
|
||||
let handle = {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
accounts
|
||||
.get(account_id)
|
||||
.ok_or_else(|| Error::AccountNotFound(account_id.to_string()))?
|
||||
.clone()
|
||||
};
|
||||
|
||||
let inner = handle.read();
|
||||
let meta = inner.meta();
|
||||
let mut total_bytes = 0u64;
|
||||
let mut deleted_bytes = 0u64;
|
||||
|
||||
for seg in meta.segments.values() {
|
||||
total_bytes += seg.total_bytes;
|
||||
deleted_bytes += seg.deleted_bytes;
|
||||
}
|
||||
|
||||
let mut total_keys = 0u64;
|
||||
for bid in 0..crate::types::BUCKET_COUNT {
|
||||
if let Ok(records) =
|
||||
self.cache
|
||||
.get_or_load(account_id, bid, handle.dir())
|
||||
{
|
||||
total_keys += records.iter().filter(|r| !r.is_tombstone()).count() as u64;
|
||||
}
|
||||
}
|
||||
|
||||
Ok(AccountStats {
|
||||
account_id: account_id.to_string(),
|
||||
total_keys,
|
||||
total_bytes,
|
||||
deleted_bytes,
|
||||
segment_count: meta.segments.len(),
|
||||
})
|
||||
}
|
||||
|
||||
pub fn shutdown(&self) -> Result<()> {
|
||||
let accounts = self.accounts.read().unwrap();
|
||||
for (_, handle) in accounts.iter() {
|
||||
let mut inner = handle.write();
|
||||
inner.flush_active()?;
|
||||
}
|
||||
let global = GlobalMeta::load(&self.root)?;
|
||||
global.save(&self.root)?;
|
||||
tracing::info!("bichon-blob shut down cleanly");
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for Engine {
|
||||
fn drop(&mut self) {
|
||||
if let Err(e) = self.shutdown() {
|
||||
tracing::error!("bichon-blob shutdown error: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
59
crates/blob/src/error.rs
Normal file
59
crates/blob/src/error.rs
Normal file
@@ -0,0 +1,59 @@
|
||||
use std::{io, path::PathBuf};
|
||||
|
||||
pub type Result<T> = std::result::Result<T, Error>;
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum Error {
|
||||
#[error("I/O error: {0}")]
|
||||
Io(#[from] io::Error),
|
||||
|
||||
#[error("JSON error: {0}")]
|
||||
Json(#[from] serde_json::Error),
|
||||
|
||||
#[error("CRC32 mismatch at {path}:{offset}")]
|
||||
CrcMismatch { path: PathBuf, offset: u64 },
|
||||
|
||||
#[error("Corrupt entry at {path}:{offset}: {reason}")]
|
||||
CorruptEntry {
|
||||
path: PathBuf,
|
||||
offset: u64,
|
||||
reason: String,
|
||||
},
|
||||
|
||||
#[error("Account not found: {0}")]
|
||||
AccountNotFound(String),
|
||||
|
||||
#[error("Account already exists: {0}")]
|
||||
AccountAlreadyExists(String),
|
||||
|
||||
#[error("Segment not found: {0}")]
|
||||
SegmentNotFound(u32),
|
||||
|
||||
#[error("Value too large: {size} bytes (max 100 MB)")]
|
||||
ValueTooLarge { size: usize },
|
||||
|
||||
#[error("Compression error: {0}")]
|
||||
Compression(String),
|
||||
|
||||
#[error("Disk full: {0}")]
|
||||
DiskFull(String),
|
||||
|
||||
#[error("Invalid config: {0}")]
|
||||
InvalidConfig(String),
|
||||
|
||||
#[error("Bucket index corrupt at {path}: {reason}")]
|
||||
BucketIndexCorrupt { path: PathBuf, reason: String },
|
||||
|
||||
#[error("Segment file truncated at {path}: expected {expected}, got {actual}")]
|
||||
SegmentTruncated {
|
||||
path: PathBuf,
|
||||
expected: u64,
|
||||
actual: u64,
|
||||
},
|
||||
|
||||
#[error("Corrupt metadata file: {0}")]
|
||||
CorruptMeta(String),
|
||||
|
||||
#[error("Unsupported metadata version {version} in {path}")]
|
||||
UnsupportedMetaVersion { path: PathBuf, version: u32 },
|
||||
}
|
||||
54
crates/blob/src/file_pool.rs
Normal file
54
crates/blob/src/file_pool.rs
Normal file
@@ -0,0 +1,54 @@
|
||||
use std::collections::VecDeque;
|
||||
use std::fs::File;
|
||||
use std::path::Path;
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use crate::error::Result;
|
||||
use crate::fs as fs_util;
|
||||
|
||||
/// Simple LRU pool of open file handles, keyed by segment_id.
|
||||
/// Uses Arc<Mutex<File>> to allow safe concurrent reads from the same segment.
|
||||
pub struct FilePool {
|
||||
max_entries: usize,
|
||||
entries: Mutex<VecDeque<(u32, Arc<Mutex<File>>)>>,
|
||||
}
|
||||
|
||||
impl FilePool {
|
||||
pub fn new(max_entries: usize) -> Self {
|
||||
Self {
|
||||
max_entries: max_entries.max(1),
|
||||
entries: Mutex::new(VecDeque::new()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Get an open File for the given segment. Reuses cached handle if available.
|
||||
pub fn get(&self, seg_id: u32, path: &Path) -> Result<Arc<Mutex<File>>> {
|
||||
let mut entries = self.entries.lock().unwrap();
|
||||
|
||||
// Check for existing entry
|
||||
for (i, (id, _)) in entries.iter().enumerate() {
|
||||
if *id == seg_id {
|
||||
let (_, file) = entries.remove(i).unwrap();
|
||||
entries.push_front((seg_id, file.clone()));
|
||||
return Ok(file);
|
||||
}
|
||||
}
|
||||
|
||||
// Open new file
|
||||
let file = Arc::new(Mutex::new(fs_util::open_read(path)?));
|
||||
|
||||
// Evict oldest if full
|
||||
if entries.len() >= self.max_entries {
|
||||
entries.pop_back();
|
||||
}
|
||||
|
||||
entries.push_front((seg_id, file.clone()));
|
||||
Ok(file)
|
||||
}
|
||||
|
||||
/// Remove a cached file handle (e.g. after GC rewrites a segment).
|
||||
pub fn invalidate(&self, seg_id: u32) {
|
||||
let mut entries = self.entries.lock().unwrap();
|
||||
entries.retain(|(id, _)| *id != seg_id);
|
||||
}
|
||||
}
|
||||
142
crates/blob/src/fs.rs
Normal file
142
crates/blob/src/fs.rs
Normal file
@@ -0,0 +1,142 @@
|
||||
use std::fs::{self, File, OpenOptions};
|
||||
use std::io::{self, Write};
|
||||
use std::path::Path;
|
||||
use std::time::Duration;
|
||||
|
||||
use crate::error::Result;
|
||||
|
||||
/// Max retries for transient filesystem errors (NFS ESTALE, CIFS sharing violations, etc.)
|
||||
const MAX_RETRIES: u32 = 5;
|
||||
const RETRY_DELAY: Duration = Duration::from_millis(20);
|
||||
|
||||
/// Check if an I/O error is transient (retryable).
|
||||
fn is_transient(err: &io::Error) -> bool {
|
||||
use std::io::ErrorKind;
|
||||
matches!(
|
||||
err.kind(),
|
||||
ErrorKind::TimedOut
|
||||
| ErrorKind::Interrupted
|
||||
| ErrorKind::WouldBlock
|
||||
| ErrorKind::UnexpectedEof
|
||||
) || err.raw_os_error() == Some(116) // ESTALE on Linux
|
||||
}
|
||||
|
||||
/// Open an existing file for reading, with retry on transient errors (NFS ESTALE etc.).
|
||||
pub fn open_read(path: &Path) -> Result<File> {
|
||||
let mut last_err = None;
|
||||
for attempt in 0..MAX_RETRIES {
|
||||
match File::open(path) {
|
||||
Ok(f) => return Ok(f),
|
||||
Err(e) if is_transient(&e) => {
|
||||
last_err = Some(e);
|
||||
if attempt > 0 {
|
||||
std::thread::sleep(RETRY_DELAY * attempt);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
}
|
||||
}
|
||||
Err(crate::error::Error::Io(last_err.unwrap()))
|
||||
}
|
||||
|
||||
/// Open an existing file for writing, with retry on transient errors.
|
||||
pub fn open_write(path: &Path) -> Result<File> {
|
||||
let mut last_err = None;
|
||||
for attempt in 0..MAX_RETRIES {
|
||||
match OpenOptions::new().write(true).open(path) {
|
||||
Ok(f) => return Ok(f),
|
||||
Err(e) if is_transient(&e) => {
|
||||
last_err = Some(e);
|
||||
if attempt > 0 {
|
||||
std::thread::sleep(RETRY_DELAY * attempt);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
}
|
||||
}
|
||||
Err(crate::error::Error::Io(last_err.unwrap()))
|
||||
}
|
||||
|
||||
/// Create a new file atomically: write content to a temp file, fsync, then rename.
|
||||
/// Avoids `create_new(true)` which is racy on NFS.
|
||||
pub fn create_atomic(path: &Path, content: &[u8]) -> Result<()> {
|
||||
let tmp = path.with_extension(
|
||||
path.extension()
|
||||
.map(|e| format!("{}.tmp", e.to_string_lossy()))
|
||||
.unwrap_or_else(|| "tmp".to_string()),
|
||||
);
|
||||
|
||||
{
|
||||
let mut f = File::create(&tmp)?;
|
||||
f.write_all(content)?;
|
||||
f.sync_all()?;
|
||||
}
|
||||
|
||||
fs::rename(&tmp, path)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Truncate an existing file to the given size, with retry.
|
||||
pub fn truncate(path: &Path, size: u64) -> Result<()> {
|
||||
let f = open_write(path)?;
|
||||
f.set_len(size)?;
|
||||
f.sync_all()?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
fn test_open_read_existing() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("test.txt");
|
||||
std::fs::write(&path, b"hello").unwrap();
|
||||
|
||||
let mut f = open_read(&path).unwrap();
|
||||
let mut s = String::new();
|
||||
std::io::Read::read_to_string(&mut f, &mut s).unwrap();
|
||||
assert_eq!(s, "hello");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_open_read_missing() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let result = open_read(&dir.path().join("nope.txt"));
|
||||
assert!(result.is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_create_atomic_success() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("data.bin");
|
||||
create_atomic(&path, b"hello world").unwrap();
|
||||
|
||||
let content = std::fs::read(&path).unwrap();
|
||||
assert_eq!(content, b"hello world");
|
||||
// Temp file should not exist
|
||||
assert!(!dir.path().join("data.bin.tmp").exists());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_create_atomic_overwrites() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("data.bin");
|
||||
create_atomic(&path, b"first").unwrap();
|
||||
create_atomic(&path, b"second").unwrap();
|
||||
assert_eq!(std::fs::read(&path).unwrap(), b"second");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_truncate() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = dir.path().join("trunc.bin");
|
||||
std::fs::write(&path, b"1234567890").unwrap();
|
||||
truncate(&path, 5).unwrap();
|
||||
assert_eq!(std::fs::metadata(&path).unwrap().len(), 5);
|
||||
}
|
||||
}
|
||||
267
crates/blob/src/gc.rs
Normal file
267
crates/blob/src/gc.rs
Normal file
@@ -0,0 +1,267 @@
|
||||
use std::collections::HashMap;
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use crate::bucket::{self, BucketFile, BucketIndex, IndexRecord};
|
||||
use crate::error::Result;
|
||||
#[cfg(test)]
|
||||
use crate::meta::SegmentStats;
|
||||
use crate::segment::{self, SegmentReader, SegmentWriter};
|
||||
|
||||
/// Result of a GC run.
|
||||
#[derive(Debug)]
|
||||
pub struct GcStats {
|
||||
pub segment_id: u32,
|
||||
pub bytes_before: u64,
|
||||
pub bytes_after: u64,
|
||||
pub entries_kept: usize,
|
||||
pub entries_skipped: usize,
|
||||
}
|
||||
|
||||
/// Run GC on an account: pick the sealed segment with highest deleted_ratio,
|
||||
/// rewrite it without deleted/overwritten entries, then rebuild all bucket files.
|
||||
pub fn gc_account(
|
||||
account_dir: &Path,
|
||||
deleted_ratio_threshold: f64,
|
||||
) -> Result<Option<GcStats>> {
|
||||
let meta = crate::meta::AccountMeta::load(account_dir)?;
|
||||
|
||||
// Find the best candidate
|
||||
let candidate = meta
|
||||
.segments
|
||||
.values()
|
||||
.filter(|s| s.sealed && s.deleted_ratio >= deleted_ratio_threshold)
|
||||
.max_by(|a, b| a.deleted_ratio.partial_cmp(&b.deleted_ratio).unwrap());
|
||||
|
||||
let target = match candidate {
|
||||
Some(s) => s.clone(),
|
||||
None => return Ok(None),
|
||||
};
|
||||
|
||||
let seg_path = account_dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(target.segment_id));
|
||||
let reader = SegmentReader::open(seg_path.clone(), target.segment_id)?;
|
||||
|
||||
// Build a global view: for each key, which entry (segment_id + offset) is the latest?
|
||||
let mut latest_key: HashMap<[u8; 32], (u32, u64)> = HashMap::new();
|
||||
|
||||
for &seg_id in meta.segments.keys() {
|
||||
let rpath = account_dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(seg_id));
|
||||
if !rpath.exists() {
|
||||
continue;
|
||||
}
|
||||
let r = SegmentReader::open(rpath, seg_id)?;
|
||||
let _ = r.scan_entries(0, |entry, offset| {
|
||||
match latest_key.get(&entry.key) {
|
||||
Some((existing_seg, existing_off)) => {
|
||||
if seg_id > *existing_seg
|
||||
|| (seg_id == *existing_seg && offset > *existing_off)
|
||||
{
|
||||
latest_key.insert(entry.key, (seg_id, offset));
|
||||
}
|
||||
}
|
||||
None => {
|
||||
latest_key.insert(entry.key, (seg_id, offset));
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
})?;
|
||||
}
|
||||
|
||||
// Create temp segment with a unique name
|
||||
let timestamp = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap()
|
||||
.as_nanos();
|
||||
let temp_name = format!("temp_{:016x}.seg", timestamp);
|
||||
let temp_path = account_dir.join("segments").join(&temp_name);
|
||||
let mut writer = SegmentWriter::create(temp_path.clone(), target.segment_id)?;
|
||||
|
||||
let mut bytes_after: u64 = 0;
|
||||
let mut entries_kept: usize = 0;
|
||||
let mut entries_skipped: usize = 0;
|
||||
|
||||
reader.scan_entries(0, |entry, offset| {
|
||||
// Skip tombstones
|
||||
if entry.is_tombstone() {
|
||||
entries_skipped += 1;
|
||||
return Ok(());
|
||||
}
|
||||
// Skip if this key has a newer entry in another segment
|
||||
if let Some((latest_seg, latest_off)) = latest_key.get(&entry.key) {
|
||||
if *latest_seg != target.segment_id || *latest_off != offset {
|
||||
entries_skipped += 1;
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
// Keep this entry
|
||||
writer.append(entry)?;
|
||||
bytes_after += entry.data.len() as u64;
|
||||
entries_kept += 1;
|
||||
Ok(())
|
||||
})?;
|
||||
|
||||
writer.fsync()?;
|
||||
|
||||
// Atomic rename: replace old segment with new one
|
||||
fs::rename(&temp_path, &seg_path)?;
|
||||
|
||||
// Rebuild all bucket files
|
||||
rebuild_buckets(account_dir, &meta)?;
|
||||
|
||||
// Update meta
|
||||
let mut meta = crate::meta::AccountMeta::load(account_dir)?;
|
||||
if let Some(stats) = meta.segments.get_mut(&target.segment_id) {
|
||||
stats.total_bytes = bytes_after;
|
||||
stats.deleted_bytes = 0;
|
||||
stats.recompute_ratio();
|
||||
}
|
||||
meta.save(account_dir)?;
|
||||
|
||||
Ok(Some(GcStats {
|
||||
segment_id: target.segment_id,
|
||||
bytes_before: target.total_bytes,
|
||||
bytes_after,
|
||||
entries_kept,
|
||||
entries_skipped,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Rebuild all 16 bucket files from scratch by scanning all segments.
|
||||
fn rebuild_buckets(account_dir: &Path, meta: &crate::meta::AccountMeta) -> Result<()> {
|
||||
let mut bucket_records: HashMap<u16, Vec<IndexRecord>> = HashMap::new();
|
||||
for i in 0..crate::types::BUCKET_COUNT {
|
||||
bucket_records.insert(i, Vec::new());
|
||||
}
|
||||
|
||||
for &seg_id in meta.segments.keys() {
|
||||
let seg_path = account_dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(seg_id));
|
||||
if !seg_path.exists() {
|
||||
continue;
|
||||
}
|
||||
let reader = SegmentReader::open(seg_path, seg_id)?;
|
||||
reader.scan_entries(0, |entry, offset| {
|
||||
let bid = bucket::bucket_id(&entry.key);
|
||||
let rec = IndexRecord::new(
|
||||
entry.key,
|
||||
seg_id,
|
||||
offset,
|
||||
entry.data.len() as u32,
|
||||
entry.flags,
|
||||
);
|
||||
bucket_records.entry(bid).or_default().push(rec);
|
||||
Ok(())
|
||||
})?;
|
||||
}
|
||||
|
||||
for (bid, records) in &bucket_records {
|
||||
let index = BucketIndex::from_records(records.clone(), *bid);
|
||||
let bf = BucketFile::open(account_dir, *bid);
|
||||
bf.rewrite(&index.records)?;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Compact bucket files: load, dedup, rewrite.
|
||||
pub fn compact_buckets(account_dir: &Path) -> Result<()> {
|
||||
for bid in 0..crate::types::BUCKET_COUNT {
|
||||
let bf = BucketFile::open(account_dir, bid);
|
||||
if bf.path().exists() {
|
||||
let index = bf.load_index()?;
|
||||
bf.rewrite(&index.records)?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::segment::Entry;
|
||||
use crate::types::Codec;
|
||||
use tempfile::TempDir;
|
||||
|
||||
fn setup_account(dir: &Path) {
|
||||
fs::create_dir_all(dir.join("segments")).unwrap();
|
||||
crate::bucket::BucketFile::ensure_dir(dir).unwrap();
|
||||
|
||||
let seg_path = dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(1));
|
||||
let mut writer = SegmentWriter::create(seg_path, 1).unwrap();
|
||||
|
||||
// Write 5 entries
|
||||
for i in 0..5u8 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0] = i;
|
||||
let entry = Entry::new(key, &vec![i; 1000], 0, Codec::None);
|
||||
writer.append(&entry).unwrap();
|
||||
}
|
||||
|
||||
// Tombstone entry 2
|
||||
let mut key2 = [0u8; 32];
|
||||
key2[0] = 2;
|
||||
let tomb = Entry::tombstone(key2);
|
||||
writer.append(&tomb).unwrap();
|
||||
|
||||
writer.fsync().unwrap();
|
||||
|
||||
// Save meta
|
||||
let mut meta = crate::meta::AccountMeta::new("test".into(), 2);
|
||||
meta.segments.insert(
|
||||
1,
|
||||
SegmentStats {
|
||||
segment_id: 1,
|
||||
total_bytes: 6000,
|
||||
deleted_bytes: 1000,
|
||||
deleted_ratio: 1000.0 / 6000.0,
|
||||
sealed: true,
|
||||
indexed_up_to_offset: 0,
|
||||
},
|
||||
);
|
||||
// Make segment 2 active so segment 1 is sealed
|
||||
let seg2_path = dir
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(2));
|
||||
SegmentWriter::create(seg2_path, 2).unwrap();
|
||||
meta.save(dir).unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_gc_removes_tombstones() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
setup_account(dir.path());
|
||||
|
||||
let result = gc_account(dir.path(), 0.01).unwrap();
|
||||
assert!(result.is_some());
|
||||
|
||||
// Verify segment 1 no longer has the tombstone'd entry
|
||||
let seg_path = dir
|
||||
.path()
|
||||
.join("segments")
|
||||
.join(segment::segment_filename(1));
|
||||
let reader = SegmentReader::open(seg_path, 1).unwrap();
|
||||
let mut count = 0;
|
||||
reader.scan_entries(0, |entry, _offset| {
|
||||
count += 1;
|
||||
assert!(entry.key[0] != 2);
|
||||
Ok(())
|
||||
}).unwrap();
|
||||
assert_eq!(count, 4); // 5 original - 1 tombstoned
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_compact_buckets() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
setup_account(dir.path());
|
||||
compact_buckets(dir.path()).unwrap();
|
||||
// Should not panic
|
||||
}
|
||||
}
|
||||
19
crates/blob/src/lib.rs
Normal file
19
crates/blob/src/lib.rs
Normal file
@@ -0,0 +1,19 @@
|
||||
pub mod account;
|
||||
pub mod bucket;
|
||||
pub mod cache;
|
||||
pub mod checksum;
|
||||
pub mod compress;
|
||||
pub mod engine;
|
||||
pub mod error;
|
||||
pub mod file_pool;
|
||||
pub mod fs;
|
||||
pub mod gc;
|
||||
pub mod meta;
|
||||
pub mod recovery;
|
||||
pub mod segment;
|
||||
pub mod types;
|
||||
|
||||
pub use account::AccountHandle;
|
||||
pub use engine::{AccountStats, Engine};
|
||||
pub use error::{Error, Result};
|
||||
pub use types::{Codec, Config};
|
||||
284
crates/blob/src/meta.rs
Normal file
284
crates/blob/src/meta.rs
Normal file
@@ -0,0 +1,284 @@
|
||||
use std::collections::BTreeMap;
|
||||
use std::path::Path;
|
||||
|
||||
use crate::checksum;
|
||||
use crate::error::Result;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
const META_VERSION: u32 = 1;
|
||||
|
||||
// ── Helpers ────────────────────────────────────────────────────────────────
|
||||
|
||||
fn write_bin<T: Serialize>(path: &Path, value: &T) -> Result<()> {
|
||||
let payload = bincode::serialize(value).map_err(|e| {
|
||||
crate::error::Error::CorruptMeta(format!("{}: bincode encode: {}", path.display(), e))
|
||||
})?;
|
||||
let crc = checksum::crc32(&payload);
|
||||
let mut buf = Vec::with_capacity(8 + payload.len());
|
||||
buf.extend_from_slice(&crc.to_le_bytes());
|
||||
buf.extend_from_slice(&META_VERSION.to_le_bytes());
|
||||
buf.extend_from_slice(&payload);
|
||||
|
||||
crate::fs::create_atomic(path, &buf)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn read_bin<T: for<'de> Deserialize<'de>>(path: &Path) -> Result<T> {
|
||||
let data = std::fs::read(path)?;
|
||||
if data.len() < 8 {
|
||||
return Err(crate::error::Error::CorruptMeta(path.display().to_string()));
|
||||
}
|
||||
let stored_crc = u32::from_le_bytes(data[0..4].try_into().unwrap());
|
||||
let version = u32::from_le_bytes(data[4..8].try_into().unwrap());
|
||||
if version != META_VERSION {
|
||||
return Err(crate::error::Error::UnsupportedMetaVersion {
|
||||
path: path.to_path_buf(),
|
||||
version,
|
||||
});
|
||||
}
|
||||
let computed = checksum::crc32(&data[8..]);
|
||||
if stored_crc != computed {
|
||||
return Err(crate::error::Error::CorruptMeta(path.display().to_string()));
|
||||
}
|
||||
bincode::deserialize(&data[8..]).map_err(|e| {
|
||||
crate::error::Error::CorruptMeta(format!("{}: bincode decode: {}", path.display(), e))
|
||||
})
|
||||
}
|
||||
|
||||
// ── GlobalMeta ─────────────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct GlobalMeta {
|
||||
pub version: u32,
|
||||
pub accounts: Vec<String>,
|
||||
}
|
||||
|
||||
impl Default for GlobalMeta {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
version: META_VERSION,
|
||||
accounts: Vec::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl GlobalMeta {
|
||||
pub fn load(store_root: &Path) -> Result<Self> {
|
||||
let bin_path = store_root.join("global_meta.bin");
|
||||
if bin_path.exists() {
|
||||
return read_bin(&bin_path);
|
||||
}
|
||||
// Migration from JSON
|
||||
let json_path = store_root.join("global_meta.json");
|
||||
if json_path.exists() {
|
||||
let data = std::fs::read_to_string(&json_path)?;
|
||||
let mut meta: Self = serde_json::from_str(&data)?;
|
||||
meta.accounts.sort();
|
||||
write_bin(&bin_path, &meta)?;
|
||||
let _ = std::fs::remove_file(&json_path);
|
||||
return Ok(meta);
|
||||
}
|
||||
Ok(Self::default())
|
||||
}
|
||||
|
||||
pub fn save(&self, store_root: &Path) -> Result<()> {
|
||||
let path = store_root.join("global_meta.bin");
|
||||
let mut meta = self.clone();
|
||||
meta.accounts.sort();
|
||||
write_bin(&path, &meta)
|
||||
}
|
||||
}
|
||||
|
||||
// ── SegmentStats ───────────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct SegmentStats {
|
||||
pub segment_id: u32,
|
||||
pub total_bytes: u64,
|
||||
pub deleted_bytes: u64,
|
||||
pub deleted_ratio: f64,
|
||||
pub sealed: bool,
|
||||
/// Byte offset up to which entries have been indexed in bucket files.
|
||||
/// Recovery starts scanning from here instead of 0.
|
||||
pub indexed_up_to_offset: u64,
|
||||
}
|
||||
|
||||
impl SegmentStats {
|
||||
pub fn new(segment_id: u32) -> Self {
|
||||
Self {
|
||||
segment_id,
|
||||
total_bytes: 0,
|
||||
deleted_bytes: 0,
|
||||
deleted_ratio: 0.0,
|
||||
sealed: false,
|
||||
indexed_up_to_offset: 0,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn recompute_ratio(&mut self) {
|
||||
if self.total_bytes > 0 {
|
||||
self.deleted_ratio = self.deleted_bytes as f64 / self.total_bytes as f64;
|
||||
} else {
|
||||
self.deleted_ratio = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── AccountMeta ────────────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct AccountMeta {
|
||||
pub account_id: String,
|
||||
pub active_segment_id: u32,
|
||||
pub segments: BTreeMap<u32, SegmentStats>,
|
||||
}
|
||||
|
||||
impl AccountMeta {
|
||||
pub fn new(account_id: String, active_segment_id: u32) -> Self {
|
||||
Self {
|
||||
account_id,
|
||||
active_segment_id,
|
||||
segments: BTreeMap::new(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn load(account_dir: &Path) -> Result<Self> {
|
||||
let bin_path = account_dir.join("meta.bin");
|
||||
if bin_path.exists() {
|
||||
return read_bin(&bin_path);
|
||||
}
|
||||
// Migration from JSON
|
||||
let json_path = account_dir.join("meta.json");
|
||||
if json_path.exists() {
|
||||
let data = std::fs::read_to_string(&json_path)?;
|
||||
let meta: Self = serde_json::from_str(&data)?;
|
||||
write_bin(&bin_path, &meta)?;
|
||||
let _ = std::fs::remove_file(&json_path);
|
||||
return Ok(meta);
|
||||
}
|
||||
Err(crate::error::Error::AccountNotFound(
|
||||
account_dir.to_string_lossy().into(),
|
||||
))
|
||||
}
|
||||
|
||||
pub fn save(&self, account_dir: &Path) -> Result<()> {
|
||||
write_bin(&account_dir.join("meta.bin"), self)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
fn test_global_meta_bin_roundtrip() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut meta = GlobalMeta::default();
|
||||
meta.accounts.push("alice".into());
|
||||
meta.save(dir.path()).unwrap();
|
||||
|
||||
let loaded = GlobalMeta::load(dir.path()).unwrap();
|
||||
assert_eq!(loaded.accounts, vec!["alice"]);
|
||||
assert!(!dir.path().join("global_meta.json").exists());
|
||||
assert!(dir.path().join("global_meta.bin").exists());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_global_meta_default_when_missing() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let meta = GlobalMeta::load(dir.path()).unwrap();
|
||||
assert!(meta.accounts.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_json_migration() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
// Write old JSON format
|
||||
let json = r#"{"version":1,"accounts":["bob","alice"]}"#;
|
||||
std::fs::write(dir.path().join("global_meta.json"), json).unwrap();
|
||||
|
||||
let meta = GlobalMeta::load(dir.path()).unwrap();
|
||||
// Should be sorted
|
||||
assert_eq!(meta.accounts, vec!["alice", "bob"]);
|
||||
// JSON should be removed
|
||||
assert!(!dir.path().join("global_meta.json").exists());
|
||||
// BIN should exist
|
||||
assert!(dir.path().join("global_meta.bin").exists());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_account_meta_bin_roundtrip() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut meta = AccountMeta::new("alice".into(), 1);
|
||||
meta.segments.insert(
|
||||
1,
|
||||
SegmentStats {
|
||||
segment_id: 1,
|
||||
total_bytes: 1000,
|
||||
deleted_bytes: 300,
|
||||
deleted_ratio: 0.3,
|
||||
sealed: false,
|
||||
indexed_up_to_offset: 0,
|
||||
},
|
||||
);
|
||||
meta.save(dir.path()).unwrap();
|
||||
|
||||
let loaded = AccountMeta::load(dir.path()).unwrap();
|
||||
assert_eq!(loaded.active_segment_id, 1);
|
||||
assert_eq!(loaded.segments[&1].total_bytes, 1000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_corrupt_bin_detected() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
std::fs::write(dir.path().join("meta.bin"), vec![0xFFu8; 100]).unwrap();
|
||||
let result = AccountMeta::load(dir.path());
|
||||
assert!(result.is_err());
|
||||
// 0xFFFFFFFF version triggers UnsupportedMetaVersion
|
||||
assert!(matches!(result.unwrap_err(), crate::error::Error::UnsupportedMetaVersion { .. }));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_crc_corruption_detected() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
// Write a well-formed header (version=1) but with wrong CRC bytes
|
||||
let mut buf = Vec::new();
|
||||
buf.extend_from_slice(&0xDEADBEEFu32.to_le_bytes()); // wrong CRC
|
||||
buf.extend_from_slice(&1u32.to_le_bytes()); // version = 1 (OK)
|
||||
buf.extend_from_slice(b"some payload bytes"); // payload
|
||||
std::fs::write(dir.path().join("meta.bin"), &buf).unwrap();
|
||||
let result = AccountMeta::load(dir.path());
|
||||
assert!(matches!(result.unwrap_err(), crate::error::Error::CorruptMeta(_)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_account_json_migration() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
// Write old JSON format for AccountMeta
|
||||
let json = r#"{"account_id":"alice","active_segment_id":5,"segments":{}}"#;
|
||||
std::fs::write(dir.path().join("meta.json"), json).unwrap();
|
||||
|
||||
let meta = AccountMeta::load(dir.path()).unwrap();
|
||||
assert_eq!(meta.account_id, "alice");
|
||||
assert_eq!(meta.active_segment_id, 5);
|
||||
// JSON should be removed
|
||||
assert!(!dir.path().join("meta.json").exists());
|
||||
// BIN should exist
|
||||
assert!(dir.path().join("meta.bin").exists());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bin_sorted_keys() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut meta = AccountMeta::new("test".into(), 1);
|
||||
meta.segments.insert(3, SegmentStats::new(3));
|
||||
meta.segments.insert(1, SegmentStats::new(1));
|
||||
meta.segments.insert(2, SegmentStats::new(2));
|
||||
meta.save(dir.path()).unwrap();
|
||||
|
||||
let loaded = AccountMeta::load(dir.path()).unwrap();
|
||||
let keys: Vec<u32> = loaded.segments.keys().copied().collect();
|
||||
assert_eq!(keys, vec![1, 2, 3]);
|
||||
}
|
||||
}
|
||||
202
crates/blob/src/recovery.rs
Normal file
202
crates/blob/src/recovery.rs
Normal file
@@ -0,0 +1,202 @@
|
||||
use std::collections::HashMap;
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
|
||||
use crate::bucket::{self, BucketFile, IndexRecord};
|
||||
use crate::error::Result;
|
||||
use crate::meta::{AccountMeta, SegmentStats};
|
||||
use crate::segment::{self, SegmentReader};
|
||||
|
||||
/// Recover an account after a crash: scan segments, repair indices, update stats.
|
||||
pub fn recover_account(account_dir: &Path) -> Result<AccountMeta> {
|
||||
let meta_bin = account_dir.join("meta.bin");
|
||||
let meta_json = account_dir.join("meta.json");
|
||||
let meta_exists = meta_bin.exists() || meta_json.exists();
|
||||
let mut meta = if meta_exists {
|
||||
AccountMeta::load(account_dir).unwrap_or_else(|_| {
|
||||
AccountMeta::new(
|
||||
account_dir
|
||||
.file_name()
|
||||
.unwrap_or_default()
|
||||
.to_string_lossy()
|
||||
.into(),
|
||||
1,
|
||||
)
|
||||
})
|
||||
} else {
|
||||
return Ok(AccountMeta::new(
|
||||
account_dir
|
||||
.file_name()
|
||||
.unwrap_or_default()
|
||||
.to_string_lossy()
|
||||
.into(),
|
||||
1,
|
||||
));
|
||||
};
|
||||
|
||||
// Discover all segment files on disk
|
||||
let seg_dir = account_dir.join("segments");
|
||||
if !seg_dir.exists() {
|
||||
fs::create_dir_all(&seg_dir)?;
|
||||
}
|
||||
|
||||
let mut disk_segments: Vec<u32> = Vec::new();
|
||||
if seg_dir.exists() {
|
||||
for entry in fs::read_dir(&seg_dir)? {
|
||||
let entry = entry?;
|
||||
let name = entry.file_name();
|
||||
let name_str = name.to_string_lossy();
|
||||
if name_str.ends_with(".seg") && !name_str.contains("temp_") {
|
||||
if let Some(id_str) = name_str.strip_suffix(".seg") {
|
||||
if let Ok(id) = id_str.parse::<u32>() {
|
||||
disk_segments.push(id);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
disk_segments.sort_unstable();
|
||||
|
||||
if disk_segments.is_empty() {
|
||||
meta.active_segment_id = 1;
|
||||
} else {
|
||||
let max_id = *disk_segments.last().unwrap();
|
||||
meta.active_segment_id = max_id;
|
||||
}
|
||||
|
||||
// Ensure buckets directory exists
|
||||
let buckets_dir = account_dir.join("buckets");
|
||||
fs::create_dir_all(&buckets_dir)?;
|
||||
|
||||
// For each segment, scan only the unindexed tail and update stats incrementally
|
||||
for &seg_id in &disk_segments {
|
||||
let seg_path = seg_dir.join(segment::segment_filename(seg_id));
|
||||
let file_size = fs::metadata(&seg_path)?.len();
|
||||
|
||||
// Preserve existing stats; start fresh if this is a newly discovered segment
|
||||
let mut stats = meta.segments.remove(&seg_id).unwrap_or_else(|| SegmentStats::new(seg_id));
|
||||
let is_sealed = seg_id != meta.active_segment_id;
|
||||
stats.sealed = is_sealed;
|
||||
|
||||
// Scan start: from last indexed offset. Clamp defensively.
|
||||
let scan_start = if stats.indexed_up_to_offset <= file_size {
|
||||
stats.indexed_up_to_offset
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
// If fully indexed, skip scanning entirely
|
||||
if scan_start >= file_size {
|
||||
meta.segments.insert(seg_id, stats);
|
||||
continue;
|
||||
}
|
||||
|
||||
let reader = SegmentReader::open(seg_path.clone(), seg_id)?;
|
||||
let mut new_records: HashMap<u16, Vec<IndexRecord>> = HashMap::new();
|
||||
|
||||
let truncation_point = reader.scan_entries(scan_start, |entry, offset| {
|
||||
let bid = bucket::bucket_id(&entry.key);
|
||||
let rec = IndexRecord::new(
|
||||
entry.key,
|
||||
seg_id,
|
||||
offset,
|
||||
entry.data.len() as u32,
|
||||
entry.flags,
|
||||
);
|
||||
new_records.entry(bid).or_default().push(rec);
|
||||
|
||||
stats.total_bytes += entry.data.len() as u64;
|
||||
if entry.is_tombstone() {
|
||||
stats.deleted_bytes += entry.raw_size as u64;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
})?;
|
||||
|
||||
// Merge new records into bucket files (only the newly discovered ones)
|
||||
for (bid, records) in &new_records {
|
||||
let bf = BucketFile::open(account_dir, *bid);
|
||||
bf.append_batch(records)?;
|
||||
}
|
||||
|
||||
// Truncate if tail corruption found
|
||||
if truncation_point < file_size {
|
||||
segment::truncate_segment(&seg_path, truncation_point)?;
|
||||
}
|
||||
|
||||
stats.indexed_up_to_offset = truncation_point;
|
||||
stats.recompute_ratio();
|
||||
meta.segments.insert(seg_id, stats);
|
||||
}
|
||||
|
||||
meta.save(account_dir)?;
|
||||
|
||||
Ok(meta)
|
||||
}
|
||||
|
||||
/// Clean up leftover temp files from interrupted GC.
|
||||
pub fn cleanup_temp_files(account_dir: &Path) -> Result<()> {
|
||||
let seg_dir = account_dir.join("segments");
|
||||
if seg_dir.exists() {
|
||||
for entry in fs::read_dir(&seg_dir)? {
|
||||
let entry = entry?;
|
||||
let name = entry.file_name();
|
||||
let name_str = name.to_string_lossy();
|
||||
if name_str.starts_with("temp_") {
|
||||
let path = entry.path();
|
||||
tracing::warn!("Removing leftover temp file: {:?}", path);
|
||||
fs::remove_file(&path)?;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Also cleanup temp bucket files
|
||||
let buckets_dir = account_dir.join("buckets");
|
||||
if buckets_dir.exists() {
|
||||
for entry in fs::read_dir(&buckets_dir)? {
|
||||
let entry = entry?;
|
||||
let name = entry.file_name();
|
||||
let name_str = name.to_string_lossy();
|
||||
if name_str.ends_with(".tmp") {
|
||||
let path = entry.path();
|
||||
tracing::warn!("Removing leftover temp bucket file: {:?}", path);
|
||||
fs::remove_file(&path)?;
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
fn test_recover_fresh_account() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let account_dir = dir.path().join("test");
|
||||
fs::create_dir_all(&account_dir).unwrap();
|
||||
|
||||
let meta = recover_account(&account_dir).unwrap();
|
||||
assert_eq!(meta.active_segment_id, 1);
|
||||
assert!(meta.segments.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_cleanup_temp_files() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let account_dir = dir.path().join("test");
|
||||
fs::create_dir_all(account_dir.join("segments")).unwrap();
|
||||
fs::create_dir_all(account_dir.join("buckets")).unwrap();
|
||||
fs::write(
|
||||
account_dir.join("segments").join("temp_ABC123.seg"),
|
||||
b"garbage",
|
||||
)
|
||||
.unwrap();
|
||||
fs::write(account_dir.join("buckets").join("00.idx.tmp"), b"garbage").unwrap();
|
||||
|
||||
cleanup_temp_files(&account_dir).unwrap();
|
||||
|
||||
assert!(!account_dir.join("segments").join("temp_ABC123.seg").exists());
|
||||
}
|
||||
}
|
||||
539
crates/blob/src/segment.rs
Normal file
539
crates/blob/src/segment.rs
Normal file
@@ -0,0 +1,539 @@
|
||||
use std::fs::{self, File};
|
||||
use std::io::{Read, Seek, SeekFrom, Write};
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::Mutex;
|
||||
|
||||
use crate::checksum;
|
||||
use crate::error::{Error, Result};
|
||||
use crate::fs as fs_util;
|
||||
use crate::types::{Codec, ENTRY_HEADER_SIZE, ENTRY_MAGIC, SEGMENT_MAX_SIZE};
|
||||
|
||||
/// In-memory representation of a stored entry.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Entry {
|
||||
pub flags: u8,
|
||||
pub codec: Codec,
|
||||
pub key: [u8; 32],
|
||||
pub raw_size: u32,
|
||||
pub data: Vec<u8>,
|
||||
}
|
||||
|
||||
impl Entry {
|
||||
/// Create a normal data entry.
|
||||
pub fn new(key: [u8; 32], raw_data: &[u8], flags: u8, codec: Codec) -> Self {
|
||||
Self {
|
||||
flags,
|
||||
codec,
|
||||
key,
|
||||
raw_size: raw_data.len() as u32,
|
||||
data: raw_data.to_vec(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Create a tombstone entry.
|
||||
pub fn tombstone(key: [u8; 32]) -> Self {
|
||||
Self {
|
||||
flags: 1,
|
||||
codec: Codec::None,
|
||||
key,
|
||||
raw_size: 0,
|
||||
data: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_tombstone(&self) -> bool {
|
||||
self.flags == 1
|
||||
}
|
||||
|
||||
/// Total on-disk size: header + data
|
||||
pub fn disk_size(&self) -> usize {
|
||||
ENTRY_HEADER_SIZE + self.data.len()
|
||||
}
|
||||
}
|
||||
|
||||
/// Write entries sequentially to a segment file.
|
||||
pub struct SegmentWriter {
|
||||
file: File,
|
||||
path: PathBuf,
|
||||
id: u32,
|
||||
bytes_written: u64,
|
||||
}
|
||||
|
||||
impl SegmentWriter {
|
||||
pub fn create(path: PathBuf, id: u32) -> Result<Self> {
|
||||
// Use create+truncate instead of create_new to avoid NFS O_EXCL issues.
|
||||
let file = File::create(&path)?;
|
||||
Ok(Self {
|
||||
file,
|
||||
path,
|
||||
id,
|
||||
bytes_written: 0,
|
||||
})
|
||||
}
|
||||
|
||||
pub fn open_append(path: PathBuf, id: u32) -> Result<Self> {
|
||||
let mut file = fs_util::open_write(&path)?;
|
||||
file.seek(SeekFrom::End(0))?;
|
||||
let bytes_written = file.stream_position()?;
|
||||
Ok(Self {
|
||||
file,
|
||||
path,
|
||||
id,
|
||||
bytes_written,
|
||||
})
|
||||
}
|
||||
|
||||
pub fn id(&self) -> u32 {
|
||||
self.id
|
||||
}
|
||||
|
||||
pub fn path(&self) -> &Path {
|
||||
&self.path
|
||||
}
|
||||
|
||||
pub fn bytes_written(&self) -> u64 {
|
||||
self.bytes_written
|
||||
}
|
||||
|
||||
pub fn is_full(&self) -> bool {
|
||||
self.bytes_written >= SEGMENT_MAX_SIZE
|
||||
}
|
||||
|
||||
/// Append an entry. Returns the offset where it was written.
|
||||
pub fn append(&mut self, entry: &Entry) -> Result<u64> {
|
||||
let offset = self.bytes_written;
|
||||
self.write_entry(entry)
|
||||
.map_err(|e| map_io_err(e, &self.path))?;
|
||||
Ok(offset)
|
||||
}
|
||||
|
||||
fn write_entry(&mut self, entry: &Entry) -> Result<()> {
|
||||
let data_size = entry.data.len() as u32;
|
||||
|
||||
// Write magic
|
||||
self.file.write_all(&ENTRY_MAGIC.to_le_bytes())?;
|
||||
|
||||
// CRC32 placeholder: write zeros, remember position
|
||||
let crc_pos = self.file.stream_position()?;
|
||||
self.file.write_all(&0u32.to_le_bytes())?;
|
||||
|
||||
// Write flags, codec, key, raw_size, data_size
|
||||
self.file.write_all(&[entry.flags])?;
|
||||
self.file.write_all(&[entry.codec as u8])?;
|
||||
self.file.write_all(&entry.key)?;
|
||||
self.file.write_all(&entry.raw_size.to_le_bytes())?;
|
||||
self.file.write_all(&data_size.to_le_bytes())?;
|
||||
|
||||
// Write data
|
||||
self.file.write_all(&entry.data)?;
|
||||
|
||||
// Calculate CRC32 over everything after the crc32 field
|
||||
let crc = {
|
||||
let mut hasher = checksum::CrcWriter::new();
|
||||
hasher.update(&[entry.flags]);
|
||||
hasher.update(&[entry.codec as u8]);
|
||||
hasher.update(&entry.key);
|
||||
hasher.update(&entry.raw_size.to_le_bytes());
|
||||
hasher.update(&data_size.to_le_bytes());
|
||||
hasher.update(&entry.data);
|
||||
hasher.finalize()
|
||||
};
|
||||
|
||||
// Seek back and write the real CRC32
|
||||
self.file.seek(SeekFrom::Start(crc_pos))?;
|
||||
self.file.write_all(&crc.to_le_bytes())?;
|
||||
|
||||
// Seek back to end
|
||||
self.file.seek(SeekFrom::End(0))?;
|
||||
|
||||
self.bytes_written += entry.disk_size() as u64;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn fsync(&self) -> Result<()> {
|
||||
self.file.sync_all().map_err(|e| {
|
||||
if e.kind() == std::io::ErrorKind::StorageFull {
|
||||
Error::DiskFull(format!("{}: {}", self.path.display(), e))
|
||||
} else {
|
||||
Error::Io(e)
|
||||
}
|
||||
})?;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Read entries from a segment file.
|
||||
pub struct SegmentReader {
|
||||
path: PathBuf,
|
||||
id: u32,
|
||||
}
|
||||
|
||||
impl SegmentReader {
|
||||
pub fn open(path: PathBuf, id: u32) -> Result<Self> {
|
||||
Ok(Self { path, id })
|
||||
}
|
||||
|
||||
pub fn id(&self) -> u32 {
|
||||
self.id
|
||||
}
|
||||
|
||||
pub fn path(&self) -> &Path {
|
||||
&self.path
|
||||
}
|
||||
|
||||
pub fn file_size(&self) -> Result<u64> {
|
||||
Ok(fs::metadata(&self.path)?.len())
|
||||
}
|
||||
|
||||
/// Read a single entry at the given offset. Returns the entry and the offset of the next entry.
|
||||
pub fn read_entry_at(&self, offset: u64) -> Result<(Entry, u64)> {
|
||||
let mut file = fs_util::open_read(&self.path)?;
|
||||
file.seek(SeekFrom::Start(offset))?;
|
||||
|
||||
// Read magic
|
||||
let mut magic_buf = [0u8; 4];
|
||||
file.read_exact(&mut magic_buf)?;
|
||||
let magic = u32::from_le_bytes(magic_buf);
|
||||
if magic != ENTRY_MAGIC {
|
||||
return Err(Error::CorruptEntry {
|
||||
path: self.path.clone(),
|
||||
offset,
|
||||
reason: format!("bad magic: 0x{:08X}", magic),
|
||||
});
|
||||
}
|
||||
|
||||
// Read CRC32
|
||||
let mut crc_buf = [0u8; 4];
|
||||
file.read_exact(&mut crc_buf)?;
|
||||
let stored_crc = u32::from_le_bytes(crc_buf);
|
||||
|
||||
// Read flags, codec
|
||||
let mut flags_buf = [0u8; 1];
|
||||
file.read_exact(&mut flags_buf)?;
|
||||
let flags = flags_buf[0];
|
||||
|
||||
let mut codec_buf = [0u8; 1];
|
||||
file.read_exact(&mut codec_buf)?;
|
||||
let codec = Codec::from_u8(codec_buf[0]).ok_or_else(|| Error::CorruptEntry {
|
||||
path: self.path.clone(),
|
||||
offset,
|
||||
reason: format!("unknown codec: {}", codec_buf[0]),
|
||||
})?;
|
||||
|
||||
// Read key, raw_size, data_size
|
||||
let mut key = [0u8; 32];
|
||||
file.read_exact(&mut key)?;
|
||||
|
||||
let mut raw_size_buf = [0u8; 4];
|
||||
file.read_exact(&mut raw_size_buf)?;
|
||||
let raw_size = u32::from_le_bytes(raw_size_buf);
|
||||
|
||||
let mut data_size_buf = [0u8; 4];
|
||||
file.read_exact(&mut data_size_buf)?;
|
||||
let data_size = u32::from_le_bytes(data_size_buf);
|
||||
|
||||
// Read data
|
||||
let mut data = vec![0u8; data_size as usize];
|
||||
file.read_exact(&mut data)?;
|
||||
|
||||
// Verify CRC32 (over everything after the crc32 field)
|
||||
let computed_crc = {
|
||||
let mut hasher = checksum::CrcWriter::new();
|
||||
hasher.update(&[flags]);
|
||||
hasher.update(&[codec as u8]);
|
||||
hasher.update(&key);
|
||||
hasher.update(&raw_size.to_le_bytes());
|
||||
hasher.update(&data_size.to_le_bytes());
|
||||
hasher.update(&data);
|
||||
hasher.finalize()
|
||||
};
|
||||
|
||||
if stored_crc != computed_crc {
|
||||
return Err(Error::CrcMismatch {
|
||||
path: self.path.clone(),
|
||||
offset,
|
||||
});
|
||||
}
|
||||
|
||||
let next_offset = offset + ENTRY_HEADER_SIZE as u64 + data_size as u64;
|
||||
|
||||
Ok((
|
||||
Entry {
|
||||
flags,
|
||||
codec,
|
||||
key,
|
||||
raw_size,
|
||||
data,
|
||||
},
|
||||
next_offset,
|
||||
))
|
||||
}
|
||||
|
||||
/// Read a single entry at the given offset using a pre-opened File (via Mutex).
|
||||
/// This avoids the per-read File::open cost for hot segments.
|
||||
pub fn read_entry_at_file(&self, offset: u64, file: &Mutex<File>) -> Result<(Entry, u64)> {
|
||||
|
||||
let mut file = file.lock().unwrap();
|
||||
|
||||
file.seek(SeekFrom::Start(offset))?;
|
||||
|
||||
// Read magic
|
||||
let mut magic_buf = [0u8; 4];
|
||||
file.read_exact(&mut magic_buf)?;
|
||||
let magic = u32::from_le_bytes(magic_buf);
|
||||
if magic != ENTRY_MAGIC {
|
||||
return Err(Error::CorruptEntry {
|
||||
path: self.path.clone(),
|
||||
offset,
|
||||
reason: format!("bad magic: 0x{:08X}", magic),
|
||||
});
|
||||
}
|
||||
|
||||
// Read CRC32
|
||||
let mut crc_buf = [0u8; 4];
|
||||
file.read_exact(&mut crc_buf)?;
|
||||
let stored_crc = u32::from_le_bytes(crc_buf);
|
||||
|
||||
// Read flags, codec
|
||||
let mut flags_buf = [0u8; 1];
|
||||
file.read_exact(&mut flags_buf)?;
|
||||
let flags = flags_buf[0];
|
||||
|
||||
let mut codec_buf = [0u8; 1];
|
||||
file.read_exact(&mut codec_buf)?;
|
||||
let codec = Codec::from_u8(codec_buf[0]).ok_or_else(|| Error::CorruptEntry {
|
||||
path: self.path.clone(),
|
||||
offset,
|
||||
reason: format!("unknown codec: {}", codec_buf[0]),
|
||||
})?;
|
||||
|
||||
// Read key, raw_size, data_size
|
||||
let mut key = [0u8; 32];
|
||||
file.read_exact(&mut key)?;
|
||||
|
||||
let mut raw_size_buf = [0u8; 4];
|
||||
file.read_exact(&mut raw_size_buf)?;
|
||||
let raw_size = u32::from_le_bytes(raw_size_buf);
|
||||
|
||||
let mut data_size_buf = [0u8; 4];
|
||||
file.read_exact(&mut data_size_buf)?;
|
||||
let data_size = u32::from_le_bytes(data_size_buf);
|
||||
|
||||
// Read data
|
||||
let mut data = vec![0u8; data_size as usize];
|
||||
file.read_exact(&mut data)?;
|
||||
|
||||
// Verify CRC32
|
||||
let computed_crc = {
|
||||
let mut hasher = crate::checksum::CrcWriter::new();
|
||||
hasher.update(&[flags]);
|
||||
hasher.update(&[codec as u8]);
|
||||
hasher.update(&key);
|
||||
hasher.update(&raw_size.to_le_bytes());
|
||||
hasher.update(&data_size.to_le_bytes());
|
||||
hasher.update(&data);
|
||||
hasher.finalize()
|
||||
};
|
||||
|
||||
if stored_crc != computed_crc {
|
||||
return Err(Error::CrcMismatch {
|
||||
path: self.path.clone(),
|
||||
offset,
|
||||
});
|
||||
}
|
||||
|
||||
let next_offset = offset + ENTRY_HEADER_SIZE as u64 + data_size as u64;
|
||||
|
||||
Ok((
|
||||
Entry {
|
||||
flags,
|
||||
codec,
|
||||
key,
|
||||
raw_size,
|
||||
data,
|
||||
},
|
||||
next_offset,
|
||||
))
|
||||
}
|
||||
|
||||
/// Read data portion of an entry (for pread-style reads when you already know offset + data_size).
|
||||
pub fn read_data(&self, offset: u64, data_size: u32) -> Result<Vec<u8>> {
|
||||
let mut file = fs_util::open_read(&self.path)?;
|
||||
// Skip magic(4) + crc32(4) + flags(1) + codec(1) + key(32) + raw_size(4) + data_size(4) = 50 bytes
|
||||
let data_start = offset + ENTRY_HEADER_SIZE as u64;
|
||||
file.seek(SeekFrom::Start(data_start))?;
|
||||
let mut buf = vec![0u8; data_size as usize];
|
||||
file.read_exact(&mut buf)?;
|
||||
Ok(buf)
|
||||
}
|
||||
|
||||
/// Read the full entry header + data for verification (used by recovery and GC).
|
||||
pub fn read_full_entry(&self, offset: u64, data_size: u32) -> Result<Vec<u8>> {
|
||||
let mut file = fs_util::open_read(&self.path)?;
|
||||
file.seek(SeekFrom::Start(offset))?;
|
||||
let total = ENTRY_HEADER_SIZE + data_size as usize;
|
||||
let mut buf = vec![0u8; total];
|
||||
file.read_exact(&mut buf)?;
|
||||
Ok(buf)
|
||||
}
|
||||
|
||||
/// Iterate over all valid entries in the segment, calling f for each.
|
||||
/// Stops when hitting a corrupt/incomplete entry at the tail.
|
||||
pub fn scan_entries<F>(&self, start_offset: u64, mut f: F) -> Result<u64>
|
||||
where
|
||||
F: FnMut(&Entry, u64) -> Result<()>,
|
||||
{
|
||||
let file_size = self.file_size()?;
|
||||
let mut offset = start_offset;
|
||||
|
||||
while offset + ENTRY_HEADER_SIZE as u64 <= file_size {
|
||||
match self.read_entry_at(offset) {
|
||||
Ok((entry, next)) => {
|
||||
f(&entry, offset)?;
|
||||
offset = next;
|
||||
}
|
||||
Err(Error::CrcMismatch { .. }) | Err(Error::CorruptEntry { .. }) => {
|
||||
// If near end of file (within one max entry), truncate
|
||||
if file_size - offset < ENTRY_HEADER_SIZE as u64 + 100 * 1024 * 1024 {
|
||||
// Likely a partial write at tail, stop here
|
||||
break;
|
||||
} else {
|
||||
return Err(Error::CorruptEntry {
|
||||
path: self.path.clone(),
|
||||
offset,
|
||||
reason: "mid-file corruption detected".into(),
|
||||
});
|
||||
}
|
||||
}
|
||||
Err(e) => return Err(e),
|
||||
}
|
||||
}
|
||||
|
||||
Ok(offset) // return the truncation point
|
||||
}
|
||||
}
|
||||
|
||||
/// Truncate a segment file to the given size.
|
||||
pub fn truncate_segment(path: &Path, size: u64) -> Result<()> {
|
||||
fs_util::truncate(path, size)
|
||||
}
|
||||
|
||||
/// Map an Error, converting Io(StorageFull) to DiskFull with path context.
|
||||
fn map_io_err(e: Error, path: &Path) -> Error {
|
||||
match e {
|
||||
Error::Io(io) if io.kind() == std::io::ErrorKind::StorageFull => {
|
||||
Error::DiskFull(format!("{}: {}", path.display(), io))
|
||||
}
|
||||
_ => e,
|
||||
}
|
||||
}
|
||||
|
||||
/// Segment file name from id: "00000001.seg"
|
||||
pub fn segment_filename(id: u32) -> String {
|
||||
format!("{:08}.seg", id)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
fn temp_segment_path(dir: &TempDir, id: u32) -> PathBuf {
|
||||
dir.path().join(segment_filename(id))
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_write_and_read_entry() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = temp_segment_path(&dir, 1);
|
||||
let key = [0xAAu8; 32];
|
||||
let data = b"hello world".to_vec();
|
||||
|
||||
let entry = Entry::new(key, &data, 0, Codec::None);
|
||||
{
|
||||
let mut writer = SegmentWriter::create(path.clone(), 1).unwrap();
|
||||
writer.append(&entry).unwrap();
|
||||
writer.fsync().unwrap();
|
||||
}
|
||||
|
||||
let reader = SegmentReader::open(path, 1).unwrap();
|
||||
let (read_entry, next) = reader.read_entry_at(0).unwrap();
|
||||
|
||||
assert_eq!(read_entry.key, key);
|
||||
assert_eq!(read_entry.data, data);
|
||||
assert_eq!(read_entry.flags, 0);
|
||||
assert_eq!(read_entry.raw_size, 11);
|
||||
assert!(next > 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_tombstone_entry() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = temp_segment_path(&dir, 1);
|
||||
let key = [0xBBu8; 32];
|
||||
|
||||
let entry = Entry::tombstone(key);
|
||||
{
|
||||
let mut writer = SegmentWriter::create(path.clone(), 1).unwrap();
|
||||
writer.append(&entry).unwrap();
|
||||
writer.fsync().unwrap();
|
||||
}
|
||||
|
||||
let reader = SegmentReader::open(path, 1).unwrap();
|
||||
let (read_entry, _) = reader.read_entry_at(0).unwrap();
|
||||
|
||||
assert!(read_entry.is_tombstone());
|
||||
assert_eq!(read_entry.data.len(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multiple_entries() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = temp_segment_path(&dir, 1);
|
||||
|
||||
let entries: Vec<_> = (0..10)
|
||||
.map(|i| {
|
||||
let mut key = [0u8; 32];
|
||||
key[0] = i;
|
||||
Entry::new(key, &vec![i; 100], 0, Codec::None)
|
||||
})
|
||||
.collect();
|
||||
|
||||
{
|
||||
let mut writer = SegmentWriter::create(path.clone(), 1).unwrap();
|
||||
for e in &entries {
|
||||
writer.append(e).unwrap();
|
||||
}
|
||||
writer.fsync().unwrap();
|
||||
}
|
||||
|
||||
let reader = SegmentReader::open(path, 1).unwrap();
|
||||
let mut offset = 0u64;
|
||||
for (i, expected) in entries.iter().enumerate() {
|
||||
let (entry, next) = reader.read_entry_at(offset).unwrap();
|
||||
assert_eq!(entry.key[0], i as u8);
|
||||
assert_eq!(entry.data, expected.data);
|
||||
offset = next;
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bad_magic_detected() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = temp_segment_path(&dir, 1);
|
||||
// Write garbage
|
||||
std::fs::write(&path, vec![0xFFu8; 100]).unwrap();
|
||||
|
||||
let reader = SegmentReader::open(path, 1).unwrap();
|
||||
let result = reader.read_entry_at(0);
|
||||
assert!(result.is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_full() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = temp_segment_path(&dir, 1);
|
||||
let writer = SegmentWriter::create(path, 1).unwrap();
|
||||
assert!(!writer.is_full());
|
||||
}
|
||||
}
|
||||
88
crates/blob/src/types.rs
Normal file
88
crates/blob/src/types.rs
Normal file
@@ -0,0 +1,88 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
/// Magic number for entry identification
|
||||
pub const ENTRY_MAGIC: u32 = 0xB3DB_0001;
|
||||
|
||||
/// Fixed header size: magic(4) + crc32(4) + flags(1) + codec(1) + key(32) + raw_size(4) + data_size(4)
|
||||
pub const ENTRY_HEADER_SIZE: usize = 50;
|
||||
|
||||
/// Index record size: key(32) + segment_id(4) + offset(8) + data_size(4) + flags(1) + _pad(3)
|
||||
pub const INDEX_RECORD_SIZE: usize = 52;
|
||||
|
||||
/// Maximum segment size (256 MB)
|
||||
pub const SEGMENT_MAX_SIZE: u64 = 256 * 1024 * 1024;
|
||||
|
||||
/// Number of hash buckets per account
|
||||
pub const BUCKET_COUNT: u16 = 16;
|
||||
|
||||
/// Maximum value size (100 MB)
|
||||
pub const MAX_VALUE_SIZE: usize = 100 * 1024 * 1024;
|
||||
|
||||
/// Default compression threshold (4 KB)
|
||||
pub const DEFAULT_COMPRESS_THRESHOLD: usize = 4096;
|
||||
|
||||
/// Default LRU bucket cache size
|
||||
pub const DEFAULT_LRU_BUCKET_COUNT: usize = 256;
|
||||
|
||||
/// Default GC deleted ratio threshold
|
||||
pub const DEFAULT_GC_DELETED_RATIO: f64 = 0.30;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub enum Codec {
|
||||
None = 0,
|
||||
Zstd = 1,
|
||||
Lz4 = 2,
|
||||
}
|
||||
|
||||
impl Codec {
|
||||
pub fn from_u8(v: u8) -> Option<Self> {
|
||||
match v {
|
||||
0 => Some(Codec::None),
|
||||
1 => Some(Codec::Zstd),
|
||||
2 => Some(Codec::Lz4),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Config {
|
||||
pub compress_threshold: usize,
|
||||
pub default_codec: Codec,
|
||||
pub compression_level: i32,
|
||||
pub lru_bucket_count: usize,
|
||||
pub gc_deleted_ratio: f64,
|
||||
}
|
||||
|
||||
impl Default for Config {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
compress_threshold: DEFAULT_COMPRESS_THRESHOLD,
|
||||
default_codec: Codec::Zstd,
|
||||
compression_level: 0,
|
||||
lru_bucket_count: DEFAULT_LRU_BUCKET_COUNT,
|
||||
gc_deleted_ratio: DEFAULT_GC_DELETED_RATIO,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Config {
|
||||
pub fn validate(&self) -> crate::error::Result<()> {
|
||||
if self.lru_bucket_count == 0 {
|
||||
return Err(crate::error::Error::InvalidConfig(
|
||||
"lru_bucket_count must be > 0".into(),
|
||||
));
|
||||
}
|
||||
if self.gc_deleted_ratio <= 0.0 || self.gc_deleted_ratio >= 1.0 {
|
||||
return Err(crate::error::Error::InvalidConfig(
|
||||
"gc_deleted_ratio must be in (0.0, 1.0)".into(),
|
||||
));
|
||||
}
|
||||
if self.compression_level < 0 {
|
||||
return Err(crate::error::Error::InvalidConfig(
|
||||
"compression_level must be >= 0".into(),
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
498
crates/blob/tests/acid_test.rs
Normal file
498
crates/blob/tests/acid_test.rs
Normal file
@@ -0,0 +1,498 @@
|
||||
/// Crash-consistency and ACID property tests for bichon-blob.
|
||||
///
|
||||
/// Since we can't kill the process mid-write in an inline test, we simulate crashes
|
||||
/// by dropping the Engine without calling any cleanup (close/drop is the "crash"),
|
||||
/// then re-opening and verifying recovery produced consistent state.
|
||||
///
|
||||
/// For true power-loss simulation, each test writes data, drops the engine abruptly,
|
||||
/// then reopens and verifies: no corruption, no lost committed data, no partial writes.
|
||||
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::thread;
|
||||
|
||||
use bichon_blob::{Codec, Config, Engine};
|
||||
use tempfile::TempDir;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn make_key(seed: u64) -> [u8; 32] {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..8].copy_from_slice(&seed.to_le_bytes());
|
||||
key
|
||||
}
|
||||
|
||||
fn make_value(size: usize) -> Vec<u8> {
|
||||
let pattern = b"The quick brown fox jumps over the lazy dog. ";
|
||||
let mut v = Vec::with_capacity(size);
|
||||
while v.len() < size {
|
||||
let rem = size - v.len();
|
||||
let n = rem.min(pattern.len());
|
||||
v.extend_from_slice(&pattern[..n]);
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Durability: committed data survives crash
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn test_durability_single_write_survives_crash() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let key = make_key(42);
|
||||
let value = make_value(8192);
|
||||
|
||||
// Write
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
} // <-- Engine dropped = simulated crash
|
||||
|
||||
// Recover
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, Some(value));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_durability_many_writes_survive_crash() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let n = 500;
|
||||
let value = make_value(2048);
|
||||
let mut keys = Vec::new();
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
for i in 0..n {
|
||||
let key = make_key(i as u64);
|
||||
keys.push(key);
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
} // crash
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
for (i, key) in keys.iter().enumerate() {
|
||||
let result = engine.read("alice", key).unwrap();
|
||||
assert_eq!(result, Some(value.clone()), "missing key at index {}", i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_durability_delete_survives_crash() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let key = make_key(99);
|
||||
let value = make_value(4096);
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
} // crash after write
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.delete("alice", &key).unwrap();
|
||||
} // crash after delete
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, None, "delete should persist across crash");
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Atomicity: no partial writes visible after crash
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn test_atomicity_no_partial_entries_after_crash() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
|
||||
// Write enough entries to fill part of a segment, then crash
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
let value = make_value(50_000); // big enough to notice
|
||||
for i in 0..200u64 {
|
||||
engine
|
||||
.write("alice", make_key(i), &value, Codec::None)
|
||||
.unwrap();
|
||||
}
|
||||
} // crash
|
||||
|
||||
// Recovery should clean up any partial tail entries and all committed
|
||||
// entries should be readable
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
let value = make_value(50_000);
|
||||
for i in 0..200u64 {
|
||||
let result = engine.read("alice", &make_key(i)).unwrap();
|
||||
assert_eq!(
|
||||
result,
|
||||
Some(value.clone()),
|
||||
"committed key {} should be intact",
|
||||
i
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_atomicity_crash_during_segment_roll() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let big_value = make_value(2 * 1024 * 1024); // 2 MB each entry
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
// Write enough to cross at least one segment boundary (256 MB)
|
||||
for i in 0..140u64 {
|
||||
engine
|
||||
.write("alice", make_key(i), &big_value, Codec::None)
|
||||
.unwrap();
|
||||
}
|
||||
} // crash mid-way or after multiple segments
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
// All committed writes (that returned Ok) must be readable
|
||||
for i in 0..140u64 {
|
||||
let result = engine.read("alice", &make_key(i)).unwrap();
|
||||
assert!(
|
||||
result.is_some(),
|
||||
"key {} should exist after segment roll recovery",
|
||||
i
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 3. Consistency: CRC detects corruption, no silent data loss
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn test_consistency_crc_detects_corruption() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let key = make_key(77);
|
||||
let value = make_value(8192);
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
// Corrupt the segment file by flipping a byte
|
||||
let seg_path = find_first_segment(dir.path(), "alice");
|
||||
let mut data = fs::read(&seg_path).unwrap();
|
||||
// Flip a byte in the data portion, not the header
|
||||
let flip_pos = data.len() - 100;
|
||||
data[flip_pos] ^= 0xFF;
|
||||
fs::write(&seg_path, &data).unwrap();
|
||||
|
||||
// Reading should detect CRC mismatch
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
let result = engine.read("alice", &key);
|
||||
// Either error or None is acceptable — never silently wrong data
|
||||
match result {
|
||||
Err(_) => {} // CRC mismatch detected — good
|
||||
Ok(None) => {} // index may point to truncated/removed data
|
||||
Ok(Some(v)) => {
|
||||
if v == value {
|
||||
panic!("CRC corruption was NOT detected — silent data corruption!");
|
||||
}
|
||||
// If value differs, index pointed elsewhere after recovery
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_consistency_corrupt_magic_truncated_on_recovery() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
for i in 0..10u64 {
|
||||
engine
|
||||
.write("alice", make_key(i), &make_value(4096), Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
// Append garbage to the segment file (simulating partial write from crash)
|
||||
let seg_path = find_first_segment(dir.path(), "alice");
|
||||
let mut data = fs::read(&seg_path).unwrap();
|
||||
let orig_len = data.len();
|
||||
// Append garbage that doesn't start with the magic number
|
||||
data.extend_from_slice(&[0xFF; 200]);
|
||||
fs::write(&seg_path, &data).unwrap();
|
||||
|
||||
// Recovery should truncate the garbage
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
// Verify committed data is still intact
|
||||
for i in 0..10u64 {
|
||||
let result = engine.read("alice", &make_key(i)).unwrap();
|
||||
assert!(result.is_some(), "committed key {} should survive tail truncation", i);
|
||||
}
|
||||
}
|
||||
|
||||
// Verify file was actually truncated
|
||||
let truncated_len = fs::metadata(&seg_path).unwrap().len();
|
||||
assert!(truncated_len <= orig_len as u64, "garbage should have been truncated");
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 4. Isolation: concurrent reader sees consistent snapshot
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn test_isolation_reader_sees_snapshot_not_partial_write() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Arc::new(Engine::open(dir.path(), Config::default()).unwrap());
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
// Pre-populate a known key
|
||||
let original_value = make_value(4096);
|
||||
let key = make_key(100);
|
||||
engine
|
||||
.write("alice", key, &original_value, Codec::Zstd)
|
||||
.unwrap();
|
||||
|
||||
let running = Arc::new(AtomicBool::new(true));
|
||||
let writer_done = Arc::new(AtomicBool::new(false));
|
||||
|
||||
// Spawn a writer that continuously overwrites the same key
|
||||
let writer_engine = engine.clone();
|
||||
let writer_running = running.clone();
|
||||
let writer_done_flag = writer_done.clone();
|
||||
let writer_key = key;
|
||||
|
||||
let writer = thread::spawn(move || {
|
||||
for i in 0..1000u64 {
|
||||
if !writer_running.load(Ordering::Relaxed) {
|
||||
break;
|
||||
}
|
||||
let val = make_value(4096 + (i as usize % 100));
|
||||
writer_engine
|
||||
.write("alice", writer_key, &val, Codec::Zstd)
|
||||
.unwrap();
|
||||
thread::yield_now();
|
||||
}
|
||||
writer_done_flag.store(true, Ordering::SeqCst);
|
||||
});
|
||||
|
||||
// Concurrent reader: reads should never panic or hang
|
||||
let reader_engine = engine.clone();
|
||||
let reader_running = running.clone();
|
||||
let reader = thread::spawn(move || {
|
||||
let mut reads = 0;
|
||||
while reads < 500 {
|
||||
if !reader_running.load(Ordering::Relaxed) && reads > 0 {
|
||||
break;
|
||||
}
|
||||
let result = reader_engine.read("alice", &key);
|
||||
match result {
|
||||
Ok(Some(_)) | Ok(None) => {} // OK
|
||||
Err(e) => {
|
||||
// Accept transient errors but report them
|
||||
eprintln!("reader saw error: {:?}", e);
|
||||
}
|
||||
}
|
||||
reads += 1;
|
||||
thread::yield_now();
|
||||
}
|
||||
});
|
||||
|
||||
reader.join().unwrap();
|
||||
running.store(false, Ordering::SeqCst);
|
||||
writer.join().unwrap();
|
||||
|
||||
// Final read should see the last committed value (not partial)
|
||||
let final_result = engine.read("alice", &key).unwrap();
|
||||
assert!(final_result.is_some(), "final read should find a value");
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Crash during GC: old data intact, no corruption
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn test_crash_during_gc_leaves_data_intact() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let value = make_value(500_000); // 500 KB each
|
||||
// Write enough entries and delete some to create GC candidate
|
||||
for i in 0..500u64 {
|
||||
engine
|
||||
.write("alice", make_key(i), &value, Codec::None)
|
||||
.unwrap();
|
||||
}
|
||||
// Delete ~40%
|
||||
for i in (0..500u64).step_by(5) {
|
||||
engine.delete("alice", &make_key(i)).unwrap();
|
||||
}
|
||||
// Single GC run (may or may not trigger)
|
||||
let _ = engine.gc("alice");
|
||||
} // crash after GC
|
||||
|
||||
// All non-deleted entries must still be readable
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
let value = make_value(500_000);
|
||||
for i in 0..500u64 {
|
||||
let key = make_key(i);
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
if i % 5 == 0 {
|
||||
// Deleted keys
|
||||
assert_eq!(result, None, "key {} should be deleted", i);
|
||||
} else {
|
||||
assert_eq!(
|
||||
result,
|
||||
Some(value.clone()),
|
||||
"key {} should survive GC+crash",
|
||||
i
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Multiple crash-reopen cycles (torture test)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn test_multiple_crash_reopen_cycles() {
|
||||
use std::collections::HashSet;
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let value = make_value(4096);
|
||||
let mut alive: HashSet<u64> = HashSet::new();
|
||||
|
||||
// Populate and crash
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
for i in 0..50u64 {
|
||||
engine
|
||||
.write("alice", make_key(i), &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
alive.insert(i);
|
||||
}
|
||||
}
|
||||
|
||||
// Reopen, verify all exist, write more, crash
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
for &k in &alive {
|
||||
assert!(engine.read("alice", &make_key(k)).unwrap().is_some());
|
||||
}
|
||||
for i in 100..150u64 {
|
||||
engine
|
||||
.write("alice", make_key(i), &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
alive.insert(i);
|
||||
}
|
||||
}
|
||||
|
||||
// Reopen, verify all exist, delete some, crash
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
for &k in &alive {
|
||||
assert!(engine.read("alice", &make_key(k)).unwrap().is_some());
|
||||
}
|
||||
for i in 0..10u64 {
|
||||
engine.delete("alice", &make_key(i)).unwrap();
|
||||
alive.remove(&i);
|
||||
}
|
||||
}
|
||||
|
||||
// Final reopen: survivors exist, deleted gone
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
for &k in &alive {
|
||||
assert!(engine.read("alice", &make_key(k)).unwrap().is_some(),
|
||||
"key {} should exist", k);
|
||||
}
|
||||
for i in 0..10u64 {
|
||||
assert_eq!(engine.read("alice", &make_key(i)).unwrap(), None,
|
||||
"key {} should be deleted", i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Account-level isolation
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn test_account_isolation_crash_one_account_does_not_affect_others() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
engine.create_account("bob").unwrap();
|
||||
|
||||
engine
|
||||
.write("alice", make_key(1), &make_value(4096), Codec::Zstd)
|
||||
.unwrap();
|
||||
engine
|
||||
.write("bob", make_key(1), &make_value(8192), Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
// Delete alice's account dir partially to simulate corruption
|
||||
// Then verify bob is intact
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
// Bob should be fine
|
||||
let result = engine.read("bob", &make_key(1)).unwrap();
|
||||
assert!(result.is_some(), "bob should be unaffected");
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn find_first_segment(store_root: &Path, account: &str) -> std::path::PathBuf {
|
||||
let seg_dir = store_root.join("accounts").join(account).join("segments");
|
||||
for entry in fs::read_dir(&seg_dir).unwrap() {
|
||||
let entry = entry.unwrap();
|
||||
let name = entry.file_name().to_string_lossy().into_owned();
|
||||
if name.ends_with(".seg") && !name.contains("temp_") {
|
||||
return entry.path();
|
||||
}
|
||||
}
|
||||
panic!("no segment found in {:?}", seg_dir);
|
||||
}
|
||||
497
crates/blob/tests/integration_test.rs
Normal file
497
crates/blob/tests/integration_test.rs
Normal file
@@ -0,0 +1,497 @@
|
||||
use bichon_blob::{Codec, Config, Engine};
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
fn test_create_and_list_accounts() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
|
||||
engine.create_account("alice").unwrap();
|
||||
engine.create_account("bob").unwrap();
|
||||
|
||||
let accounts = engine.list_accounts();
|
||||
assert!(accounts.contains(&"alice".to_string()));
|
||||
assert!(accounts.contains(&"bob".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_write_and_read() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let key = [0xAA; 32];
|
||||
let value = b"Hello, this is a test email!".to_vec();
|
||||
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, Some(value));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_read_missing_key() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let key = [0xFF; 32];
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_delete() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let key = [0xBB; 32];
|
||||
let value = b"Some email content".to_vec();
|
||||
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
engine.delete("alice", &key).unwrap();
|
||||
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_delete_account() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
engine.delete_account("alice").unwrap();
|
||||
|
||||
let accounts = engine.list_accounts();
|
||||
assert!(!accounts.contains(&"alice".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_small_value_not_compressed() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let key = [0xCC; 32];
|
||||
let value = b"hi"; // Smaller than 4KB threshold
|
||||
|
||||
engine
|
||||
.write("alice", key, value, Codec::Zstd)
|
||||
.unwrap();
|
||||
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, Some(value.to_vec()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_large_value() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let key = [0xDD; 32];
|
||||
let value = vec![b'X'; 100_000]; // 100KB
|
||||
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, Some(value));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multiple_keys() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let n = 100;
|
||||
for i in 0..n {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&(i as u32).to_le_bytes());
|
||||
let value = format!("email number {}", i).into_bytes();
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
for i in 0..n {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&(i as u32).to_le_bytes());
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, Some(format!("email number {}", i).into_bytes()));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_gc() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
// Write many entries
|
||||
let value = vec![b'Y'; 5000];
|
||||
let n = 100;
|
||||
|
||||
for i in 0..n {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&(i as u32).to_le_bytes());
|
||||
engine
|
||||
.write("alice", key, &value, Codec::None)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
// Delete even-numbered keys
|
||||
for i in (0..n).step_by(2) {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&(i as u32).to_le_bytes());
|
||||
engine.delete("alice", &key).unwrap();
|
||||
}
|
||||
|
||||
// Run GC
|
||||
let _result = engine.gc("alice").unwrap();
|
||||
|
||||
// Verify remaining keys still readable
|
||||
for i in (1..n).step_by(2) {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&(i as u32).to_le_bytes());
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, Some(value.clone()));
|
||||
}
|
||||
|
||||
// Deleted keys should not exist
|
||||
for i in (0..n).step_by(2) {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&(i as u32).to_le_bytes());
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, None);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_reopen_persistence() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let key = [0xEE; 32];
|
||||
let value = b"persistent data".to_vec();
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
engine
|
||||
.write("alice", key, &value, Codec::Zstd)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
// Reopen
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
let result = engine.read("alice", &key).unwrap();
|
||||
assert_eq!(result, Some(value));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_stats() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
engine
|
||||
.write("alice", [1u8; 32], b"hello", Codec::None)
|
||||
.unwrap();
|
||||
|
||||
let stats = engine.stats("alice").unwrap();
|
||||
assert!(stats.total_bytes > 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_batch_write() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let n = 50;
|
||||
let entries: Vec<_> = (0..n)
|
||||
.map(|i: u64| {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..8].copy_from_slice(&i.to_le_bytes());
|
||||
let value = format!("batch email {}", i).into_bytes();
|
||||
(key, value, Codec::Zstd)
|
||||
})
|
||||
.collect();
|
||||
|
||||
engine.write_batch("alice", &entries).unwrap();
|
||||
|
||||
for (key, value, _) in &entries {
|
||||
let result = engine.read("alice", key).unwrap();
|
||||
assert_eq!(result.as_ref(), Some(value));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_batch_write_persistence() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let entries: Vec<_> = (0..30u64)
|
||||
.map(|i| {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..8].copy_from_slice(&i.to_le_bytes());
|
||||
(key, format!("persist {}", i).into_bytes(), Codec::Zstd)
|
||||
})
|
||||
.collect();
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
engine.write_batch("alice", &entries).unwrap();
|
||||
}
|
||||
|
||||
{
|
||||
let engine = Engine::open(dir.path(), Config::default()).unwrap();
|
||||
for (key, value, _) in &entries {
|
||||
let result = engine.read("alice", key).unwrap();
|
||||
assert_eq!(result.as_ref(), Some(value));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_invalid_config_rejected() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let mut config = Config::default();
|
||||
config.lru_bucket_count = 0;
|
||||
assert!(Engine::open(dir.path(), config).is_err());
|
||||
|
||||
let mut config = Config::default();
|
||||
config.gc_deleted_ratio = 1.5;
|
||||
assert!(Engine::open(dir.path(), config).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_concurrent_reads() {
|
||||
use std::sync::Arc;
|
||||
use std::thread;
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Arc::new(Engine::open(dir.path(), Config::default()).unwrap());
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
// Write some data
|
||||
for i in 0..50u32 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&i.to_le_bytes());
|
||||
engine.write("alice", key, &vec![i as u8; 1024], Codec::None).unwrap();
|
||||
}
|
||||
|
||||
// Spawn 4 threads, each reading a different subset
|
||||
let mut handles = vec![];
|
||||
for t in 0..4 {
|
||||
let engine = engine.clone();
|
||||
handles.push(thread::spawn(move || {
|
||||
for i in (t * 12)..((t + 1) * 12) {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&(i as u32).to_le_bytes());
|
||||
let read = engine.read("alice", &key).unwrap();
|
||||
assert!(read.is_some(), "key {} should exist", i);
|
||||
}
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_concurrent_writes_different_accounts() {
|
||||
use std::sync::Arc;
|
||||
use std::thread;
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Arc::new(Engine::open(dir.path(), Config::default()).unwrap());
|
||||
|
||||
for name in &["alice", "bob", "carol"] {
|
||||
engine.create_account(name).unwrap();
|
||||
}
|
||||
|
||||
let mut handles = vec![];
|
||||
for (t, name) in ["alice", "bob", "carol"].iter().enumerate() {
|
||||
let engine = engine.clone();
|
||||
let account_name = name.to_string();
|
||||
handles.push(thread::spawn(move || {
|
||||
for i in 0..20 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&((t * 100 + i) as u32).to_le_bytes());
|
||||
let value = vec![(t * 100 + i) as u8; 512];
|
||||
engine.write(&account_name, key, &value, Codec::None).unwrap();
|
||||
}
|
||||
}));
|
||||
}
|
||||
for h in handles {
|
||||
h.join().unwrap();
|
||||
}
|
||||
|
||||
// Verify all writes persisted
|
||||
for (t, name) in ["alice", "bob", "carol"].iter().enumerate() {
|
||||
for i in 0..20 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&((t * 100 + i) as u32).to_le_bytes());
|
||||
let read = engine.read(name, &key).unwrap();
|
||||
assert!(read.is_some(), "account {} key {} should exist", name, i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_crash_recovery() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let dir_path = dir.path().to_path_buf();
|
||||
|
||||
// Phase 1: write data, then drop without shutdown (simulates crash)
|
||||
{
|
||||
let engine = Engine::open(&dir_path, Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
for i in 0..50u32 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&i.to_le_bytes());
|
||||
engine.write("alice", key, &vec![i as u8; 512], Codec::None).unwrap();
|
||||
}
|
||||
// Engine dropped here without calling shutdown()
|
||||
}
|
||||
|
||||
// Phase 2: reopen — recovery should run, data should be intact
|
||||
let engine = Engine::open(&dir_path, Config::default()).unwrap();
|
||||
let stats = engine.stats("alice").unwrap();
|
||||
assert!(stats.total_keys > 0, "recovery should preserve data");
|
||||
|
||||
// Verify reads work
|
||||
for i in 0..50u32 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&i.to_le_bytes());
|
||||
let read = engine.read("alice", &key).unwrap();
|
||||
assert!(read.is_some(), "key {} should survive crash recovery", i);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_meta_bin_durability() {
|
||||
// Verify meta.bin has valid CRC and can be read after a write cycle.
|
||||
let dir = TempDir::new().unwrap();
|
||||
let dir_path = dir.path().to_path_buf();
|
||||
|
||||
{
|
||||
let engine = Engine::open(&dir_path, Config::default()).unwrap();
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
let key = [0x42u8; 32];
|
||||
engine.write("alice", key, b"durable", Codec::None).unwrap();
|
||||
}
|
||||
// Engine dropped → shutdown() called → meta saved via write_bin (with fsync)
|
||||
|
||||
// Verify meta.bin exists and has valid CRC
|
||||
let meta_path = dir_path
|
||||
.join("accounts")
|
||||
.join("alice")
|
||||
.join("meta.bin");
|
||||
assert!(meta_path.exists(), "meta.bin should exist after clean shutdown");
|
||||
|
||||
let data = std::fs::read(&meta_path).unwrap();
|
||||
assert!(data.len() >= 8, "meta.bin should have at least 8 bytes (crc + version)");
|
||||
|
||||
let stored_crc = u32::from_le_bytes(data[0..4].try_into().unwrap());
|
||||
assert_ne!(stored_crc, 0, "stored CRC should be non-zero");
|
||||
|
||||
// Reopen and verify data is intact
|
||||
let engine = Engine::open(&dir_path, Config::default()).unwrap();
|
||||
let read = engine.read("alice", &[0x42u8; 32]).unwrap();
|
||||
assert_eq!(read, Some(b"durable".to_vec()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_gc_concurrent_with_writes() {
|
||||
// GC should not lose entries that are written concurrently.
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::thread;
|
||||
|
||||
let dir = TempDir::new().unwrap();
|
||||
let engine = Arc::new(Engine::open(dir.path(), Config::default()).unwrap());
|
||||
engine.create_account("alice").unwrap();
|
||||
|
||||
// Pre-fill: write enough to trigger eventual GC
|
||||
let big_value = vec![b'X'; 8192];
|
||||
for i in 0..500u32 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&i.to_le_bytes());
|
||||
engine.write("alice", key, &big_value, Codec::None).unwrap();
|
||||
}
|
||||
|
||||
// Delete some to create GC candidates
|
||||
for i in 0..250u32 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&i.to_le_bytes());
|
||||
engine.delete("alice", &key).unwrap();
|
||||
}
|
||||
|
||||
let running = Arc::new(AtomicBool::new(true));
|
||||
let engine_gc = engine.clone();
|
||||
let running_gc = running.clone();
|
||||
|
||||
// Thread 1: run GC in a loop
|
||||
let gc_handle = thread::spawn(move || {
|
||||
while running_gc.load(Ordering::Relaxed) {
|
||||
let _ = engine_gc.gc("alice");
|
||||
thread::sleep(std::time::Duration::from_millis(10));
|
||||
}
|
||||
});
|
||||
|
||||
// Thread 2: keep writing new entries
|
||||
let engine_write = engine.clone();
|
||||
let running_write = running.clone();
|
||||
let write_handle = thread::spawn(move || {
|
||||
let mut counter = 10000u32;
|
||||
while running_write.load(Ordering::Relaxed) {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&counter.to_le_bytes());
|
||||
engine_write
|
||||
.write("alice", key, &vec![counter as u8; 256], Codec::None)
|
||||
.unwrap();
|
||||
counter += 1;
|
||||
}
|
||||
counter
|
||||
});
|
||||
|
||||
// Let them race for a bit
|
||||
thread::sleep(std::time::Duration::from_millis(500));
|
||||
running.store(false, Ordering::Relaxed);
|
||||
|
||||
gc_handle.join().unwrap();
|
||||
let final_counter = write_handle.join().unwrap();
|
||||
|
||||
// All written entries must be readable
|
||||
let mut missing = 0;
|
||||
for i in 0..500u32 {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&i.to_le_bytes());
|
||||
if engine.read("alice", &key).unwrap().is_none() {
|
||||
// Entries 0..250 were deleted, they should be gone
|
||||
if i >= 250 {
|
||||
missing += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
assert_eq!(missing, 0, "pre-existing entries should survive concurrent GC");
|
||||
|
||||
// Entries written during the race should be readable
|
||||
for i in 10000..final_counter {
|
||||
let mut key = [0u8; 32];
|
||||
key[0..4].copy_from_slice(&i.to_le_bytes());
|
||||
let read = engine.read("alice", &key).unwrap();
|
||||
assert!(read.is_some(), "concurrently written key {} should exist after GC", i);
|
||||
}
|
||||
}
|
||||
23
crates/cli/Cargo.toml
Normal file
23
crates/cli/Cargo.toml
Normal file
@@ -0,0 +1,23 @@
|
||||
[package]
|
||||
name = "bichon-cli"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
|
||||
|
||||
[dependencies]
|
||||
bichon-core = { path = "../core" }
|
||||
tokio.workspace = true
|
||||
serde.workspace = true
|
||||
clap.workspace = true
|
||||
dialoguer.workspace = true
|
||||
console.workspace = true
|
||||
mail-parser.workspace = true
|
||||
reqwest.workspace = true
|
||||
toml = "0.9.8"
|
||||
memmap2 = "0.9.10"
|
||||
outlook-pst = { git = "https://github.com/rustmailer/outlook-pst-rs.git", branch = "main" }
|
||||
chrono.workspace = true
|
||||
base64.workspace = true
|
||||
sysinfo.workspace = true
|
||||
indicatif.workspace = true
|
||||
serde_json.workspace = true
|
||||
81
crates/cli/src/api/download.rs
Normal file
81
crates/cli/src/api/download.rs
Normal file
@@ -0,0 +1,81 @@
|
||||
use crate::BichonCliConfig;
|
||||
use bichon_core::{base64_encode, envelope::meta::BichonMetadata, store::envelope::Envelope};
|
||||
use chrono::{TimeZone, Utc};
|
||||
use reqwest::Client;
|
||||
use tokio::io::AsyncWriteExt;
|
||||
|
||||
pub async fn download_and_export_with_json_header(
|
||||
client: &Client,
|
||||
config: &BichonCliConfig,
|
||||
envelope: Envelope,
|
||||
file: &mut tokio::fs::File,
|
||||
) -> bool {
|
||||
let url = format!(
|
||||
"{}/api/v1/download-message/{}/{}",
|
||||
config.base_url, &envelope.account_id, &envelope.id
|
||||
);
|
||||
|
||||
let response = match client
|
||||
.get(&url)
|
||||
.header("Authorization", format!("Bearer {}", config.api_token))
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
Ok(res) => {
|
||||
if !res.status().is_success() {
|
||||
eprintln!(
|
||||
" ✘ HTTP Error {}: Failed for {}",
|
||||
res.status(),
|
||||
&envelope.id
|
||||
);
|
||||
return false;
|
||||
}
|
||||
res
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(" ✘ Network error: {} for {}", e, &envelope.id);
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
let email_bytes = match response.bytes().await {
|
||||
Ok(b) => b,
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
" ✘ Failed to read response body for {}: {}",
|
||||
&envelope.id, e
|
||||
);
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
let date_dt = Utc.timestamp_opt(envelope.date / 1000, 0).unwrap();
|
||||
let date_str = date_dt.format("%a %b %e %H:%M:%S %Y").to_string();
|
||||
let from_line = format!("From {} {}\n", envelope.from.clone(), date_str);
|
||||
|
||||
let custom_header = build_metadata_header(BichonMetadata {
|
||||
account_email: envelope.account_email,
|
||||
mailbox_name: envelope.mailbox_name,
|
||||
tags: envelope.tags,
|
||||
});
|
||||
|
||||
let mut final_buffer =
|
||||
Vec::with_capacity(from_line.len() + custom_header.len() + email_bytes.len() + 2);
|
||||
final_buffer.extend_from_slice(from_line.as_bytes());
|
||||
final_buffer.extend_from_slice(custom_header.as_bytes());
|
||||
final_buffer.extend_from_slice(&email_bytes);
|
||||
final_buffer.extend_from_slice(b"\n\n");
|
||||
|
||||
if let Err(e) = file.write_all(&final_buffer).await {
|
||||
eprintln!(" ✘ IO Error: Failed to write to mbox: {}", e);
|
||||
return false;
|
||||
}
|
||||
|
||||
true
|
||||
}
|
||||
|
||||
fn build_metadata_header(meta: BichonMetadata) -> String {
|
||||
let json_str = serde_json::to_string(&meta).ok().unwrap();
|
||||
let encoded = base64_encode!(json_str);
|
||||
format!("X-Bichon-Metadata: {}\r\n", encoded)
|
||||
}
|
||||
4
crates/cli/src/api/mod.rs
Normal file
4
crates/cli/src/api/mod.rs
Normal file
@@ -0,0 +1,4 @@
|
||||
pub mod download;
|
||||
pub mod search;
|
||||
pub mod sender;
|
||||
pub mod stats;
|
||||
58
crates/cli/src/api/search.rs
Normal file
58
crates/cli/src/api/search.rs
Normal file
@@ -0,0 +1,58 @@
|
||||
use bichon_core::{
|
||||
common::paginated::DataPage,
|
||||
message::search::{EmailSearchFilter, EmailSearchRequest, SortBy},
|
||||
store::envelope::Envelope,
|
||||
};
|
||||
use reqwest::Client;
|
||||
|
||||
use crate::BichonCliConfig;
|
||||
|
||||
pub async fn search_messages(
|
||||
client: &Client,
|
||||
config: &BichonCliConfig,
|
||||
account_ids: Option<std::collections::HashSet<u64>>,
|
||||
page: u64,
|
||||
page_size: u64,
|
||||
) -> Option<DataPage<Envelope>> {
|
||||
let url = format!("{}/api/v1/search-messages", config.base_url);
|
||||
|
||||
let payload = EmailSearchRequest {
|
||||
filter: EmailSearchFilter {
|
||||
account_ids,
|
||||
..Default::default()
|
||||
},
|
||||
page,
|
||||
page_size,
|
||||
sort_by: Some(SortBy::DATE),
|
||||
desc: Some(false),
|
||||
};
|
||||
|
||||
match client
|
||||
.post(&url)
|
||||
.header("Authorization", format!("Bearer {}", config.api_token))
|
||||
.json(&payload)
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
Ok(res) if res.status().is_success() => match res.json::<DataPage<Envelope>>().await {
|
||||
Ok(data) => Some(data),
|
||||
Err(e) => {
|
||||
eprintln!(" ✘ Failed to parse search response: {}", e);
|
||||
None
|
||||
}
|
||||
},
|
||||
Ok(res) => {
|
||||
let status = res.status();
|
||||
let error_body = res.text().await.unwrap_or_default();
|
||||
eprintln!(
|
||||
" ✘ Failed to search messages. Status: {}\n Server error: {}",
|
||||
status, error_body
|
||||
);
|
||||
None
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(" ✘ Network error performing search: {}", e);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
77
crates/cli/src/api/sender.rs
Normal file
77
crates/cli/src/api/sender.rs
Normal file
@@ -0,0 +1,77 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use console::style;
|
||||
use reqwest::Client;
|
||||
|
||||
use bichon_core::import::BatchEmlRequest;
|
||||
|
||||
use crate::BichonCliConfig;
|
||||
|
||||
pub async fn send_batch_request(
|
||||
client: &Client,
|
||||
config: &BichonCliConfig,
|
||||
account_id: u64,
|
||||
folder: &str,
|
||||
emls: Vec<String>,
|
||||
) {
|
||||
let url = format!("{}/api/v1/import", config.base_url);
|
||||
let payload = BatchEmlRequest {
|
||||
account_id,
|
||||
mail_folder: folder.to_string(),
|
||||
emls,
|
||||
};
|
||||
|
||||
let count = payload.emls.len();
|
||||
|
||||
match client
|
||||
.post(&url)
|
||||
.header("Authorization", format!("Bearer {}", config.api_token))
|
||||
.json(&payload)
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
Ok(res) if res.status().is_success() => {
|
||||
println!(
|
||||
" {} Sent {} emails to [{}]",
|
||||
style("✔").green(),
|
||||
count,
|
||||
folder
|
||||
);
|
||||
}
|
||||
Ok(res) => {
|
||||
let status = res.status();
|
||||
let error_body = res.text().await.unwrap_or_default();
|
||||
eprintln!(
|
||||
" {} Failed to send to [{}]. Status: {}\n Server error: {}",
|
||||
style("✘").red(),
|
||||
folder,
|
||||
status,
|
||||
error_body
|
||||
);
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
" {} Network error on [{}]: {}",
|
||||
style("✘").red(),
|
||||
folder,
|
||||
e
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
48
crates/cli/src/api/stats.rs
Normal file
48
crates/cli/src/api/stats.rs
Normal file
@@ -0,0 +1,48 @@
|
||||
use bichon_core::account::stats::AccountStats;
|
||||
use reqwest::Client;
|
||||
|
||||
use crate::BichonCliConfig;
|
||||
|
||||
pub async fn fetch_account_stats(
|
||||
client: &Client,
|
||||
config: &BichonCliConfig,
|
||||
account_id: u64,
|
||||
) -> Option<AccountStats> {
|
||||
let url = format!("{}/api/v1/accounts/{}/stats", config.base_url, account_id);
|
||||
|
||||
match client
|
||||
.get(&url)
|
||||
.header("Authorization", format!("Bearer {}", config.api_token))
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
Ok(res) if res.status().is_success() => {
|
||||
match res.json::<AccountStats>().await {
|
||||
Ok(stats) => Some(stats),
|
||||
Err(e) => {
|
||||
eprintln!(" ✘ Failed to parse stats response: {}", e);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(res) => {
|
||||
let status = res.status();
|
||||
let error_body = res.text().await.unwrap_or_default();
|
||||
eprintln!(
|
||||
" ✘ Failed to fetch stats for account [{}]. Status: {}\n Server error: {}",
|
||||
account_id,
|
||||
status,
|
||||
error_body
|
||||
);
|
||||
None
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
" ✘ Network error fetching stats for [{}]: {}",
|
||||
account_id,
|
||||
e
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,22 +1,43 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use std::process;
|
||||
|
||||
use console::style;
|
||||
use dialoguer::{theme::ColorfulTheme, Select};
|
||||
use reqwest::Client;
|
||||
|
||||
use crate::modules::{
|
||||
use bichon_core::{
|
||||
account::payload::MinimalAccount,
|
||||
cli::BichonCtlConfig,
|
||||
users::{permissions::Permission, view::UserView},
|
||||
};
|
||||
|
||||
pub async fn verify_user_and_get_account(config: &BichonCtlConfig, theme: &ColorfulTheme) -> u64 {
|
||||
let client = Client::new();
|
||||
let url = format!("{}/api/v1/current-user", config.base_url);
|
||||
use crate::BichonCliConfig;
|
||||
|
||||
async fn fetch_json<T: serde::de::DeserializeOwned>(
|
||||
client: &Client,
|
||||
url: &str,
|
||||
token: &str,
|
||||
label: &str,
|
||||
) -> T {
|
||||
let response = match client
|
||||
.get(&url)
|
||||
.header("Authorization", format!("Bearer {}", config.api_token))
|
||||
.get(url)
|
||||
.header("Authorization", format!("Bearer {}", token))
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
@@ -36,19 +57,15 @@ pub async fn verify_user_and_get_account(config: &BichonCtlConfig, theme: &Color
|
||||
}
|
||||
};
|
||||
|
||||
if !response.status().is_success() {
|
||||
let status = response.status();
|
||||
let error_body = response
|
||||
.text()
|
||||
.await
|
||||
.unwrap_or_else(|_| "No error detail provided".to_string());
|
||||
let status = response.status();
|
||||
let body = response.text().await.unwrap_or_else(|_| String::new());
|
||||
|
||||
if !status.is_success() {
|
||||
eprintln!(
|
||||
"\n{} Server returned an error (Status: {})",
|
||||
style("✘ API Error:").red().bold(),
|
||||
style(status).yellow()
|
||||
);
|
||||
|
||||
if status == 401 {
|
||||
eprintln!(
|
||||
"{} Your API Token seems to be invalid or expired.",
|
||||
@@ -60,36 +77,66 @@ pub async fn verify_user_and_get_account(config: &BichonCtlConfig, theme: &Color
|
||||
style("Context:").dim()
|
||||
);
|
||||
}
|
||||
|
||||
eprintln!("{} {}", style("Response:").dim(), error_body);
|
||||
eprintln!("{} {}", style("Response:").dim(), body);
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
let user: UserView = response.json().await.expect("Failed to parse user data");
|
||||
println!("Welcome, {}!", style(&user.username).cyan());
|
||||
|
||||
let account_list_url = format!(
|
||||
"{}/api/v1/minimal-account-list?only_nosync=true",
|
||||
config.base_url
|
||||
);
|
||||
let acc_response = client
|
||||
.get(&account_list_url)
|
||||
.header("Authorization", format!("Bearer {}", config.api_token))
|
||||
.send()
|
||||
.await
|
||||
.expect("Failed to fetch account list");
|
||||
|
||||
if !acc_response.status().is_success() {
|
||||
panic!(
|
||||
"Failed to retrieve accounts. Status: {}",
|
||||
acc_response.status()
|
||||
if body.is_empty() {
|
||||
eprintln!(
|
||||
"\n{} Server returned an empty response for [{}] (Status: {})",
|
||||
style("✘ Empty Response:").red().bold(),
|
||||
label,
|
||||
status
|
||||
);
|
||||
eprintln!(
|
||||
"{} This may be caused by a reverse proxy or middleware issue.",
|
||||
style("Tip:").cyan()
|
||||
);
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
let accounts: Vec<MinimalAccount> = acc_response
|
||||
.json()
|
||||
.await
|
||||
.expect("Failed to parse minimal account list");
|
||||
match serde_json::from_str::<T>(&body) {
|
||||
Ok(data) => data,
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
"\n{} Failed to parse response for [{}]: {}",
|
||||
style("✘ Parse Error:").red().bold(),
|
||||
label,
|
||||
e
|
||||
);
|
||||
eprintln!("{} Raw body: {}", style("Debug:").dim(), body);
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn verify_user_and_get_account(
|
||||
config: &BichonCliConfig,
|
||||
theme: &ColorfulTheme,
|
||||
only_nosync: bool,
|
||||
) -> MinimalAccount {
|
||||
let client = Client::new();
|
||||
|
||||
let user: UserView = fetch_json(
|
||||
&client,
|
||||
&format!("{}/api/v1/current-user", config.base_url),
|
||||
&config.api_token,
|
||||
"current-user",
|
||||
)
|
||||
.await;
|
||||
|
||||
println!("Welcome, {}!", style(&user.username).cyan());
|
||||
|
||||
let accounts: Vec<MinimalAccount> = fetch_json(
|
||||
&client,
|
||||
&format!(
|
||||
"{}/api/v1/minimal-account-list?only_nosync={only_nosync}",
|
||||
config.base_url
|
||||
),
|
||||
&config.api_token,
|
||||
"minimal-account-list",
|
||||
)
|
||||
.await;
|
||||
|
||||
if accounts.is_empty() {
|
||||
println!(
|
||||
@@ -106,6 +153,7 @@ pub async fn verify_user_and_get_account(config: &BichonCtlConfig, theme: &Color
|
||||
);
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
let required_permission = Permission::DATA_IMPORT_BATCH;
|
||||
let mut selectable_accounts = Vec::new();
|
||||
let mut options = Vec::new();
|
||||
@@ -165,5 +213,5 @@ pub async fn verify_user_and_get_account(config: &BichonCtlConfig, theme: &Color
|
||||
style(&selected_acc.email).cyan().bold()
|
||||
);
|
||||
|
||||
selected_acc.id
|
||||
selected_acc.clone()
|
||||
}
|
||||
@@ -1,3 +1,21 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use std::{
|
||||
collections::HashMap,
|
||||
fs,
|
||||
@@ -9,13 +27,12 @@ use dialoguer::{theme::ColorfulTheme, Input};
|
||||
use mail_parser::MessageParser;
|
||||
use reqwest::Client;
|
||||
|
||||
use crate::{
|
||||
base64_encode_url_safe,
|
||||
modules::cli::{sender::send_batch_request, BichonCtlConfig},
|
||||
};
|
||||
use bichon_core::base64_encode_url_safe;
|
||||
|
||||
use crate::{BichonCliConfig, api::sender::send_batch_request};
|
||||
|
||||
pub async fn handle_eml_directory_import(
|
||||
config: &BichonCtlConfig,
|
||||
config: &BichonCliConfig,
|
||||
account_id: u64,
|
||||
theme: &ColorfulTheme,
|
||||
) {
|
||||
@@ -83,7 +100,7 @@ fn scan_dir(
|
||||
}
|
||||
|
||||
async fn process_and_upload(
|
||||
config: &BichonCtlConfig,
|
||||
config: &BichonCliConfig,
|
||||
account_id: u64,
|
||||
tasks: HashMap<String, Vec<PathBuf>>,
|
||||
) {
|
||||
215
crates/cli/src/export/mod.rs
Normal file
215
crates/cli/src/export/mod.rs
Normal file
@@ -0,0 +1,215 @@
|
||||
use crate::api::download::download_and_export_with_json_header;
|
||||
use crate::api::search::search_messages;
|
||||
use crate::api::stats::fetch_account_stats;
|
||||
use crate::BichonCliConfig;
|
||||
use bichon_core::account::payload::MinimalAccount;
|
||||
use console::style;
|
||||
use dialoguer::Confirm;
|
||||
use dialoguer::{theme::ColorfulTheme, Input};
|
||||
use indicatif::{ProgressBar, ProgressStyle};
|
||||
use reqwest::Client;
|
||||
use std::path::{Path, PathBuf};
|
||||
use sysinfo::Disks;
|
||||
|
||||
pub async fn handle_account_export(
|
||||
config: &BichonCliConfig,
|
||||
account: MinimalAccount,
|
||||
theme: &ColorfulTheme,
|
||||
) {
|
||||
let client = Client::new();
|
||||
|
||||
println!("Fetching account statistics...");
|
||||
let stats = match fetch_account_stats(&client, config, account.id).await {
|
||||
Some(s) => s,
|
||||
None => {
|
||||
eprintln!("{} Failed to fetch account statistics.", style("✘").red());
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
println!("\n--- Account Statistics ---");
|
||||
println!(" Total Emails: {}", style(stats.total_count).cyan());
|
||||
println!(
|
||||
" Total Size: {}",
|
||||
style(format_bytes(stats.total_size)).cyan()
|
||||
);
|
||||
|
||||
let path = loop {
|
||||
let input: String = Input::with_theme(theme)
|
||||
.with_prompt("Enter ABSOLUTE directory path for MBOX file")
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
let p = PathBuf::from(&input);
|
||||
|
||||
if !p.is_absolute() {
|
||||
eprintln!(
|
||||
" {} {}",
|
||||
style("✘").red(),
|
||||
style("Invalid path: Must be an absolute path.").red()
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
if !p.exists() {
|
||||
eprintln!(
|
||||
" {} {}",
|
||||
style("✘").red(),
|
||||
style("Invalid path: Directory does not exist.").red()
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
if !p.is_dir() {
|
||||
eprintln!(
|
||||
" {} {}",
|
||||
style("✘").red(),
|
||||
style("Invalid path: The path provided is not a directory.").red()
|
||||
);
|
||||
continue;
|
||||
}
|
||||
break p;
|
||||
};
|
||||
|
||||
let disks = Disks::new_with_refreshed_list();
|
||||
let disk_result = disks
|
||||
.list()
|
||||
.iter()
|
||||
.find(|d| path.starts_with(d.mount_point()))
|
||||
.ok_or_else(|| "Could not identify the disk for the provided path.");
|
||||
|
||||
match disk_result {
|
||||
Ok(disk) => {
|
||||
let free_space = disk.available_space();
|
||||
let required_space = (stats.total_size as f64 * 1.2) as u64;
|
||||
|
||||
if free_space < required_space {
|
||||
eprintln!(
|
||||
" {} Insufficient disk space (including 10% safety buffer)!\n Required: {} (Base: {})\n Available: {}",
|
||||
style("✘").red(),
|
||||
style(format_bytes(required_space)).yellow(),
|
||||
style(format_bytes(stats.total_size)).yellow(),
|
||||
style(format_bytes(free_space)).yellow()
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
println!(
|
||||
" {} Disk space check passed. (Required: {}, Available: {})",
|
||||
style("✔").green(),
|
||||
style(format_bytes(required_space)).cyan(),
|
||||
style(format_bytes(free_space)).cyan()
|
||||
);
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(" {} {}", style("✘").red(), style(e).red());
|
||||
return;
|
||||
}
|
||||
}
|
||||
let mbox_file = get_unique_mbox_path(&path, account.id, &account.email);
|
||||
if Confirm::with_theme(theme)
|
||||
.with_prompt(format!(
|
||||
"Export {} emails to '{}'?",
|
||||
stats.total_count,
|
||||
mbox_file.display()
|
||||
))
|
||||
.default(true)
|
||||
.interact()
|
||||
.unwrap()
|
||||
{
|
||||
println!(
|
||||
" {} Starting export ({} items per page)...",
|
||||
style("✔").green(),
|
||||
100
|
||||
);
|
||||
|
||||
let pb = ProgressBar::new(stats.total_count as u64);
|
||||
pb.set_style(ProgressStyle::with_template(
|
||||
"{spinner:.green} [{elapsed_precise}] [{bar:40.cyan/blue}] {pos}/{len} ({eta}) {msg}"
|
||||
).unwrap());
|
||||
|
||||
let mut file = match tokio::fs::OpenOptions::new()
|
||||
.append(true)
|
||||
.create(true)
|
||||
.open(&mbox_file)
|
||||
.await
|
||||
{
|
||||
Ok(f) => f,
|
||||
Err(e) => {
|
||||
eprintln!(" ✘ Failed to open file '{}': {}", path.display(), e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let page_size = 100;
|
||||
let mut current_page = 1;
|
||||
let mut total_pages;
|
||||
|
||||
loop {
|
||||
let account_ids = Some(std::collections::HashSet::from([account.id]));
|
||||
if let Some(batch) = search_messages(&client, config, account_ids, current_page, page_size).await {
|
||||
total_pages = batch.total_pages.unwrap();
|
||||
|
||||
pb.set_message(format!("Page {}/{}", current_page, total_pages));
|
||||
|
||||
for envelope in batch.items {
|
||||
let success =
|
||||
download_and_export_with_json_header(&client, config, envelope.clone(), &mut file)
|
||||
.await;
|
||||
|
||||
if !success {
|
||||
eprintln!(
|
||||
" ✘ Failed to export email {}, skipping...",
|
||||
envelope.id
|
||||
);
|
||||
continue;
|
||||
}
|
||||
pb.inc(1);
|
||||
}
|
||||
if current_page >= total_pages {
|
||||
break;
|
||||
}
|
||||
current_page += 1;
|
||||
} else {
|
||||
pb.finish_with_message("Error");
|
||||
eprintln!(
|
||||
" ✘ Failed to fetch page {}. Aborting process...",
|
||||
current_page
|
||||
);
|
||||
return;
|
||||
}
|
||||
}
|
||||
pb.finish();
|
||||
println!(" {} Export complete!", style("✔").green());
|
||||
}
|
||||
}
|
||||
|
||||
fn format_bytes(bytes: u64) -> String {
|
||||
if bytes < 1024 {
|
||||
format!("{:.2} B", bytes)
|
||||
} else if bytes < 1024 * 1024 {
|
||||
format!("{:.2} KB", bytes / 1024)
|
||||
} else if bytes < 1024 * 1024 * 1024 {
|
||||
format!("{:.2} MB", bytes / 1024 / 1024)
|
||||
} else {
|
||||
format!("{:.2} GB", bytes / 1024 / 1024 / 1024)
|
||||
}
|
||||
}
|
||||
|
||||
fn get_unique_mbox_path(base_dir: &Path, account_id: u64, email: &str) -> PathBuf {
|
||||
let email_part = email.replace(' ', "_");
|
||||
|
||||
let mut base_name = format!("account_{}_{}", account_id, email_part);
|
||||
if base_name.starts_with('.') {
|
||||
base_name = format!("_{}", base_name);
|
||||
}
|
||||
|
||||
let mut final_path = base_dir.join(format!("{}.mbox", base_name));
|
||||
|
||||
let mut counter = 1;
|
||||
while final_path.exists() {
|
||||
final_path = base_dir.join(format!("{}_{}.mbox", base_name, counter));
|
||||
counter += 1;
|
||||
}
|
||||
final_path
|
||||
}
|
||||
172
crates/cli/src/main.rs
Normal file
172
crates/cli/src/main.rs
Normal file
@@ -0,0 +1,172 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use bichon_core::bichon_version;
|
||||
use clap::Parser;
|
||||
use console::style;
|
||||
use dialoguer::{theme::ColorfulTheme, Confirm, Input, Select};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::fs;
|
||||
|
||||
use crate::{
|
||||
auth::verify_user_and_get_account, eml::handle_eml_directory_import,
|
||||
export::handle_account_export, mbox::handle_mbox_single_file_import, pst::handle_pst_import,
|
||||
thunderbird::handle_thunderbird_import,
|
||||
};
|
||||
|
||||
pub mod api;
|
||||
pub mod auth;
|
||||
pub mod eml;
|
||||
pub mod export;
|
||||
pub mod mbox;
|
||||
pub mod pst;
|
||||
pub mod thunderbird;
|
||||
|
||||
#[derive(Parser, Debug)]
|
||||
#[command(
|
||||
name = "bichon-cli",
|
||||
author = "rustmailer",
|
||||
version = bichon_version!(),
|
||||
about = "A CLI tool to import email data into Bichon service"
|
||||
)]
|
||||
pub struct BichonCli {
|
||||
/// Path to the configuration file
|
||||
#[arg(
|
||||
short,
|
||||
long,
|
||||
default_value = "config.toml",
|
||||
value_name = "FILE",
|
||||
help = "Sets a custom config file"
|
||||
)]
|
||||
pub config: std::path::PathBuf,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||
pub struct BichonCliConfig {
|
||||
pub base_url: String,
|
||||
pub api_token: String,
|
||||
}
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() {
|
||||
let cli = BichonCli::parse();
|
||||
let theme = ColorfulTheme::default();
|
||||
let config_path = &cli.config;
|
||||
let mut current_config: Option<BichonCliConfig> = None;
|
||||
|
||||
if config_path.exists() {
|
||||
if let Ok(content) = fs::read_to_string(config_path) {
|
||||
if let Ok(config) = toml::from_str::<BichonCliConfig>(&content) {
|
||||
println!("{}", style("✔ Existing configuration found:").green());
|
||||
println!(" Base URL: {}", style(&config.base_url).yellow());
|
||||
println!(" API Token: {}", style(&config.api_token).yellow());
|
||||
|
||||
// Confirm with user
|
||||
if Confirm::with_theme(&theme)
|
||||
.with_prompt("Do you want to use this configuration?")
|
||||
.default(true)
|
||||
.interact()
|
||||
.unwrap()
|
||||
{
|
||||
current_config = Some(config);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let final_config = match current_config {
|
||||
Some(conf) => conf,
|
||||
None => {
|
||||
println!("\n{}", style("Please enter Bichon service details:").bold());
|
||||
|
||||
let url: String = Input::with_theme(&theme)
|
||||
.with_prompt("Bichon Base URL")
|
||||
.default("http://localhost:15630".into())
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
let token: String = Input::with_theme(&theme)
|
||||
.with_prompt("API Token")
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
let conf = BichonCliConfig {
|
||||
base_url: url,
|
||||
api_token: token,
|
||||
};
|
||||
|
||||
// 3. Offer to save the new configuration
|
||||
if Confirm::with_theme(&theme)
|
||||
.with_prompt("Save this configuration for future use?")
|
||||
.default(true)
|
||||
.interact()
|
||||
.unwrap()
|
||||
{
|
||||
let toml_str = toml::to_string(&conf).unwrap();
|
||||
fs::write(config_path, toml_str).expect("Failed to save config file");
|
||||
println!("{}", style("Configuration saved successfully!").green());
|
||||
}
|
||||
conf
|
||||
}
|
||||
};
|
||||
|
||||
let operations = &[
|
||||
"1. Import: Upload email data to Bichon",
|
||||
"2. Export: Download account data as MBOX file",
|
||||
];
|
||||
|
||||
let op_idx = Select::with_theme(&theme)
|
||||
.with_prompt("Select operation")
|
||||
.items(operations)
|
||||
.default(0)
|
||||
.interact()
|
||||
.unwrap();
|
||||
|
||||
match op_idx {
|
||||
0 => {
|
||||
let target_account = verify_user_and_get_account(&final_config, &theme, true).await;
|
||||
|
||||
let import_modes = &[
|
||||
"1. EML: Scan directory recursively (Maintains folder structure)",
|
||||
"2. MBOX: Single archive file (Stream from one file)",
|
||||
"3. Thunderbird: Import from local profile directory",
|
||||
"4. PST: Outlook Personal Storage (Single .pst file)",
|
||||
];
|
||||
|
||||
let mode_idx = Select::with_theme(&theme)
|
||||
.with_prompt("Select import method")
|
||||
.items(import_modes)
|
||||
.default(0)
|
||||
.interact()
|
||||
.unwrap();
|
||||
|
||||
match mode_idx {
|
||||
0 => handle_eml_directory_import(&final_config, target_account.id, &theme).await,
|
||||
1 => handle_mbox_single_file_import(&final_config, target_account.id, &theme).await,
|
||||
2 => handle_thunderbird_import(&final_config, target_account.id, &theme).await,
|
||||
3 => handle_pst_import(&final_config, target_account.id, &theme).await,
|
||||
_ => unreachable!(),
|
||||
}
|
||||
}
|
||||
1 => {
|
||||
let target_account = verify_user_and_get_account(&final_config, &theme, false).await;
|
||||
handle_account_export(&final_config, target_account, &theme).await;
|
||||
}
|
||||
_ => unreachable!(),
|
||||
}
|
||||
}
|
||||
134
crates/cli/src/mbox/gmail.rs
Normal file
134
crates/cli/src/mbox/gmail.rs
Normal file
@@ -0,0 +1,134 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use std::collections::HashSet;
|
||||
|
||||
pub fn determine_folder(labels_raw: &str) -> String {
|
||||
let mut status_blacklist = HashSet::new();
|
||||
status_blacklist.insert("Opened");
|
||||
status_blacklist.insert("Unread");
|
||||
status_blacklist.insert("Archived");
|
||||
|
||||
let all_labels: Vec<&str> = labels_raw
|
||||
.split(',')
|
||||
.map(|s| s.trim())
|
||||
.filter(|s| !s.is_empty())
|
||||
.collect();
|
||||
|
||||
if all_labels.is_empty() {
|
||||
return "Unknown".to_string();
|
||||
}
|
||||
|
||||
let filtered: Vec<&str> = all_labels
|
||||
.iter()
|
||||
.filter(|&&l| !status_blacklist.contains(l))
|
||||
.cloned()
|
||||
.collect();
|
||||
|
||||
match filtered.len() {
|
||||
// Case A: If all labels were status labels, fallback to the first original label
|
||||
0 => all_labels[0].to_string(),
|
||||
// Case B: If only one label remains, that's our target destination
|
||||
1 => filtered[0].to_string(),
|
||||
// Case C: Multiple labels remain (e.g., ["Inbox", "medium"])
|
||||
_ => {
|
||||
// Prioritize custom business labels by excluding generic locations like "Inbox" or "Sent"
|
||||
let business_label = filtered.iter().find(|&&l| l != "Inbox" && l != "Sent");
|
||||
|
||||
match business_label {
|
||||
// Return the first non-generic label found
|
||||
Some(label) => label.to_string(),
|
||||
// If only generic labels remain (e.g., ["Sent", "Inbox"]), pick the first available
|
||||
None => filtered[0].to_string(),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use mail_parser::{HeaderValue, MessageParser};
|
||||
|
||||
use super::*;
|
||||
|
||||
fn parse_x_gmail_labels(raw_message: &[u8]) -> Option<String> {
|
||||
// MessageParser::new() has an empty header_map so the hardcoded match at
|
||||
// parsers/header.rs:76 treats ALL unknown headers as raw (no RFC 2047
|
||||
// decoding). We need three things to get decoding:
|
||||
// 1. A non-empty header_map (so the else branch runs)
|
||||
// 2. default_header_text() so the fallback fn is parse_unstructured
|
||||
// 3. OR register X-Gmail-Labels explicitly via header_text()
|
||||
let message = MessageParser::new()
|
||||
.with_minimal_headers()
|
||||
.default_header_text()
|
||||
.parse(raw_message)?;
|
||||
let value: &HeaderValue<'_> = message.header("X-Gmail-Labels")?;
|
||||
value.as_text().map(|s| s.to_string())
|
||||
}
|
||||
|
||||
/// Construct a raw MIME message with RFC 2047 encoded X-Gmail-Labels,
|
||||
/// parse it, and verify the header is correctly decoded.
|
||||
fn build_email(x_gmail_labels: &str) -> Vec<u8> {
|
||||
format!(
|
||||
"From: sender@example.com\r\n\
|
||||
To: recipient@example.com\r\n\
|
||||
Subject: Test\r\n\
|
||||
X-Gmail-Labels: {}\r\n\
|
||||
\r\n\
|
||||
Body text here.\r\n",
|
||||
x_gmail_labels
|
||||
)
|
||||
.into_bytes()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rfc2047_encoded_labels_are_decoded() {
|
||||
// Exactly the format the user reported: French Gmail labels
|
||||
let raw = build_email("=?UTF-8?Q?Corbeille?=, =?UTF-8?Q?Messages_archiv=C3=A9s?=");
|
||||
let labels = parse_x_gmail_labels(&raw).expect("failed to parse X-Gmail-Labels");
|
||||
|
||||
// mail-parser decodes RFC 2047 header values during initial parsing.
|
||||
// The decoded text should NOT contain raw =?UTF-8?Q?... sequences.
|
||||
assert!(!labels.contains("=?UTF-8"), "labels still encoded: {labels:?}");
|
||||
assert!(labels.contains("Corbeille"), "missing 'Corbeille': {labels:?}");
|
||||
assert!(
|
||||
labels.contains("archivés"),
|
||||
"missing decoded 'archivés': {labels:?}",
|
||||
);
|
||||
|
||||
// Full pipeline: decoded labels → determine_folder
|
||||
let folder = determine_folder(&labels);
|
||||
assert_eq!(folder, "Corbeille");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn plain_ascii_labels_passthrough() {
|
||||
let raw = build_email("Inbox, Important");
|
||||
let labels = parse_x_gmail_labels(&raw).expect("failed to parse X-Gmail-Labels");
|
||||
assert_eq!(labels, "Inbox, Important");
|
||||
assert_eq!(determine_folder(&labels), "Important");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_x_gmail_labels_header() {
|
||||
let raw = b"From: sender@example.com\r\nTo: r@example.com\r\n\r\nBody.\r\n";
|
||||
let message = MessageParser::new().parse(raw.as_slice()).unwrap();
|
||||
assert!(message.header("X-Gmail-Labels").is_none());
|
||||
}
|
||||
}
|
||||
418
crates/cli/src/mbox/mod.rs
Normal file
418
crates/cli/src/mbox/mod.rs
Normal file
@@ -0,0 +1,418 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::path::PathBuf;
|
||||
|
||||
use crate::api::sender::send_batch_request;
|
||||
use crate::mbox::gmail::determine_folder;
|
||||
use bichon_core::import::reader::MboxFile;
|
||||
use crate::BichonCliConfig;
|
||||
use bichon_core::base64_encode_url_safe;
|
||||
use bichon_core::envelope::meta::{parse_bichon_metadata, BichonMetadata};
|
||||
use console::style;
|
||||
use dialoguer::{theme::ColorfulTheme, Input};
|
||||
use dialoguer::{Confirm, Select};
|
||||
use mail_parser::MessageParser;
|
||||
use reqwest::Client;
|
||||
|
||||
/// Skip emails larger than this with a warning (100 MB).
|
||||
const MAX_EMAIL_BYTES: usize = 100 * 1024 * 1024;
|
||||
/// Flush a folder buffer when accumulated base64 bytes exceed this (200 MB).
|
||||
const MAX_BUFFER_BYTES: usize = 200 * 1024 * 1024;
|
||||
|
||||
pub mod gmail;
|
||||
|
||||
pub async fn handle_mbox_single_file_import(
|
||||
config: &BichonCliConfig,
|
||||
account_id: u64,
|
||||
theme: &ColorfulTheme,
|
||||
) {
|
||||
let path_str: String = Input::with_theme(theme)
|
||||
.with_prompt("Enter the path to your SINGLE .mbox file")
|
||||
.validate_with(|input: &String| {
|
||||
let p = std::path::Path::new(input);
|
||||
if !p.exists() {
|
||||
return Err("The specified path does not exist.");
|
||||
}
|
||||
if !p.is_file() {
|
||||
return Err("MBOX mode requires a SINGLE file, not a directory.");
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
let mbox_path = PathBuf::from(path_str);
|
||||
|
||||
let options = vec![
|
||||
"Use labels from mail headers (X-Gmail-Labels)",
|
||||
"Specify a single target folder for all emails",
|
||||
"Use X-Bichon-Metadata header (Automatic)",
|
||||
];
|
||||
|
||||
let selection = Select::with_theme(theme)
|
||||
.with_prompt("How should we determine the target folder?")
|
||||
.items(&options)
|
||||
.default(0)
|
||||
.interact()
|
||||
.unwrap();
|
||||
|
||||
let target_folder: Option<String> = match selection {
|
||||
0 => None,
|
||||
1 => {
|
||||
let folder: String = Input::with_theme(theme)
|
||||
.with_prompt("Target folder name")
|
||||
.default("INBOX".into())
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
Some(folder)
|
||||
}
|
||||
2 => None,
|
||||
_ => unreachable!(),
|
||||
};
|
||||
|
||||
if let Some(ref folder) = target_folder {
|
||||
println!(
|
||||
"{}",
|
||||
style(format!("Mode: Fixed folder ({})", folder)).dim()
|
||||
);
|
||||
} else {
|
||||
println!("{}", style("Mode: Dynamic (header-based)").dim());
|
||||
}
|
||||
|
||||
println!(
|
||||
"\n{} Ready to process MBOX file: {}",
|
||||
style("✔").green(),
|
||||
style(mbox_path.display()).cyan()
|
||||
);
|
||||
|
||||
if let Ok(meta) = std::fs::metadata(&mbox_path) {
|
||||
let size_mb = meta.len() as f64 / 1024.0 / 1024.0;
|
||||
println!(
|
||||
"{}",
|
||||
style(format!("Processing file: {:.1} MB", size_mb)).dim()
|
||||
);
|
||||
}
|
||||
|
||||
if Confirm::with_theme(theme)
|
||||
.with_prompt("Start importing?")
|
||||
.default(true)
|
||||
.interact()
|
||||
.unwrap()
|
||||
{
|
||||
run_import(account_id, &mbox_path, config, target_folder).await
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn run_import(
|
||||
account_id: u64,
|
||||
mbox_path: &PathBuf,
|
||||
config: &BichonCliConfig,
|
||||
target_folder: Option<String>,
|
||||
) {
|
||||
let client = Client::new();
|
||||
let mbox = match MboxFile::from_file(mbox_path) {
|
||||
Ok(mbox) => mbox,
|
||||
Err(err) => {
|
||||
println!("Skipping invalid MBOX: {} ({})", mbox_path.display(), err);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let mut folder_buffers: HashMap<String, Vec<String>> = HashMap::new();
|
||||
let mut total_buffered_bytes: usize = 0;
|
||||
let batch_limit = 50;
|
||||
let mut skipped_count: u64 = 0;
|
||||
|
||||
println!("Starting import process...");
|
||||
|
||||
for (index, e) in mbox.iter().enumerate() {
|
||||
let msg_num = index + 1;
|
||||
let body = e.data;
|
||||
|
||||
if body.len() > MAX_EMAIL_BYTES {
|
||||
let size_mb = body.len() as f64 / 1024.0 / 1024.0;
|
||||
eprintln!(
|
||||
"{} {}: email #{} is {:.1} MB (limit 100 MB). Skipping...",
|
||||
style("Warning").yellow().bold(),
|
||||
style(format!("oversized")).dim(),
|
||||
msg_num,
|
||||
size_mb,
|
||||
);
|
||||
skipped_count += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
let message = match MessageParser::new()
|
||||
.with_minimal_headers()
|
||||
.default_header_text()
|
||||
.parse(body)
|
||||
{
|
||||
Some(msg) => msg,
|
||||
None => {
|
||||
eprintln!(
|
||||
"{} {}: {}",
|
||||
style("Warning").yellow().bold(),
|
||||
style(format!("at message #{}", msg_num)).dim(),
|
||||
"Failed to parse email structure. Skipping..."
|
||||
);
|
||||
skipped_count += 1;
|
||||
continue;
|
||||
}
|
||||
};
|
||||
|
||||
let mut metadata: Option<BichonMetadata> = None;
|
||||
if let Some(meta_header) = message.header_raw("X-Bichon-Metadata") {
|
||||
metadata = parse_bichon_metadata(meta_header);
|
||||
}
|
||||
|
||||
let get_default_folder = || {
|
||||
let labels = message
|
||||
.header("X-Gmail-Labels")
|
||||
.and_then(|h| h.as_text())
|
||||
.map(|s| s.to_string())
|
||||
.unwrap_or_else(|| "INBOX".to_string());
|
||||
determine_folder(&labels)
|
||||
};
|
||||
|
||||
let folder_name = if let Some(ref folder) = target_folder {
|
||||
folder.clone()
|
||||
} else if let Some(ref meta) = metadata {
|
||||
meta.mailbox_name.clone().unwrap_or_else(get_default_folder)
|
||||
} else {
|
||||
get_default_folder()
|
||||
};
|
||||
|
||||
// Drop message before base64-encoding to free MIME parse memory.
|
||||
drop(message);
|
||||
|
||||
let b64_eml = base64_encode_url_safe!(&body);
|
||||
let encoded_len = b64_eml.len();
|
||||
|
||||
let buffer = folder_buffers
|
||||
.entry(folder_name.clone())
|
||||
.or_insert_with(Vec::new);
|
||||
buffer.push(b64_eml);
|
||||
total_buffered_bytes += encoded_len;
|
||||
|
||||
if buffer.len() >= batch_limit || total_buffered_bytes >= MAX_BUFFER_BYTES {
|
||||
let emls_to_send = folder_buffers.remove(&folder_name).unwrap();
|
||||
let freed: usize = emls_to_send.iter().map(|s| s.len()).sum();
|
||||
total_buffered_bytes = total_buffered_bytes.saturating_sub(freed);
|
||||
send_batch_request(&client, config, account_id, &folder_name, emls_to_send).await;
|
||||
}
|
||||
}
|
||||
|
||||
for (folder_name, emls) in folder_buffers {
|
||||
if !emls.is_empty() {
|
||||
send_batch_request(&client, config, account_id, &folder_name, emls).await;
|
||||
}
|
||||
}
|
||||
|
||||
if skipped_count > 0 {
|
||||
println!(
|
||||
"{}",
|
||||
style(format!(
|
||||
"Skipped {} email(s) (oversized or unparseable).",
|
||||
skipped_count
|
||||
))
|
||||
.yellow()
|
||||
.bold()
|
||||
);
|
||||
}
|
||||
|
||||
println!("{}", style("Import completed successfully!").green().bold());
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::collections::HashMap;
|
||||
|
||||
/// Fake sender: records every flushed batch as (folder_name, email_count, total_bytes).
|
||||
struct FakeSender {
|
||||
batches: Vec<(String, usize, usize)>,
|
||||
}
|
||||
|
||||
impl FakeSender {
|
||||
fn new() -> Self {
|
||||
Self { batches: vec![] }
|
||||
}
|
||||
fn send(&mut self, folder: &str, emls: Vec<String>) {
|
||||
let count = emls.len();
|
||||
let bytes: usize = emls.iter().map(|s| s.len()).sum();
|
||||
self.batches.push((folder.to_string(), count, bytes));
|
||||
// emls is dropped here, simulating real send
|
||||
}
|
||||
}
|
||||
|
||||
fn fake_encode(size: usize) -> String {
|
||||
// base64 expands ~1.33x, so the encoded string is roughly this long.
|
||||
// We just need a predictable byte size, so use a repeated character.
|
||||
"x".repeat(size)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn flush_on_global_byte_threshold() {
|
||||
let mut buffers: HashMap<String, Vec<String>> = HashMap::new();
|
||||
let mut total_bytes: usize = 0;
|
||||
let batch_limit = 50;
|
||||
let mut sender = FakeSender::new();
|
||||
|
||||
// Simulate 3 emails, each 80 MB encoded, spread across 3 folders.
|
||||
// After each email, global total goes up by 80 MB.
|
||||
// After the 3rd email: 240 MB > 200 MB → flush the folder that got the 3rd email.
|
||||
let emails = vec![
|
||||
("Inbox", 80_000_000),
|
||||
("Sent", 80_000_000),
|
||||
("Archive", 80_000_000),
|
||||
];
|
||||
|
||||
for (folder, eml_size) in emails {
|
||||
let encoded = fake_encode(eml_size);
|
||||
let len = encoded.len();
|
||||
let buffer = buffers.entry(folder.to_string()).or_insert_with(Vec::new);
|
||||
buffer.push(encoded);
|
||||
total_bytes += len;
|
||||
|
||||
if buffer.len() >= batch_limit || total_bytes >= MAX_BUFFER_BYTES {
|
||||
let sent = buffers.remove(folder).unwrap();
|
||||
let freed: usize = sent.iter().map(|s| s.len()).sum();
|
||||
total_bytes = total_bytes.saturating_sub(freed);
|
||||
sender.send(folder, sent);
|
||||
}
|
||||
}
|
||||
|
||||
// The 3rd email should trigger a global flush of "Archive".
|
||||
assert_eq!(sender.batches.len(), 1);
|
||||
assert_eq!(sender.batches[0].0, "Archive");
|
||||
assert_eq!(sender.batches[0].1, 1);
|
||||
// "Inbox" and "Sent" are still buffered (160 MB total).
|
||||
assert_eq!(buffers.len(), 2);
|
||||
assert!(buffers.contains_key("Inbox"));
|
||||
assert!(buffers.contains_key("Sent"));
|
||||
assert_eq!(total_bytes, 160_000_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn flush_on_count_threshold() {
|
||||
let mut buffers: HashMap<String, Vec<String>> = HashMap::new();
|
||||
let mut total_bytes: usize = 0;
|
||||
let batch_limit = 3;
|
||||
let mut sender = FakeSender::new();
|
||||
|
||||
// 4 small emails all to Inbox, well under byte threshold.
|
||||
for _ in 0..4 {
|
||||
let encoded = fake_encode(100); // tiny
|
||||
let len = encoded.len();
|
||||
let buffer = buffers
|
||||
.entry("Inbox".to_string())
|
||||
.or_insert_with(Vec::new);
|
||||
buffer.push(encoded);
|
||||
total_bytes += len;
|
||||
|
||||
if buffer.len() >= batch_limit || total_bytes >= MAX_BUFFER_BYTES {
|
||||
let sent = buffers.remove("Inbox").unwrap();
|
||||
let freed: usize = sent.iter().map(|s| s.len()).sum();
|
||||
total_bytes = total_bytes.saturating_sub(freed);
|
||||
sender.send("Inbox", sent);
|
||||
}
|
||||
}
|
||||
|
||||
// Count=3 should trigger flush once; the 4th email stays buffered.
|
||||
assert_eq!(sender.batches.len(), 1);
|
||||
assert_eq!(sender.batches[0].1, 3); // 3 emails flushed
|
||||
let remaining = buffers.get("Inbox").unwrap();
|
||||
assert_eq!(remaining.len(), 1); // 1 still buffered
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn global_bytes_exact_boundary() {
|
||||
let mut buffers: HashMap<String, Vec<String>> = HashMap::new();
|
||||
let mut total_bytes: usize = 0;
|
||||
let mut sender = FakeSender::new();
|
||||
|
||||
// Push one email that puts us right at 200 MB.
|
||||
let encoded = fake_encode(MAX_BUFFER_BYTES);
|
||||
let len = encoded.len();
|
||||
buffers
|
||||
.entry("Inbox".to_string())
|
||||
.or_insert_with(Vec::new)
|
||||
.push(encoded);
|
||||
total_bytes += len;
|
||||
|
||||
if total_bytes >= MAX_BUFFER_BYTES {
|
||||
let sent = buffers.remove("Inbox").unwrap();
|
||||
let freed: usize = sent.iter().map(|s| s.len()).sum();
|
||||
total_bytes = total_bytes.saturating_sub(freed);
|
||||
sender.send("Inbox", sent);
|
||||
}
|
||||
|
||||
// Should have flushed on the boundary.
|
||||
assert_eq!(sender.batches.len(), 1);
|
||||
assert_eq!(total_bytes, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn flush_one_folder_does_not_lose_others() {
|
||||
let mut buffers: HashMap<String, Vec<String>> = HashMap::new();
|
||||
let mut total_bytes: usize = 0;
|
||||
let batch_limit = 50;
|
||||
let mut sender = FakeSender::new();
|
||||
|
||||
// Build up A to 150 MB, B to 100 MB (total 250 MB > 200 MB).
|
||||
// A should trigger flush; B should stay buffered.
|
||||
let folder_a = "A".to_string();
|
||||
let folder_b = "B".to_string();
|
||||
|
||||
// Folder A: 150 MB
|
||||
let encoded = fake_encode(150_000_000);
|
||||
let len = encoded.len();
|
||||
buffers.entry(folder_a.clone()).or_insert_with(Vec::new).push(encoded);
|
||||
total_bytes += len;
|
||||
|
||||
// Folder B: 100 MB → total 250 MB → trigger flush on B
|
||||
let encoded = fake_encode(100_000_000);
|
||||
let len = encoded.len();
|
||||
buffers.entry(folder_b.clone()).or_insert_with(Vec::new).push(encoded);
|
||||
total_bytes += len;
|
||||
|
||||
// Check trigger on B
|
||||
let b_buffer = buffers.get(&folder_b).unwrap();
|
||||
if b_buffer.len() >= batch_limit || total_bytes >= MAX_BUFFER_BYTES {
|
||||
let sent = buffers.remove(&folder_b).unwrap();
|
||||
let freed: usize = sent.iter().map(|s| s.len()).sum();
|
||||
total_bytes = total_bytes.saturating_sub(freed);
|
||||
sender.send(&folder_b, sent);
|
||||
}
|
||||
|
||||
assert_eq!(sender.batches.len(), 1);
|
||||
assert_eq!(sender.batches[0].0, "B"); // B flushed
|
||||
assert!(buffers.contains_key("A")); // A still there
|
||||
assert_eq!(total_bytes, 150_000_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn skip_oversized_email() {
|
||||
assert!(100 <= MAX_EMAIL_BYTES);
|
||||
// Use vec! so the 100 MB array lives on the heap, not the stack.
|
||||
let huge = vec![0u8; MAX_EMAIL_BYTES + 1];
|
||||
assert!(huge.len() > MAX_EMAIL_BYTES);
|
||||
}
|
||||
}
|
||||
224
crates/cli/src/pst/mod.rs
Normal file
224
crates/cli/src/pst/mod.rs
Normal file
@@ -0,0 +1,224 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::api::sender::send_batch_request;
|
||||
use crate::BichonCliConfig;
|
||||
use bichon_core::import::pst::build_eml_base64;
|
||||
use dialoguer::theme::ColorfulTheme;
|
||||
use dialoguer::{Confirm, Input};
|
||||
use outlook_pst::messaging::folder::Folder;
|
||||
use outlook_pst::ndb::node_id::NodeId;
|
||||
use reqwest::Client;
|
||||
use std::future::Future;
|
||||
use std::path::PathBuf;
|
||||
use std::pin::Pin;
|
||||
use std::rc::Rc;
|
||||
|
||||
pub async fn handle_pst_import(config: &BichonCliConfig, account_id: u64, theme: &ColorfulTheme) {
|
||||
let path_str: String = Input::with_theme(theme)
|
||||
.with_prompt("Enter the path to your SINGLE .pst file")
|
||||
.validate_with(|input: &String| {
|
||||
let p = std::path::Path::new(input);
|
||||
if !p.exists() {
|
||||
return Err("The specified path does not exist.");
|
||||
}
|
||||
|
||||
if !p.is_file() {
|
||||
return Err("PST mode requires a SINGLE file, not a directory.");
|
||||
}
|
||||
let is_pst = p
|
||||
.extension()
|
||||
.and_then(|ext| ext.to_str())
|
||||
.map(|ext| ext.eq_ignore_ascii_case("pst"))
|
||||
.unwrap_or(false);
|
||||
|
||||
if !is_pst {
|
||||
return Err("The selected file must have a .pst extension.");
|
||||
}
|
||||
|
||||
Ok(())
|
||||
})
|
||||
.interact_text()
|
||||
.unwrap();
|
||||
|
||||
let pst_path = std::path::PathBuf::from(path_str);
|
||||
|
||||
println!(
|
||||
"\n{} Ready to process PST file: {}",
|
||||
console::style("✔").green(),
|
||||
console::style(pst_path.display()).cyan()
|
||||
);
|
||||
|
||||
if let Ok(meta) = std::fs::metadata(&pst_path) {
|
||||
let size_mb = meta.len() as f64 / 1024.0 / 1024.0;
|
||||
println!(
|
||||
"{}",
|
||||
console::style(format!("PST File Size: {:.1} MB", size_mb)).dim()
|
||||
);
|
||||
}
|
||||
|
||||
if Confirm::with_theme(theme)
|
||||
.with_prompt("Start importing emails from this PST?")
|
||||
.default(true)
|
||||
.interact()
|
||||
.unwrap()
|
||||
{
|
||||
parse_pst(pst_path, config, account_id).await;
|
||||
} else {
|
||||
println!("{}", console::style("Operation cancelled by user.").red());
|
||||
}
|
||||
}
|
||||
|
||||
async fn parse_pst(pst_path: PathBuf, config: &BichonCliConfig, account_id: u64) {
|
||||
let client = Client::new();
|
||||
|
||||
let pst_store = match outlook_pst::open_store(&pst_path) {
|
||||
Ok(store) => store,
|
||||
Err(e) => {
|
||||
println!(
|
||||
"{} Failed to open PST file: {}",
|
||||
console::style("✘").red(),
|
||||
console::style(format!("{:#?}", e)).dim()
|
||||
);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let ipm_sub_tree = match pst_store.properties().ipm_sub_tree_entry_id() {
|
||||
Ok(id) => id,
|
||||
Err(e) => {
|
||||
println!(
|
||||
"{} Could not find IPM_SUBTREE (Mailbox Root): {}",
|
||||
console::style("✘").red(),
|
||||
console::style(format!("{:#?}", e)).dim()
|
||||
);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let ipm_subtree_folder = match pst_store.open_folder(&ipm_sub_tree) {
|
||||
Ok(folder) => folder,
|
||||
Err(e) => {
|
||||
println!(
|
||||
"{} Failed to open the root mailbox folder: {}",
|
||||
console::style("✘").red(),
|
||||
console::style(format!("{:#?}", e)).dim()
|
||||
);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
process_folder_recursively(&client, &ipm_subtree_folder, "", config, account_id).await;
|
||||
}
|
||||
|
||||
fn process_folder_recursively<'a>(
|
||||
client: &'a Client,
|
||||
folder: &'a Rc<dyn Folder>,
|
||||
parent_path: &'a str,
|
||||
config: &'a BichonCliConfig,
|
||||
account_id: u64,
|
||||
) -> Pin<Box<dyn Future<Output = ()> + 'a>> {
|
||||
Box::pin(async move {
|
||||
let folder_name = folder
|
||||
.properties()
|
||||
.display_name()
|
||||
.unwrap_or_else(|_| "Unknown".to_string());
|
||||
|
||||
let current_path = if parent_path.is_empty() {
|
||||
folder_name
|
||||
} else {
|
||||
format!("{}/{}", parent_path, folder_name)
|
||||
};
|
||||
|
||||
println!(
|
||||
"{} {}",
|
||||
console::style("📁 Folder:").dim(),
|
||||
console::style(¤t_path).cyan()
|
||||
);
|
||||
|
||||
let mut emls_batch = Vec::new();
|
||||
|
||||
if let Some(contents_table) = folder.contents_table() {
|
||||
for row in contents_table.rows_matrix() {
|
||||
let store = folder.store().clone();
|
||||
|
||||
let entry_id = match store
|
||||
.properties()
|
||||
.make_entry_id(NodeId::from(u32::from(row.id())))
|
||||
{
|
||||
Ok(id) => id,
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
" {} Skip row {}: {:?}",
|
||||
console::style("⚠").yellow(),
|
||||
row.unique(),
|
||||
e
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
|
||||
match store.open_message(&entry_id, None) {
|
||||
Ok(message) => match build_eml_base64(message) {
|
||||
Some(base64_eml) => emls_batch.push(base64_eml),
|
||||
None => {}
|
||||
},
|
||||
Err(e) => eprintln!(" {} Open error: {:?}", console::style("⚠").yellow(), e),
|
||||
}
|
||||
|
||||
if emls_batch.len() >= 50 {
|
||||
let batch = emls_batch.clone();
|
||||
emls_batch.clear();
|
||||
send_to_bichon(client, config, account_id, ¤t_path, batch).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if !emls_batch.is_empty() {
|
||||
send_to_bichon(client, config, account_id, ¤t_path, emls_batch).await;
|
||||
}
|
||||
|
||||
if let Some(hierarchy_table) = folder.hierarchy_table() {
|
||||
for row in hierarchy_table.rows_matrix() {
|
||||
let node = NodeId::from(u32::from(row.id()));
|
||||
if let Ok(entry_id) = folder.store().properties().make_entry_id(node) {
|
||||
if let Ok(sub_folder) = folder.store().open_folder(&entry_id) {
|
||||
process_folder_recursively(
|
||||
client,
|
||||
&sub_folder,
|
||||
¤t_path,
|
||||
config,
|
||||
account_id,
|
||||
)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
async fn send_to_bichon(
|
||||
client: &Client,
|
||||
config: &BichonCliConfig,
|
||||
account_id: u64,
|
||||
folder_path: &str,
|
||||
emls: Vec<String>,
|
||||
) {
|
||||
send_batch_request(client, config, account_id, folder_path, emls).await;
|
||||
}
|
||||
@@ -1,11 +1,29 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use std::{collections::HashMap, path::PathBuf};
|
||||
|
||||
use crate::modules::cli::{mbox::run_import, BichonCtlConfig};
|
||||
use crate::{mbox::run_import, BichonCliConfig};
|
||||
use console::style;
|
||||
use dialoguer::{theme::ColorfulTheme, Confirm, Input};
|
||||
|
||||
pub async fn handle_thunderbird_import(
|
||||
config: &BichonCtlConfig,
|
||||
config: &BichonCliConfig,
|
||||
account_id: u64,
|
||||
theme: &ColorfulTheme,
|
||||
) {
|
||||
79
crates/core/Cargo.toml
Normal file
79
crates/core/Cargo.toml
Normal file
@@ -0,0 +1,79 @@
|
||||
[package]
|
||||
name = "bichon-core"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
|
||||
|
||||
[features]
|
||||
default = ["web-api"]
|
||||
web-api = ["dep:poem-openapi"]
|
||||
|
||||
[dependencies]
|
||||
poem-openapi = { version = "5.1.16", features = [
|
||||
"openapi-explorer",
|
||||
"rapidoc",
|
||||
"scalar",
|
||||
"redoc",
|
||||
"swagger-ui",
|
||||
"email",
|
||||
], optional = true }
|
||||
chrono.workspace = true
|
||||
clap.workspace = true
|
||||
bichon-memdb.workspace = true
|
||||
itertools.workspace = true
|
||||
ring.workspace = true
|
||||
serde.workspace = true
|
||||
serde_json.workspace = true
|
||||
tokio.workspace = true
|
||||
tracing.workspace = true
|
||||
tracing-appender.workspace = true
|
||||
tracing-subscriber.workspace = true
|
||||
base64.workspace = true
|
||||
snafu.workspace = true
|
||||
reqwest.workspace = true
|
||||
tokio-socks.workspace = true
|
||||
regex.workspace = true
|
||||
email_address.workspace = true
|
||||
futures.workspace = true
|
||||
utf7-imap.workspace = true
|
||||
mail-parser.workspace = true
|
||||
tokio-rustls.workspace = true
|
||||
oauth2.workspace = true
|
||||
sysinfo.workspace = true
|
||||
num_cpus.workspace = true
|
||||
rand.workspace = true
|
||||
encoding_rs.workspace = true
|
||||
async-imap = { git = "https://github.com/rustmailer/async-imap.git", branch = "main", default-features = false, features = [
|
||||
"runtime-tokio",
|
||||
"compress",
|
||||
] }
|
||||
tantivy = { version = "0.26.1", features = ["zstd-compression", "quickwit"] }
|
||||
webpki-roots.workspace = true
|
||||
rustls.workspace = true
|
||||
rustls-pki-types.workspace = true
|
||||
tokio-io-timeout.workspace = true
|
||||
governor.workspace = true
|
||||
lru.workspace = true
|
||||
time.workspace = true
|
||||
murmur3.workspace = true
|
||||
dashmap.workspace = true
|
||||
itoa.workspace = true
|
||||
html2text.workspace = true
|
||||
bytes.workspace = true
|
||||
mail-send.workspace = true
|
||||
blake3.workspace = true
|
||||
uuid.workspace = true
|
||||
fjall.workspace = true
|
||||
tracing-log.workspace = true
|
||||
tokio-util.workspace = true
|
||||
whichlang = "0.1.1"
|
||||
deunicode = "1.6.2"
|
||||
scopeguard = "1.2.0"
|
||||
cron = "0.15"
|
||||
quick-xml = { version = "0.40.0", features = ["serialize"] }
|
||||
hickory-resolver = "0.26.0-alpha.1"
|
||||
memmap2 = "0.9.10"
|
||||
outlook-pst = { git = "https://github.com/rustmailer/outlook-pst-rs.git", branch = "main" }
|
||||
compressed-rtf = "1.0.1"
|
||||
codepage-strings = "1.0.2"
|
||||
hex.workspace = true
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,19 +16,25 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::{encrypt, error::BichonResult};
|
||||
|
||||
use crate::{encrypt, modules::error::BichonResult};
|
||||
|
||||
use poem_openapi::{Enum, Object};
|
||||
//use poem_openapi::{Enum, Object};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
#[derive(Clone, Default, Debug, Eq, PartialEq, Serialize, Deserialize, Object)]
|
||||
#[derive(Clone, Default, Debug, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct ImapConfig {
|
||||
/// IMAP server hostname or IP address
|
||||
#[oai(validator(max_length = 253, pattern = r"^[a-zA-Z0-9\-\.]+$"))]
|
||||
#[cfg_attr(
|
||||
feature = "web-api",
|
||||
oai(validator(max_length = 253, pattern = r"^[a-zA-Z0-9\-\.]+$"))
|
||||
)]
|
||||
pub host: String,
|
||||
/// IMAP server port number
|
||||
#[oai(validator(minimum(value = "1"), maximum(value = "65535")))]
|
||||
#[cfg_attr(
|
||||
feature = "web-api",
|
||||
oai(validator(minimum(value = "1"), maximum(value = "65535")))
|
||||
)]
|
||||
pub port: u16,
|
||||
/// Connection encryption method
|
||||
pub encryption: Encryption,
|
||||
@@ -52,8 +58,8 @@ impl ImapConfig {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#[derive(Enum, Default, Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[derive(Default, Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum AuthType {
|
||||
/// Standard password authentication (PLAIN/LOGIN)
|
||||
#[default]
|
||||
@@ -62,7 +68,8 @@ pub enum AuthType {
|
||||
OAuth2,
|
||||
}
|
||||
|
||||
#[derive(Object, Default, Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[derive(Default, Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct AuthConfig {
|
||||
///Authentication method to use
|
||||
pub auth_type: AuthType,
|
||||
@@ -70,7 +77,7 @@ pub struct AuthConfig {
|
||||
///
|
||||
/// Users should provide a plaintext password (1 to 256 characters).
|
||||
/// The server will encrypt the password using AES-256-GCM and securely store it.
|
||||
#[oai(validator(max_length = 256, min_length = 1))]
|
||||
#[cfg_attr(feature = "web-api", oai(validator(max_length = 256, min_length = 1)))]
|
||||
pub password: Option<String>,
|
||||
}
|
||||
|
||||
@@ -97,7 +104,8 @@ impl AuthConfig {
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Default, Debug, Eq, PartialEq, Serialize, Deserialize, Enum)]
|
||||
#[derive(Clone, Default, Debug, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum Encryption {
|
||||
/// SSL/TLS encrypted connection
|
||||
#[default]
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,14 +16,15 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use poem_openapi::Object;
|
||||
//use poem_openapi::Object;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::{
|
||||
modules::{
|
||||
raise_error, utc_now,
|
||||
{
|
||||
account::migration::AccountModel,
|
||||
common::auth::ClientContext,
|
||||
database::{manager::DB_MANAGER, with_transaction},
|
||||
database::{manager::DB_MANAGER, with_transaction, MemDbModel},
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
users::{
|
||||
permissions::Permission,
|
||||
@@ -31,10 +32,10 @@ use crate::{
|
||||
UserModel,
|
||||
},
|
||||
},
|
||||
raise_error, utc_now,
|
||||
};
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct BatchAccountRoleRequest {
|
||||
pub account_ids: Vec<u64>,
|
||||
pub user_ids: Vec<u64>,
|
||||
@@ -42,8 +43,8 @@ pub struct BatchAccountRoleRequest {
|
||||
}
|
||||
|
||||
impl BatchAccountRoleRequest {
|
||||
pub async fn validate_existence(&self) -> BichonResult<()> {
|
||||
let role = UserRole::find(self.role_id).await?.ok_or_else(|| {
|
||||
pub fn validate_existence(&self) -> BichonResult<()> {
|
||||
let role = UserRole::find(self.role_id)?.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("Role ID {} not found", self.role_id),
|
||||
ErrorCode::ResourceNotFound
|
||||
@@ -58,7 +59,7 @@ impl BatchAccountRoleRequest {
|
||||
}
|
||||
|
||||
for id in &self.account_ids {
|
||||
let exists = AccountModel::find(*id).await?; // Assuming an exists helper
|
||||
let exists = AccountModel::find(*id)?; // Assuming an exists helper
|
||||
if exists.is_none() {
|
||||
return Err(raise_error!(
|
||||
format!("Account ID {} not found", id),
|
||||
@@ -68,7 +69,7 @@ impl BatchAccountRoleRequest {
|
||||
}
|
||||
|
||||
for id in &self.user_ids {
|
||||
let exists = UserModel::find(*id).await?; // Assuming an exists helper
|
||||
let exists = UserModel::find(*id)?; // Assuming an exists helper
|
||||
if exists.is_none() {
|
||||
return Err(raise_error!(
|
||||
format!("User ID {} not found", id),
|
||||
@@ -80,44 +81,38 @@ impl BatchAccountRoleRequest {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn grant_batch_account_access(
|
||||
fn grant_batch_account_access(
|
||||
account_ids: Vec<u64>,
|
||||
user_ids: Vec<u64>,
|
||||
role_id: u64,
|
||||
) -> BichonResult<()> {
|
||||
with_transaction(DB_MANAGER.meta_db(), move |rw| {
|
||||
with_transaction(DB_MANAGER.db(), move |txn| {
|
||||
let mut txn = txn;
|
||||
for &uid in &user_ids {
|
||||
// Fetch the current user record from the database
|
||||
let user = rw
|
||||
.get()
|
||||
.primary::<UserModel>(uid)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("User with id={} not found.", uid),
|
||||
ErrorCode::ResourceNotFound
|
||||
)
|
||||
})?;
|
||||
let db = DB_MANAGER.db();
|
||||
let coll = db.collection(UserModel::collection());
|
||||
let key = uid.to_string();
|
||||
let user: UserModel = coll
|
||||
.get_required(&key)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
let mut updated_user = user.clone();
|
||||
|
||||
// Apply the role to each specified account_id
|
||||
for &aid in &account_ids {
|
||||
updated_user.account_access_map.insert(aid, role_id);
|
||||
}
|
||||
|
||||
updated_user.updated_at = utc_now!();
|
||||
|
||||
// Save the updated user back to the database within the transaction
|
||||
rw.update(user, updated_user)
|
||||
txn = txn
|
||||
.upsert(UserModel::collection(), key, &updated_user)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
}
|
||||
Ok(())
|
||||
Ok(txn)
|
||||
})
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn do_assign(self, context: &ClientContext) -> BichonResult<()> {
|
||||
pub fn do_assign(self, context: &ClientContext) -> BichonResult<()> {
|
||||
for account_id in &self.account_ids {
|
||||
// Get the user's specific access for this account
|
||||
let assigned_role_id =
|
||||
@@ -133,7 +128,7 @@ impl BatchAccountRoleRequest {
|
||||
})?;
|
||||
|
||||
// Fetch the role definition from the database
|
||||
let user_scoped_role = UserRole::find(*assigned_role_id).await?.ok_or_else(|| {
|
||||
let user_scoped_role = UserRole::find(*assigned_role_id)?.ok_or_else(|| {
|
||||
raise_error!(
|
||||
"Assigned account role no longer exists".into(),
|
||||
ErrorCode::InternalError
|
||||
@@ -155,6 +150,6 @@ impl BatchAccountRoleRequest {
|
||||
// This is where you'd compare target_role.permissions vs manager's perms
|
||||
}
|
||||
|
||||
Self::grant_batch_account_access(self.account_ids, self.user_ids, self.role_id).await
|
||||
Self::grant_batch_account_access(self.account_ids, self.user_ids, self.role_id)
|
||||
}
|
||||
}
|
||||
996
crates/core/src/account/migration.rs
Normal file
996
crates/core/src/account/migration.rs
Normal file
@@ -0,0 +1,996 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::BTreeSet;
|
||||
use tracing::info;
|
||||
|
||||
use crate::{
|
||||
account::{
|
||||
entity::ImapConfig,
|
||||
payload::{AccountCreateRequest, AccountUpdateRequest, MinimalAccount},
|
||||
since::{DateSince, RelativeDate},
|
||||
state::DownloadState,
|
||||
},
|
||||
cache::imap::{mailbox::MailBox, task::SYNC_TASKS},
|
||||
common::paginated::DataPage,
|
||||
context::controller::DOWNLOAD_CONTROLLER,
|
||||
database::{
|
||||
count_impl, delete_impl, find_impl, insert_impl, list_all_impl, manager::DB_MANAGER,
|
||||
paginate_impl, update_impl, MemDbModel,
|
||||
},
|
||||
encrypt,
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
id,
|
||||
oauth2::token::OAuth2AccessToken,
|
||||
raise_error,
|
||||
store::tantivy::{attachment::ATTACHMENT_MANAGER, envelope::ENVELOPE_MANAGER},
|
||||
users::{payload::UserUpdateRequest, role::DEFAULT_ACCOUNT_MANAGER_ROLE_ID, UserModel},
|
||||
utc_now,
|
||||
};
|
||||
|
||||
pub type AccountModel = Account;
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum AccountType {
|
||||
#[default]
|
||||
IMAP,
|
||||
NoSync,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum QuotaWindow {
|
||||
Hourly,
|
||||
#[default]
|
||||
Daily,
|
||||
Weekly,
|
||||
Monthly,
|
||||
}
|
||||
|
||||
/// Include/exclude filter rule.
|
||||
///
|
||||
/// - `include` non-empty: only values matching these patterns pass.
|
||||
/// - `exclude` non-empty: values matching these patterns are rejected.
|
||||
/// - Both empty: all values pass.
|
||||
/// - Both set: include checked first, then exclude.
|
||||
///
|
||||
/// Extension patterns use case-insensitive exact match; all others use regex.
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct FilterRule {
|
||||
#[serde(default)]
|
||||
pub include: Vec<String>,
|
||||
#[serde(default)]
|
||||
pub exclude: Vec<String>,
|
||||
}
|
||||
|
||||
impl FilterRule {
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.include.is_empty() && self.exclude.is_empty()
|
||||
}
|
||||
|
||||
fn matches_exact(&self, value: &str) -> bool {
|
||||
if !self.include.is_empty() && !self.include.iter().any(|e| e.eq_ignore_ascii_case(value)) {
|
||||
return false;
|
||||
}
|
||||
if !self.exclude.is_empty() && self.exclude.iter().any(|e| e.eq_ignore_ascii_case(value)) {
|
||||
return false;
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
fn matches_regex(&self, value: &str) -> bool {
|
||||
if !self.include.is_empty() && !matches_any_regex(&self.include, value) {
|
||||
return false;
|
||||
}
|
||||
if !self.exclude.is_empty() && matches_any_regex(&self.exclude, value) {
|
||||
return false;
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
fn validate_regex(&self, field: &str) -> Result<(), String> {
|
||||
validate_patterns(&self.include, &format!("{field}.include"))?;
|
||||
validate_patterns(&self.exclude, &format!("{field}.exclude"))?;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct ExtractionRules {
|
||||
/// Type 0: Master switch.
|
||||
#[serde(default)]
|
||||
pub enabled: bool,
|
||||
/// Type 1: File extensions (exact match, e.g. `{"include": ["pdf","docx"]}`).
|
||||
#[serde(default)]
|
||||
pub extensions: FilterRule,
|
||||
/// Type 2: Folder patterns (regex, e.g. `{"include": ["^INBOX/Invoices"]}`).
|
||||
#[serde(default)]
|
||||
pub folders: FilterRule,
|
||||
/// Type 3: Attachment filename patterns (regex).
|
||||
#[serde(default)]
|
||||
pub attachment_names: FilterRule,
|
||||
/// Type 4: Sender patterns (regex).
|
||||
#[serde(default)]
|
||||
pub senders: FilterRule,
|
||||
}
|
||||
|
||||
impl ExtractionRules {
|
||||
/// Returns `true` if the attachment should be extracted under these rules.
|
||||
pub fn should_extract(
|
||||
&self,
|
||||
ext: &str,
|
||||
folder: Option<&str>,
|
||||
attachment_name: Option<&str>,
|
||||
sender: Option<&str>,
|
||||
) -> bool {
|
||||
if !self.enabled {
|
||||
return false;
|
||||
}
|
||||
if !self.extensions.matches_exact(ext) {
|
||||
return false;
|
||||
}
|
||||
if !self.folders.is_empty() {
|
||||
if let Some(folder) = folder {
|
||||
if !self.folders.matches_regex(folder) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
if !self.attachment_names.is_empty() {
|
||||
if let Some(name) = attachment_name {
|
||||
if !self.attachment_names.matches_regex(name) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
if !self.senders.is_empty() {
|
||||
if let Some(sender) = sender {
|
||||
if !self.senders.matches_regex(sender) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
pub fn validate(&self) -> Result<(), String> {
|
||||
self.folders.validate_regex("folders")?;
|
||||
self.attachment_names.validate_regex("attachment_names")?;
|
||||
self.senders.validate_regex("senders")?;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Archive filtering rules — skip unwanted emails before storage.
|
||||
///
|
||||
/// Rule types:
|
||||
/// 0 — Master switch
|
||||
/// 1 — Sender filter (regex)
|
||||
/// 2 — Subject filter (regex)
|
||||
/// 3 — Skip emails larger than this (bytes)
|
||||
/// 4 — Skip emails with spam headers (X-Spam-Flag, X-Spam)
|
||||
///
|
||||
/// `None` = archive everything (backward compatible).
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct ArchiveRules {
|
||||
/// Type 0: Master switch. `false` = archive everything.
|
||||
#[serde(default)]
|
||||
pub enabled: bool,
|
||||
/// Type 1: Sender filter (regex, include/exclude).
|
||||
#[serde(default)]
|
||||
pub senders: FilterRule,
|
||||
/// Type 2: Subject filter (regex, include/exclude).
|
||||
#[serde(default)]
|
||||
pub subjects: FilterRule,
|
||||
/// Type 3: Skip emails larger than this (bytes). `None` = no size limit.
|
||||
#[serde(default)]
|
||||
pub skip_larger_than: Option<u64>,
|
||||
/// Type 4: Spam header names to check (e.g. `["X-Spam-Flag", "X-Spam"]`).
|
||||
/// When the value is `yes` or `true` (case-insensitive), the email is skipped.
|
||||
/// Empty = don't check. Common headers: `X-Spam-Flag` (SpamAssassin),
|
||||
/// `X-Spam` (rspamd), `X-MS-Exchange-Organization-SCL` (Exchange).
|
||||
#[serde(default)]
|
||||
pub spam_headers: Vec<String>,
|
||||
}
|
||||
|
||||
impl ArchiveRules {
|
||||
/// Returns `true` if the email should be archived under these rules.
|
||||
pub fn should_archive(
|
||||
&self,
|
||||
sender: Option<&str>,
|
||||
subject: Option<&str>,
|
||||
size: u32,
|
||||
is_spam: bool,
|
||||
) -> bool {
|
||||
if !self.enabled {
|
||||
return true;
|
||||
}
|
||||
if !self.senders.is_empty() {
|
||||
if let Some(sender) = sender {
|
||||
if !self.senders.matches_regex(sender) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
if !self.subjects.is_empty() {
|
||||
if let Some(subject) = subject {
|
||||
if !self.subjects.matches_regex(subject) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
if let Some(limit) = self.skip_larger_than {
|
||||
if size as u64 > limit {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if !self.spam_headers.is_empty() && is_spam {
|
||||
return false;
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Validate all regex patterns are well-formed.
|
||||
pub fn validate(&self) -> Result<(), String> {
|
||||
self.senders.validate_regex("senders")?;
|
||||
self.subjects.validate_regex("subjects")?;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn matches_any_regex(patterns: &[String], value: &str) -> bool {
|
||||
patterns.iter().any(|p| {
|
||||
regex::Regex::new(p)
|
||||
.map(|re| re.is_match(value))
|
||||
.unwrap_or(false)
|
||||
})
|
||||
}
|
||||
|
||||
fn validate_patterns(patterns: &[String], field_name: &str) -> Result<(), String> {
|
||||
for p in patterns {
|
||||
regex::Regex::new(p)
|
||||
.map_err(|e| format!("{} pattern '{}' is invalid regex: {}", field_name, p, e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct Account {
|
||||
pub id: u64,
|
||||
pub imap: Option<ImapConfig>,
|
||||
pub enabled: bool,
|
||||
#[cfg_attr(
|
||||
feature = "web-api",
|
||||
oai(validator(custom = "crate::common::validator::EmailValidator"))
|
||||
)]
|
||||
pub email: String,
|
||||
pub account_name: Option<String>,
|
||||
pub login_name: Option<String>,
|
||||
pub capabilities: Option<Vec<String>>,
|
||||
pub date_since: Option<DateSince>,
|
||||
pub date_before: Option<RelativeDate>,
|
||||
pub download_folders: Option<Vec<String>>,
|
||||
pub account_type: AccountType,
|
||||
pub download_interval_min: Option<i64>,
|
||||
pub download_batch_size: Option<u32>,
|
||||
#[serde(default)]
|
||||
pub max_email_size_bytes: Option<u64>,
|
||||
pub known_folders: Option<BTreeSet<String>>,
|
||||
pub created_at: i64,
|
||||
pub updated_at: i64,
|
||||
pub created_by: u64, //user id
|
||||
pub use_proxy: Option<u64>,
|
||||
pub use_dangerous: bool,
|
||||
pub pgp_key: Option<String>,
|
||||
pub imap_quota_bytes: Option<u64>,
|
||||
pub imap_quota_window: Option<QuotaWindow>,
|
||||
pub auto_download_new_mailboxes: Option<bool>,
|
||||
pub download_schedule: Option<String>,
|
||||
#[serde(default)]
|
||||
pub deleting: bool,
|
||||
/// Email-level filtering rules (Pro feature).
|
||||
/// `None` = archive everything (backward compatible).
|
||||
#[serde(default)]
|
||||
pub archive_rules: Option<ArchiveRules>,
|
||||
/// Attachment text extraction rules (Pro feature).
|
||||
/// `None` = extract everything (backward compatible).
|
||||
#[serde(default)]
|
||||
pub extraction_rules: Option<ExtractionRules>,
|
||||
}
|
||||
|
||||
impl MemDbModel for Account {
|
||||
fn collection() -> &'static str {
|
||||
"accounts"
|
||||
}
|
||||
fn key(&self) -> String {
|
||||
self.id.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
impl Account {
|
||||
pub fn new(user_id: u64, request: AccountCreateRequest) -> BichonResult<Self> {
|
||||
Ok(Self {
|
||||
id: id!(64),
|
||||
email: request.email,
|
||||
login_name: request.login_name,
|
||||
account_name: request.account_name,
|
||||
imap: request.imap.map(|i| i.try_encrypt_password()).transpose()?,
|
||||
enabled: request.enabled,
|
||||
capabilities: None,
|
||||
date_since: request.date_since,
|
||||
download_folders: None,
|
||||
known_folders: None,
|
||||
account_type: request.account_type,
|
||||
download_interval_min: request.download_interval_min,
|
||||
created_at: utc_now!(),
|
||||
updated_at: utc_now!(),
|
||||
use_proxy: request.use_proxy,
|
||||
use_dangerous: request.use_dangerous,
|
||||
pgp_key: request.pgp_key,
|
||||
created_by: user_id,
|
||||
download_batch_size: request.download_batch_size,
|
||||
max_email_size_bytes: request.max_email_size_bytes,
|
||||
date_before: request.date_before,
|
||||
auto_download_new_mailboxes: request.auto_download_new_mailboxes,
|
||||
imap_quota_bytes: request.imap_quota_bytes,
|
||||
imap_quota_window: request.imap_quota_window,
|
||||
download_schedule: request.download_schedule,
|
||||
deleting: false,
|
||||
archive_rules: request.archive_rules,
|
||||
extraction_rules: request.extraction_rules,
|
||||
})
|
||||
}
|
||||
|
||||
pub fn check_account_exists(account_id: u64) -> BichonResult<AccountModel> {
|
||||
Self::get(account_id)
|
||||
}
|
||||
|
||||
pub fn get(account_id: u64) -> BichonResult<AccountModel> {
|
||||
let result: AccountModel = Self::find(account_id)?.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("Account with ID '{account_id}' not found"),
|
||||
ErrorCode::ResourceNotFound
|
||||
)
|
||||
})?;
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
pub fn find(account_id: u64) -> BichonResult<Option<AccountModel>> {
|
||||
let result = find_impl::<AccountModel>(DB_MANAGER.db(), &account_id.to_string())?;
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
pub async fn create_account(
|
||||
user_id: u64,
|
||||
request: AccountCreateRequest,
|
||||
) -> BichonResult<AccountModel> {
|
||||
let entity = request.create_entity(user_id)?;
|
||||
let cloned = entity.clone();
|
||||
|
||||
// Insert account into memdb
|
||||
insert_impl(DB_MANAGER.db(), entity)?;
|
||||
|
||||
// Update user's account_access_map
|
||||
let user = UserModel::find(user_id)?.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("User with id={} not found.", user_id),
|
||||
ErrorCode::ResourceNotFound
|
||||
)
|
||||
})?;
|
||||
|
||||
let mut updated_map = user.account_access_map.clone();
|
||||
updated_map.insert(cloned.id, DEFAULT_ACCOUNT_MANAGER_ROLE_ID);
|
||||
|
||||
UserModel::update(
|
||||
user_id,
|
||||
UserUpdateRequest {
|
||||
username: None,
|
||||
email: None,
|
||||
password: None,
|
||||
avatar_base64: None,
|
||||
global_roles: None,
|
||||
account_access_map: Some(updated_map),
|
||||
acl: None,
|
||||
description: None,
|
||||
theme: None,
|
||||
language: None,
|
||||
},
|
||||
)?;
|
||||
|
||||
if matches!(cloned.account_type, AccountType::IMAP) {
|
||||
DOWNLOAD_CONTROLLER
|
||||
.trigger_schedule(cloned.id, cloned.email.clone())
|
||||
.await;
|
||||
}
|
||||
Ok(cloned)
|
||||
}
|
||||
|
||||
pub fn update(
|
||||
account_id: u64,
|
||||
request: AccountUpdateRequest,
|
||||
validate: bool,
|
||||
) -> BichonResult<()> {
|
||||
let account = AccountModel::get(account_id)?;
|
||||
if validate {
|
||||
request.validate_update_request(&account)?;
|
||||
}
|
||||
update_impl(
|
||||
DB_MANAGER.db(),
|
||||
&account_id.to_string(),
|
||||
move |current: Account| Self::apply_update_fields(¤t, request),
|
||||
)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn delete(account_id: u64) -> BichonResult<()> {
|
||||
let account = Self::get(account_id)?;
|
||||
|
||||
// Immediately stop scheduling to prevent new downloads
|
||||
if matches!(account.account_type, AccountType::IMAP) {
|
||||
SYNC_TASKS.stop(account.id).await?;
|
||||
}
|
||||
|
||||
// Mark as deleting and disabled so frontend shows status and download tasks skip it
|
||||
update_impl(
|
||||
DB_MANAGER.db(),
|
||||
&account_id.to_string(),
|
||||
move |current: Account| {
|
||||
let mut updated = current.clone();
|
||||
updated.deleting = true;
|
||||
updated.enabled = false;
|
||||
Ok(updated)
|
||||
},
|
||||
)?;
|
||||
|
||||
// Spawn background cleanup — heavy work (Tantivy, attachments) runs off the request path
|
||||
tokio::spawn(async move {
|
||||
if let Err(error) = Self::cleanup_account_resources_sequential(&account).await {
|
||||
tracing::error!(
|
||||
"[CLEANUP_ACCOUNT_ERROR] Account {}: cleanup failed, reverting deleting flag: {:#?}",
|
||||
account_id,
|
||||
error
|
||||
);
|
||||
// Revert deleting flag so the user can retry (only if account record still exists)
|
||||
let _ = update_impl(
|
||||
DB_MANAGER.db(),
|
||||
&account_id.to_string(),
|
||||
move |current: Account| {
|
||||
let mut updated = current.clone();
|
||||
updated.deleting = false;
|
||||
updated.enabled = true;
|
||||
Ok(updated)
|
||||
},
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn delete_account(account: &AccountModel) -> BichonResult<()> {
|
||||
delete_impl::<AccountModel>(DB_MANAGER.db(), &account.id.to_string())
|
||||
}
|
||||
|
||||
async fn cleanup_account_resources_sequential(account: &AccountModel) -> BichonResult<()> {
|
||||
// Sync task already stopped in delete() before spawning this background task
|
||||
if matches!(account.account_type, AccountType::IMAP) {
|
||||
DownloadState::delete(account.id)?;
|
||||
}
|
||||
OAuth2AccessToken::try_delete(account.id)?;
|
||||
UserModel::cleanup_account(account.id)?;
|
||||
MailBox::clean(account.id)?;
|
||||
ENVELOPE_MANAGER
|
||||
.delete_account_envelopes(account.id)
|
||||
.await?;
|
||||
ATTACHMENT_MANAGER
|
||||
.delete_account_attachments(account.id)
|
||||
.await?;
|
||||
Self::delete_account(account)?;
|
||||
info!("Sequential cleanup completed for account: {}", account.id);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn update_download_folders(
|
||||
account_id: u64,
|
||||
download_folders: Vec<String>,
|
||||
) -> BichonResult<()> {
|
||||
update_impl(
|
||||
DB_MANAGER.db(),
|
||||
&account_id.to_string(),
|
||||
move |current: Account| {
|
||||
let mut updated = current.clone();
|
||||
updated.download_folders = Some(download_folders);
|
||||
Ok(updated)
|
||||
},
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn update_known_folders(
|
||||
account_id: u64,
|
||||
known_folders: BTreeSet<String>,
|
||||
) -> BichonResult<()> {
|
||||
update_impl(
|
||||
DB_MANAGER.db(),
|
||||
&account_id.to_string(),
|
||||
move |current: Account| {
|
||||
let mut updated = current.clone();
|
||||
updated.known_folders = Some(known_folders);
|
||||
Ok(updated)
|
||||
},
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn update_capabilities(account_id: u64, capabilities: Vec<String>) -> BichonResult<()> {
|
||||
update_impl(
|
||||
DB_MANAGER.db(),
|
||||
&account_id.to_string(),
|
||||
move |current: Account| {
|
||||
let mut updated = current.clone();
|
||||
updated.capabilities = Some(capabilities);
|
||||
Ok(updated)
|
||||
},
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Retrieves a list of all `AccountEntity` instances.
|
||||
pub fn list_all() -> BichonResult<Vec<AccountModel>> {
|
||||
list_all_impl::<AccountModel>(DB_MANAGER.db())
|
||||
}
|
||||
|
||||
pub fn find_by_email(email: &str) -> BichonResult<Option<AccountModel>> {
|
||||
let all: Vec<AccountModel> = list_all_impl::<AccountModel>(DB_MANAGER.db())?;
|
||||
let target_email = email.trim().to_lowercase();
|
||||
|
||||
let first_match = all
|
||||
.into_iter()
|
||||
.find(|acc| acc.email.to_lowercase() == target_email);
|
||||
|
||||
Ok(first_match)
|
||||
}
|
||||
|
||||
pub fn minimal_list(only_nosync: bool) -> BichonResult<Vec<MinimalAccount>> {
|
||||
let result = list_all_impl::<AccountModel>(DB_MANAGER.db())?
|
||||
.into_iter()
|
||||
.filter(|account: &AccountModel| {
|
||||
!only_nosync || matches!(account.account_type, AccountType::NoSync)
|
||||
})
|
||||
.map(|account: AccountModel| MinimalAccount {
|
||||
id: account.id,
|
||||
email: account.email,
|
||||
name: account.account_name,
|
||||
})
|
||||
.collect::<Vec<MinimalAccount>>();
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
pub fn count() -> BichonResult<usize> {
|
||||
count_impl::<AccountModel>(DB_MANAGER.db())
|
||||
}
|
||||
|
||||
pub fn paginate_list(
|
||||
page: Option<u64>,
|
||||
page_size: Option<u64>,
|
||||
desc: Option<bool>,
|
||||
) -> BichonResult<DataPage<AccountModel>> {
|
||||
paginate_impl::<AccountModel>(DB_MANAGER.db(), page, page_size, desc).map(DataPage::from)
|
||||
}
|
||||
|
||||
// This method applies the updates from the request to the old account entity
|
||||
fn apply_update_fields(
|
||||
old: &AccountModel,
|
||||
request: AccountUpdateRequest,
|
||||
) -> BichonResult<AccountModel> {
|
||||
let mut new = old.clone();
|
||||
|
||||
if let Some(date_since) = request.date_since {
|
||||
new.date_since = Some(date_since);
|
||||
new.date_before = None;
|
||||
}
|
||||
|
||||
if let Some(date_before) = request.date_before {
|
||||
new.date_before = Some(date_before);
|
||||
new.date_since = None;
|
||||
}
|
||||
|
||||
if let Some(clear_date_range) = request.clear_date_range {
|
||||
if clear_date_range {
|
||||
new.date_since = None;
|
||||
new.date_before = None;
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(account_name) = request.account_name {
|
||||
new.account_name = Some(account_name);
|
||||
}
|
||||
|
||||
if matches!(old.account_type, AccountType::IMAP) {
|
||||
if let Some(imap) = &request.imap {
|
||||
if let Some(current_imap) = &mut new.imap {
|
||||
current_imap.host = imap.host.clone();
|
||||
current_imap.port = imap.port.clone();
|
||||
current_imap.encryption = imap.encryption.clone();
|
||||
current_imap.auth.auth_type = imap.auth.auth_type.clone();
|
||||
if let Some(password) = &imap.auth.password {
|
||||
let encrypted_password = encrypt!(password)?;
|
||||
current_imap.auth.password = Some(encrypted_password);
|
||||
}
|
||||
current_imap.use_proxy = imap.use_proxy;
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(folder_names) = request.sync_folders {
|
||||
new.download_folders = Some(folder_names);
|
||||
}
|
||||
if let Some(sync_interval_min) = &request.download_interval_min {
|
||||
new.download_interval_min = Some(*sync_interval_min);
|
||||
}
|
||||
|
||||
if let Some(download_batch_size) = &request.download_batch_size {
|
||||
new.download_batch_size = Some(*download_batch_size);
|
||||
}
|
||||
|
||||
if let Some(max_email_size_bytes) = request.max_email_size_bytes {
|
||||
new.max_email_size_bytes = Some(max_email_size_bytes);
|
||||
}
|
||||
|
||||
if let Some(use_proxy) = request.use_proxy {
|
||||
new.use_proxy = Some(use_proxy);
|
||||
}
|
||||
}
|
||||
|
||||
if matches!(old.account_type, AccountType::NoSync) {
|
||||
if let Some(email) = &request.email {
|
||||
new.email = email.clone();
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(enabled) = request.enabled {
|
||||
new.enabled = enabled;
|
||||
}
|
||||
|
||||
if let Some(use_dangerous) = request.use_dangerous {
|
||||
new.use_dangerous = use_dangerous;
|
||||
}
|
||||
|
||||
if let Some(pgp_key) = request.pgp_key {
|
||||
new.pgp_key = Some(pgp_key);
|
||||
}
|
||||
|
||||
if let Some(imap_quota_bytes) = request.imap_quota_bytes {
|
||||
new.imap_quota_bytes = Some(imap_quota_bytes);
|
||||
}
|
||||
|
||||
if let Some(imap_quota_window) = request.imap_quota_window {
|
||||
new.imap_quota_window = Some(imap_quota_window);
|
||||
}
|
||||
|
||||
if let Some(auto_download_new_mailboxes) = request.auto_download_new_mailboxes {
|
||||
new.auto_download_new_mailboxes = Some(auto_download_new_mailboxes);
|
||||
}
|
||||
if let Some(download_schedule) = request.download_schedule {
|
||||
new.download_schedule = Some(download_schedule);
|
||||
}
|
||||
if request.clear_download_schedule == Some(true) {
|
||||
new.download_schedule = None;
|
||||
}
|
||||
if request.extraction_rules.is_some() {
|
||||
new.extraction_rules = request.extraction_rules;
|
||||
}
|
||||
if request.archive_rules.is_some() {
|
||||
new.archive_rules = request.archive_rules;
|
||||
}
|
||||
new.updated_at = utc_now!();
|
||||
Ok(new)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
// ── FilterRule ───────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn filter_rule_include_only() {
|
||||
let r = FilterRule {
|
||||
include: vec![r"@ok\.com$".into()],
|
||||
..Default::default()
|
||||
};
|
||||
assert!(r.matches_regex("bob@ok.com"));
|
||||
assert!(!r.matches_regex("spam@bad.com"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn filter_rule_exclude_only() {
|
||||
let r = FilterRule {
|
||||
exclude: vec![r"@spam\.com$".into()],
|
||||
..Default::default()
|
||||
};
|
||||
assert!(r.matches_regex("bob@ok.com"));
|
||||
assert!(!r.matches_regex("bot@spam.com"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn filter_rule_include_then_exclude() {
|
||||
let r = FilterRule {
|
||||
include: vec![r"@company\.com$".into()],
|
||||
exclude: vec![r"noreply@company\.com$".into()],
|
||||
..Default::default()
|
||||
};
|
||||
assert!(r.matches_regex("bob@company.com"));
|
||||
assert!(!r.matches_regex("noreply@company.com"));
|
||||
assert!(!r.matches_regex("spam@other.com"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn filter_rule_exact_match() {
|
||||
let r = FilterRule {
|
||||
include: vec!["pdf".into(), "docx".into()],
|
||||
exclude: vec!["xlsx".into()],
|
||||
..Default::default()
|
||||
};
|
||||
assert!(r.matches_exact("pdf"));
|
||||
assert!(r.matches_exact("docx"));
|
||||
assert!(r.matches_exact("DOCX")); // case-insensitive
|
||||
assert!(!r.matches_exact("xlsx"));
|
||||
assert!(!r.matches_exact("txt"));
|
||||
}
|
||||
|
||||
// ── ExtractionRules ─────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn extraction_rules_master_switch() {
|
||||
let rules = ExtractionRules {
|
||||
enabled: false,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!rules.should_extract("pdf", None, None, None));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extraction_rules_extension_include() {
|
||||
let rules = ExtractionRules {
|
||||
enabled: true,
|
||||
extensions: FilterRule {
|
||||
include: vec!["pdf".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.should_extract("pdf", None, None, None));
|
||||
assert!(!rules.should_extract("docx", None, None, None));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extraction_rules_extension_exclude() {
|
||||
let rules = ExtractionRules {
|
||||
enabled: true,
|
||||
extensions: FilterRule {
|
||||
exclude: vec!["xlsx".into(), "pptx".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.should_extract("pdf", None, None, None));
|
||||
assert!(!rules.should_extract("xlsx", None, None, None));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extraction_rules_folder_regex() {
|
||||
let rules = ExtractionRules {
|
||||
enabled: true,
|
||||
folders: FilterRule {
|
||||
include: vec![r"^INBOX/Invoices".into(), r"Contracts$".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.should_extract("pdf", Some("INBOX/Invoices"), None, None));
|
||||
assert!(rules.should_extract("pdf", Some("Finance/Contracts"), None, None));
|
||||
assert!(!rules.should_extract("pdf", Some("INBOX/Junk"), None, None));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extraction_rules_attachment_name_regex() {
|
||||
let rules = ExtractionRules {
|
||||
enabled: true,
|
||||
attachment_names: FilterRule {
|
||||
include: vec![r"^invoice-.*\.pdf$".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.should_extract("pdf", None, Some("invoice-2024.pdf"), None));
|
||||
assert!(!rules.should_extract("pdf", None, Some("newsletter.pdf"), None));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extraction_rules_sender_regex() {
|
||||
let rules = ExtractionRules {
|
||||
enabled: true,
|
||||
senders: FilterRule {
|
||||
exclude: vec![r"@noreply\.com$".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
// non-excluded sender passes
|
||||
assert!(rules.should_extract("pdf", None, None, Some("bob@ok.com")));
|
||||
// excluded sender blocked
|
||||
assert!(!rules.should_extract("pdf", None, None, Some("bot@noreply.com")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extraction_rules_empty_filters_pass_everything() {
|
||||
let rules = ExtractionRules {
|
||||
enabled: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.should_extract(
|
||||
"anything",
|
||||
Some("any/folder"),
|
||||
Some("any.pdf"),
|
||||
Some("any@x.com")
|
||||
));
|
||||
}
|
||||
|
||||
// ── ArchiveRules ────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn archive_rules_disabled_archives_everything() {
|
||||
let rules = ArchiveRules {
|
||||
enabled: false,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.should_archive(Some("spam@x.com"), Some("BUY NOW"), 999, false));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn archive_rules_sender_exclude() {
|
||||
let rules = ArchiveRules {
|
||||
enabled: true,
|
||||
senders: FilterRule {
|
||||
exclude: vec![r"@spam\.com$".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!rules.should_archive(Some("bot@spam.com"), None, 100, false));
|
||||
assert!(rules.should_archive(Some("friend@ok.com"), None, 100, false));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn archive_rules_subject_exclude() {
|
||||
let rules = ArchiveRules {
|
||||
enabled: true,
|
||||
subjects: FilterRule {
|
||||
exclude: vec![r"(?i)unsubscribe|buy now|limited offer".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!rules.should_archive(None, Some("UNSUBSCRIBE NOW"), 100, false));
|
||||
assert!(!rules.should_archive(None, Some("Limited Offer!!"), 100, false));
|
||||
assert!(rules.should_archive(None, Some("Meeting tomorrow"), 100, false));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn archive_rules_sender_include() {
|
||||
// Only archive emails from specific senders
|
||||
let rules = ArchiveRules {
|
||||
enabled: true,
|
||||
senders: FilterRule {
|
||||
include: vec![r"@partner\.com$".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.should_archive(Some("bob@partner.com"), None, 100, false));
|
||||
assert!(!rules.should_archive(Some("spam@random.com"), None, 100, false));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn archive_rules_skip_larger_than() {
|
||||
let rules = ArchiveRules {
|
||||
enabled: true,
|
||||
skip_larger_than: Some(50_000_000),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.should_archive(None, None, 1_000_000, false));
|
||||
assert!(!rules.should_archive(None, None, 60_000_000, false));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn archive_rules_skip_spam_headers() {
|
||||
let rules = ArchiveRules {
|
||||
enabled: true,
|
||||
spam_headers: vec!["X-Spam-Flag".into()],
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!rules.should_archive(None, None, 100, true));
|
||||
assert!(rules.should_archive(None, None, 100, false));
|
||||
}
|
||||
|
||||
// ── Validation ──────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn validate_extraction_rules_valid() {
|
||||
let rules = ExtractionRules {
|
||||
folders: FilterRule {
|
||||
include: vec![r"^INBOX/.*".into()],
|
||||
..Default::default()
|
||||
},
|
||||
senders: FilterRule {
|
||||
exclude: vec![r"@spam\.com$".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.validate().is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_extraction_rules_invalid_regex() {
|
||||
let rules = ExtractionRules {
|
||||
folders: FilterRule {
|
||||
include: vec!["***bad[".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.validate().is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_archive_rules_valid() {
|
||||
let rules = ArchiveRules {
|
||||
senders: FilterRule {
|
||||
exclude: vec![r"@spam\.com$".into()],
|
||||
..Default::default()
|
||||
},
|
||||
subjects: FilterRule {
|
||||
include: vec![r"(?i)invoice".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.validate().is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_archive_rules_invalid_regex() {
|
||||
let rules = ArchiveRules {
|
||||
senders: FilterRule {
|
||||
include: vec!["[unclosed".into()],
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
assert!(rules.validate().is_err());
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,11 +16,12 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
pub mod dispatcher;
|
||||
pub mod entity;
|
||||
pub mod grant;
|
||||
pub mod migration;
|
||||
pub mod old_state;
|
||||
pub mod payload;
|
||||
pub mod since;
|
||||
pub mod state;
|
||||
pub mod stats;
|
||||
pub mod view;
|
||||
245
crates/core/src/account/old_state.rs
Normal file
245
crates/core/src/account/old_state.rs
Normal file
@@ -0,0 +1,245 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
pub struct MailboxBatchProgress {
|
||||
pub total_batches: u32,
|
||||
pub current_batch: u32,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
pub struct AccountRunningState {
|
||||
pub account_id: u64,
|
||||
pub last_incremental_sync_start: i64,
|
||||
pub last_incremental_sync_end: Option<i64>,
|
||||
pub errors: Vec<AccountError>,
|
||||
pub is_initial_sync_completed: bool,
|
||||
pub progress: Option<BTreeMap<String, MailboxBatchProgress>>,
|
||||
pub initial_sync_start_time: Option<i64>,
|
||||
pub initial_sync_end_time: Option<i64>,
|
||||
pub initial_sync_failed_time: Option<i64>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
pub struct AccountError {
|
||||
pub error: String,
|
||||
pub at: i64,
|
||||
}
|
||||
|
||||
// impl AccountRunningState {
|
||||
// pub async fn add(account_id: u64) -> BichonResult<()> {
|
||||
// let info = AccountRunningState {
|
||||
// account_id,
|
||||
// last_incremental_sync_start: 0,
|
||||
// last_incremental_sync_end: None,
|
||||
// errors: vec![],
|
||||
// is_initial_sync_completed: false,
|
||||
// progress: None,
|
||||
// initial_sync_start_time: Some(utc_now!()),
|
||||
// initial_sync_end_time: None,
|
||||
// initial_sync_failed_time: None,
|
||||
// };
|
||||
// upsert_impl(DB_MANAGER.envelope_db(), info).await
|
||||
// }
|
||||
|
||||
// pub async fn get(account_id: u64) -> BichonResult<Option<AccountRunningState>> {
|
||||
// async_find_impl(DB_MANAGER.envelope_db(), account_id).await
|
||||
// }
|
||||
|
||||
// async fn update_account_running_state(
|
||||
// account_id: u64,
|
||||
// updater: impl FnOnce(&AccountRunningState) -> BichonResult<AccountRunningState> + Send + 'static,
|
||||
// ) -> BichonResult<()> {
|
||||
// if Self::get(account_id).await?.is_some() {
|
||||
// update_impl(
|
||||
// DB_MANAGER.envelope_db(),
|
||||
// move |rw| {
|
||||
// rw.get()
|
||||
// .primary::<AccountRunningState>(account_id)
|
||||
// .map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
// .ok_or_else(|| {
|
||||
// raise_error!(
|
||||
// format!("Cannot find sync info of account={}", account_id),
|
||||
// ErrorCode::ResourceNotFound
|
||||
// )
|
||||
// })
|
||||
// },
|
||||
// updater,
|
||||
// )
|
||||
// .await?;
|
||||
// }
|
||||
// Ok(())
|
||||
// }
|
||||
|
||||
// pub async fn delete(account_id: u64) -> BichonResult<()> {
|
||||
// if Self::get(account_id).await?.is_none() {
|
||||
// return Ok(());
|
||||
// }
|
||||
|
||||
// delete_impl(DB_MANAGER.envelope_db(), move |rw| {
|
||||
// rw.get()
|
||||
// .primary::<AccountRunningState>(account_id)
|
||||
// .map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
// .ok_or_else(|| {
|
||||
// raise_error!(
|
||||
// format!(
|
||||
// "AccountRunningState '{}' not found during deletion process.",
|
||||
// account_id
|
||||
// ),
|
||||
// ErrorCode::ResourceNotFound
|
||||
// )
|
||||
// })
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// // pub async fn set_initial_sync_start(account_id: u64) -> BichonResult<()> {
|
||||
// // Self::update_account_running_state(account_id, move |current| {
|
||||
// // let mut updated = current.clone();
|
||||
// // updated.initial_sync_start_time = Some(utc_now!());
|
||||
// // Ok(updated)
|
||||
// // })
|
||||
// // .await
|
||||
// // }
|
||||
|
||||
// pub async fn set_initial_sync_completed(account_id: u64) -> BichonResult<()> {
|
||||
// Self::update_account_running_state(account_id, move |current| {
|
||||
// let mut updated = current.clone();
|
||||
// updated.is_initial_sync_completed = true;
|
||||
// updated.initial_sync_end_time = Some(utc_now!());
|
||||
// Ok(updated)
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// pub async fn set_initial_sync_failed(account_id: u64) -> BichonResult<()> {
|
||||
// Self::update_account_running_state(account_id, move |current| {
|
||||
// let mut updated = current.clone();
|
||||
// updated.initial_sync_failed_time = Some(utc_now!());
|
||||
// Ok(updated)
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// pub async fn set_current_sync_batch_number(
|
||||
// account_id: u64,
|
||||
// syncing_folder: String,
|
||||
// batch_number: u32,
|
||||
// ) -> BichonResult<()> {
|
||||
// Self::update_account_running_state(account_id, move |current| {
|
||||
// let mut updated = current.clone();
|
||||
// let mut progress_map = updated.progress.clone().unwrap_or_default();
|
||||
// let entry =
|
||||
// progress_map
|
||||
// .entry(syncing_folder.to_string())
|
||||
// .or_insert(MailboxBatchProgress {
|
||||
// total_batches: 0,
|
||||
// current_batch: 0,
|
||||
// });
|
||||
// entry.current_batch = batch_number;
|
||||
// updated.progress = Some(progress_map);
|
||||
// Ok(updated)
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// pub async fn set_folder_initial_sync_completed(
|
||||
// account_id: u64,
|
||||
// syncing_folder: String,
|
||||
// ) -> BichonResult<()> {
|
||||
// Self::update_account_running_state(account_id, move |current| {
|
||||
// let mut updated = current.clone();
|
||||
// let mut progress_map = updated.progress.clone().unwrap_or_default();
|
||||
// let entry =
|
||||
// progress_map
|
||||
// .entry(syncing_folder.to_string())
|
||||
// .or_insert(MailboxBatchProgress {
|
||||
// total_batches: 0,
|
||||
// current_batch: 0,
|
||||
// });
|
||||
// entry.current_batch = entry.total_batches;
|
||||
// updated.progress = Some(progress_map);
|
||||
// Ok(updated)
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// pub async fn set_initial_current_syncing_folder(
|
||||
// account_id: u64,
|
||||
// current_syncing_folder: String,
|
||||
// total_sync_batches: u32,
|
||||
// ) -> BichonResult<()> {
|
||||
// Self::update_account_running_state(account_id, move |current| {
|
||||
// let mut updated = current.clone();
|
||||
// let mut progress_map = updated.progress.clone().unwrap_or_default();
|
||||
// progress_map.insert(
|
||||
// current_syncing_folder.clone(),
|
||||
// MailboxBatchProgress {
|
||||
// total_batches: total_sync_batches,
|
||||
// current_batch: 0,
|
||||
// },
|
||||
// );
|
||||
// updated.progress = Some(progress_map);
|
||||
// Ok(updated)
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// pub async fn set_incremental_sync_start(account_id: u64) -> BichonResult<()> {
|
||||
// Self::update_account_running_state(account_id, move |current| {
|
||||
// let mut updated = current.clone();
|
||||
// updated.last_incremental_sync_start = utc_now!();
|
||||
// updated.last_incremental_sync_end = None;
|
||||
// Ok(updated)
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// pub async fn set_incremental_sync_end(account_id: u64) -> BichonResult<()> {
|
||||
// Self::update_account_running_state(account_id, move |current| {
|
||||
// let mut updated = current.clone();
|
||||
// updated.last_incremental_sync_end = Some(utc_now!());
|
||||
// Ok(updated)
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// pub async fn append_error_message(account_id: u64, error: String) -> BichonResult<()> {
|
||||
// Self::update_account_running_state(account_id, move |current| {
|
||||
// let mut updated = current.clone();
|
||||
// updated.append_error_log(error);
|
||||
// Ok(updated)
|
||||
// })
|
||||
// .await
|
||||
// }
|
||||
|
||||
// pub fn append_error_log(&mut self, error: String) {
|
||||
// let new_error = AccountError {
|
||||
// error,
|
||||
// at: utc_now!(),
|
||||
// };
|
||||
|
||||
// self.errors.push(new_error);
|
||||
// if self.errors.len() > ERROR_COUNT_PER_ACCOUNT {
|
||||
// self.errors.remove(0);
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,34 +16,54 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::modules::account::entity::ImapConfig;
|
||||
use crate::modules::account::migration::{AccountModel, AccountType};
|
||||
use crate::modules::account::since::{DateSince, RelativeDate};
|
||||
use crate::modules::error::code::ErrorCode;
|
||||
use crate::modules::error::BichonResult;
|
||||
use std::str::FromStr;
|
||||
|
||||
use crate::account::entity::ImapConfig;
|
||||
use crate::account::migration::{
|
||||
AccountModel, AccountType, ArchiveRules, ExtractionRules, QuotaWindow,
|
||||
};
|
||||
use crate::account::since::{DateSince, RelativeDate};
|
||||
use crate::error::code::ErrorCode;
|
||||
use crate::error::BichonResult;
|
||||
use crate::{raise_error, validate_email};
|
||||
use poem_openapi::Object;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct AccountCreateRequest {
|
||||
#[oai(validator(custom = "crate::modules::common::validator::EmailValidator"))]
|
||||
#[cfg_attr(
|
||||
feature = "web-api",
|
||||
oai(validator(custom = "crate::common::validator::EmailValidator"))
|
||||
)]
|
||||
pub email: String,
|
||||
pub name: Option<String>,
|
||||
pub login_name: Option<String>,
|
||||
pub account_name: Option<String>,
|
||||
pub imap: Option<ImapConfig>,
|
||||
pub enabled: bool,
|
||||
pub date_since: Option<DateSince>,
|
||||
pub date_before: Option<RelativeDate>,
|
||||
pub account_type: AccountType,
|
||||
#[oai(validator(minimum(value = "100")))]
|
||||
pub folder_limit: Option<u32>,
|
||||
#[oai(validator(minimum(value = "10")))]
|
||||
pub sync_interval_min: Option<i64>,
|
||||
#[oai(validator(minimum(value = "30"), maximum(value = "200")))]
|
||||
pub sync_batch_size: Option<u32>,
|
||||
#[cfg_attr(feature = "web-api", oai(validator(minimum(value = "10"))))]
|
||||
pub download_interval_min: Option<i64>,
|
||||
#[cfg_attr(
|
||||
feature = "web-api",
|
||||
oai(validator(minimum(value = "10"), maximum(value = "200")))
|
||||
)]
|
||||
pub download_batch_size: Option<u32>,
|
||||
pub max_email_size_bytes: Option<u64>,
|
||||
pub use_proxy: Option<u64>,
|
||||
pub use_dangerous: bool,
|
||||
pub pgp_key: Option<String>,
|
||||
pub imap_quota_bytes: Option<u64>,
|
||||
pub imap_quota_window: Option<QuotaWindow>,
|
||||
pub auto_download_new_mailboxes: Option<bool>,
|
||||
pub download_schedule: Option<String>,
|
||||
/// Email archive filtering rules (Pro feature).
|
||||
/// `None` = archive everything (backward compatible).
|
||||
pub archive_rules: Option<ArchiveRules>,
|
||||
/// Attachment text extraction rules (Pro feature).
|
||||
/// `None` = extract everything (backward compatible).
|
||||
pub extraction_rules: Option<ExtractionRules>,
|
||||
}
|
||||
|
||||
impl AccountCreateRequest {
|
||||
@@ -56,6 +76,13 @@ impl AccountCreateRequest {
|
||||
));
|
||||
}
|
||||
|
||||
if self.imap_quota_bytes.is_some() ^ self.imap_quota_window.is_some() {
|
||||
return Err(raise_error!(
|
||||
"Quota bytes and quota window must be provided together or omitted together".into(),
|
||||
ErrorCode::InvalidParameter
|
||||
));
|
||||
}
|
||||
|
||||
if let Some(date_since) = self.date_since.as_ref() {
|
||||
date_since.validate()?;
|
||||
}
|
||||
@@ -75,15 +102,31 @@ impl AccountCreateRequest {
|
||||
))
|
||||
}
|
||||
}
|
||||
if self.sync_interval_min.is_none() {
|
||||
if self.download_interval_min.is_none() && self.download_schedule.is_none() {
|
||||
return Err(raise_error!(
|
||||
"`sync_interval_min` is required for IMAP account type".into(),
|
||||
"`sync_interval_min` or `download_schedule` is required for IMAP account type".into(),
|
||||
ErrorCode::InvalidParameter
|
||||
));
|
||||
}
|
||||
if let Some(ref schedule) = self.download_schedule {
|
||||
validate_cron_expression(schedule)?;
|
||||
}
|
||||
}
|
||||
AccountType::NoSync => {}
|
||||
}
|
||||
if let Some(ref rules) = self.extraction_rules {
|
||||
rules.validate().map_err(|e| {
|
||||
raise_error!(
|
||||
format!("extraction_rules: {}", e),
|
||||
ErrorCode::InvalidParameter
|
||||
)
|
||||
})?;
|
||||
}
|
||||
if let Some(ref rules) = self.archive_rules {
|
||||
rules.validate().map_err(|e| {
|
||||
raise_error!(format!("archive_rules: {}", e), ErrorCode::InvalidParameter)
|
||||
})?;
|
||||
}
|
||||
Ok(AccountModel::new(user_id, self)?)
|
||||
}
|
||||
|
||||
@@ -96,7 +139,8 @@ impl AccountCreateRequest {
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct AccountUpdateRequest {
|
||||
pub email: Option<String>,
|
||||
/// Represents the account activation status.
|
||||
@@ -105,8 +149,7 @@ pub struct AccountUpdateRequest {
|
||||
/// and any attempts to access them should return an error indicating the account
|
||||
/// is inactive.
|
||||
pub enabled: Option<bool>,
|
||||
/// Display name for the account (optional)
|
||||
pub name: Option<String>,
|
||||
pub account_name: Option<String>,
|
||||
/// IMAP server configuration
|
||||
pub imap: Option<ImapConfig>,
|
||||
/// Controls initial synchronization time range
|
||||
@@ -122,12 +165,6 @@ pub struct AccountUpdateRequest {
|
||||
pub date_since: Option<DateSince>,
|
||||
pub date_before: Option<RelativeDate>,
|
||||
pub clear_date_range: Option<bool>,
|
||||
/// Max emails to sync for this folder.
|
||||
/// If not set, sync all emails.
|
||||
/// otherwise sync up to `n` most recent emails (min 10).
|
||||
#[oai(validator(minimum(value = "100")))]
|
||||
pub folder_limit: Option<u32>,
|
||||
pub clear_folder_limit: Option<bool>,
|
||||
/// Configuration for selective folder (mailbox/label) synchronization
|
||||
///
|
||||
/// - For IMAP/SMTP accounts:
|
||||
@@ -142,11 +179,15 @@ pub struct AccountUpdateRequest {
|
||||
/// Defaults to standard folders (`INBOX`, `Sent`) if empty.
|
||||
/// Modified folders will be automatically synced on the next update.
|
||||
pub sync_folders: Option<Vec<String>>,
|
||||
/// Incremental sync interval (seconds)
|
||||
#[oai(validator(minimum(value = "10")))]
|
||||
pub sync_interval_min: Option<i64>,
|
||||
#[oai(validator(minimum(value = "30"), maximum(value = "200")))]
|
||||
pub sync_batch_size: Option<u32>,
|
||||
/// Incremental download interval (seconds)
|
||||
#[cfg_attr(feature = "web-api", oai(validator(minimum(value = "10"))))]
|
||||
pub download_interval_min: Option<i64>,
|
||||
#[cfg_attr(
|
||||
feature = "web-api",
|
||||
oai(validator(minimum(value = "10"), maximum(value = "200")))
|
||||
)]
|
||||
pub download_batch_size: Option<u32>,
|
||||
pub max_email_size_bytes: Option<u64>,
|
||||
/// Optional proxy ID for establishing the connection to external APIs (e.g., Gmail, Outlook).
|
||||
/// - If `None` or not provided, the client will connect directly to the API server.
|
||||
/// - If `Some(proxy_id)`, the client will use the pre-configured proxy with the given ID for API requests.
|
||||
@@ -155,6 +196,17 @@ pub struct AccountUpdateRequest {
|
||||
pub use_dangerous: Option<bool>,
|
||||
|
||||
pub pgp_key: Option<String>,
|
||||
pub imap_quota_bytes: Option<u64>,
|
||||
pub imap_quota_window: Option<QuotaWindow>,
|
||||
pub auto_download_new_mailboxes: Option<bool>,
|
||||
pub download_schedule: Option<String>,
|
||||
pub clear_download_schedule: Option<bool>,
|
||||
/// Email archive filtering rules (Pro feature).
|
||||
/// `None` = no change. Use `Some(ArchiveRules { .. })` to set.
|
||||
pub archive_rules: Option<ArchiveRules>,
|
||||
/// Attachment text extraction rules (Pro feature).
|
||||
/// `None` = no change. Use `Some(ExtractionRules { .. })` to set.
|
||||
pub extraction_rules: Option<ExtractionRules>,
|
||||
}
|
||||
|
||||
impl AccountUpdateRequest {
|
||||
@@ -167,9 +219,9 @@ impl AccountUpdateRequest {
|
||||
));
|
||||
}
|
||||
|
||||
if self.clear_folder_limit == Some(true) && self.folder_limit.is_some() {
|
||||
if self.imap_quota_bytes.is_some() ^ self.imap_quota_window.is_some() {
|
||||
return Err(raise_error!(
|
||||
"clear_folder_limit cannot be combined with folder_limit".into(),
|
||||
"Quota bytes and quota window must be provided together or omitted together".into(),
|
||||
ErrorCode::InvalidParameter
|
||||
));
|
||||
}
|
||||
@@ -200,16 +252,56 @@ impl AccountUpdateRequest {
|
||||
));
|
||||
}
|
||||
}
|
||||
if self.clear_download_schedule == Some(true) && self.download_schedule.is_some() {
|
||||
return Err(raise_error!(
|
||||
"clear_download_schedule cannot be combined with download_schedule".into(),
|
||||
ErrorCode::InvalidParameter
|
||||
));
|
||||
}
|
||||
if let Some(ref schedule) = self.download_schedule {
|
||||
validate_cron_expression(schedule)?;
|
||||
}
|
||||
}
|
||||
if let Some(ref rules) = self.extraction_rules {
|
||||
rules.validate().map_err(|e| {
|
||||
raise_error!(
|
||||
format!("extraction_rules: {}", e),
|
||||
ErrorCode::InvalidParameter
|
||||
)
|
||||
})?;
|
||||
}
|
||||
if let Some(ref rules) = self.archive_rules {
|
||||
rules.validate().map_err(|e| {
|
||||
raise_error!(format!("archive_rules: {}", e), ErrorCode::InvalidParameter)
|
||||
})?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
fn validate_cron_expression(expr: &str) -> BichonResult<()> {
|
||||
if expr.trim().is_empty() {
|
||||
return Err(raise_error!(
|
||||
"download_schedule must not be empty".into(),
|
||||
ErrorCode::InvalidParameter
|
||||
));
|
||||
}
|
||||
cron::Schedule::from_str(expr).map_err(|e| {
|
||||
raise_error!(
|
||||
format!("Invalid cron expression '{}': {}", expr, e),
|
||||
ErrorCode::InvalidParameter
|
||||
)
|
||||
})?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
|
||||
pub struct MinimalAccount {
|
||||
pub id: u64,
|
||||
pub email: String,
|
||||
pub name: Option<String>,
|
||||
}
|
||||
|
||||
pub fn filter_accessible_accounts<'a>(
|
||||
@@ -222,3 +314,34 @@ pub fn filter_accessible_accounts<'a>(
|
||||
.cloned()
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod test {
|
||||
use super::validate_cron_expression;
|
||||
|
||||
#[test]
|
||||
fn valid_cron_expressions() {
|
||||
assert!(validate_cron_expression("0 0 0 * * *").is_ok()); // daily at midnight
|
||||
assert!(validate_cron_expression("0 */5 * * * *").is_ok()); // every 5 minutes
|
||||
assert!(validate_cron_expression("0 0 12 * * 1-5").is_ok()); // weekdays at noon
|
||||
assert!(validate_cron_expression("0 30 4 1 * *").is_ok()); // 1st of month at 04:30
|
||||
assert!(validate_cron_expression("0 0 * * * *").is_ok()); // every hour
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_cron_expression_too_few_fields() {
|
||||
assert!(validate_cron_expression("0 0 * *").is_err());
|
||||
assert!(validate_cron_expression("* * * * *").is_err()); // 5 fields, needs seconds
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_cron_expression_empty() {
|
||||
assert!(validate_cron_expression("").is_err());
|
||||
assert!(validate_cron_expression(" ").is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_cron_expression_garbage() {
|
||||
assert!(validate_cron_expression("not a cron").is_err());
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -17,14 +17,14 @@
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::{
|
||||
modules::error::{code::ErrorCode, BichonResult},
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
raise_error,
|
||||
};
|
||||
use chrono::{Datelike, Days, Local, Months, NaiveDate, Utc};
|
||||
use poem_openapi::{Enum, Object};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct DateSince {
|
||||
/// Absolute date boundary in ISO 8601 format (YYYY-MM-DD)
|
||||
///
|
||||
@@ -38,7 +38,7 @@ pub struct DateSince {
|
||||
/// "fixed": "2025-05-01"
|
||||
/// }
|
||||
/// ```
|
||||
#[oai(validator(pattern = r"^\d{4}-\d{2}-\d{2}$"))]
|
||||
#[cfg_attr(feature = "web-api", oai(validator(pattern = r"^\d{4}-\d{2}-\d{2}$")))]
|
||||
pub fixed: Option<String>,
|
||||
/// Relative time period from current date
|
||||
///
|
||||
@@ -58,7 +58,8 @@ pub struct DateSince {
|
||||
pub relative: Option<RelativeDate>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Enum)]
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum Unit {
|
||||
#[default]
|
||||
Days,
|
||||
@@ -66,12 +67,13 @@ pub enum Unit {
|
||||
Years,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct RelativeDate {
|
||||
/// The time unit to use for the offset (days, months, or years)
|
||||
pub unit: Unit,
|
||||
/// The quantity of time units to offset (must be a positive integer)
|
||||
#[oai(validator(minimum(value = "1")))]
|
||||
#[cfg_attr(feature = "web-api", oai(validator(minimum(value = "1"))))]
|
||||
pub value: u32,
|
||||
}
|
||||
|
||||
@@ -255,19 +257,47 @@ impl DateSince {
|
||||
|
||||
#[cfg(test)]
|
||||
mod test {
|
||||
use crate::modules::account::since::{DateSince, RelativeDate, Unit};
|
||||
use crate::account::since::{DateSince, RelativeDate, Unit};
|
||||
|
||||
#[test]
|
||||
fn test1() {
|
||||
fn fixed_date_valid() {
|
||||
let e = DateSince {
|
||||
fixed: Some("2014-09-12".to_string()),
|
||||
relative: None,
|
||||
};
|
||||
assert!(e.validate().is_ok());
|
||||
assert!(!e.since_date().unwrap().is_empty());
|
||||
}
|
||||
|
||||
e.validate().unwrap();
|
||||
#[test]
|
||||
fn fixed_date_in_future_fails() {
|
||||
let e = DateSince {
|
||||
fixed: Some("2099-01-01".to_string()),
|
||||
relative: None,
|
||||
};
|
||||
assert!(e.validate().is_err());
|
||||
}
|
||||
|
||||
println!("{}", e.since_date().unwrap());
|
||||
#[test]
|
||||
fn fixed_date_before_1970_fails() {
|
||||
let e = DateSince {
|
||||
fixed: Some("1960-01-01".to_string()),
|
||||
relative: None,
|
||||
};
|
||||
assert!(e.validate().is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fixed_date_bad_format_fails() {
|
||||
let e = DateSince {
|
||||
fixed: Some("01-01-2020".to_string()),
|
||||
relative: None,
|
||||
};
|
||||
assert!(e.validate().is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn relative_date_days_valid() {
|
||||
let e = DateSince {
|
||||
fixed: None,
|
||||
relative: Some(RelativeDate {
|
||||
@@ -275,9 +305,102 @@ mod test {
|
||||
value: 1,
|
||||
}),
|
||||
};
|
||||
assert!(e.validate().is_ok());
|
||||
}
|
||||
|
||||
e.validate().unwrap();
|
||||
#[test]
|
||||
fn relative_date_months_valid() {
|
||||
let e = DateSince {
|
||||
fixed: None,
|
||||
relative: Some(RelativeDate {
|
||||
unit: Unit::Months,
|
||||
value: 3,
|
||||
}),
|
||||
};
|
||||
assert!(e.validate().is_ok());
|
||||
}
|
||||
|
||||
println!("{}", e.since_date().unwrap());
|
||||
#[test]
|
||||
fn relative_date_years_valid() {
|
||||
let e = DateSince {
|
||||
fixed: None,
|
||||
relative: Some(RelativeDate {
|
||||
unit: Unit::Years,
|
||||
value: 1,
|
||||
}),
|
||||
};
|
||||
assert!(e.validate().is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn relative_date_zero_value_fails() {
|
||||
let e = DateSince {
|
||||
fixed: None,
|
||||
relative: Some(RelativeDate {
|
||||
unit: Unit::Days,
|
||||
value: 0,
|
||||
}),
|
||||
};
|
||||
assert!(e.validate().is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn both_fixed_and_relative_fails() {
|
||||
let e = DateSince {
|
||||
fixed: Some("2014-09-12".to_string()),
|
||||
relative: Some(RelativeDate {
|
||||
unit: Unit::Days,
|
||||
value: 1,
|
||||
}),
|
||||
};
|
||||
assert!(e.validate().is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn neither_fixed_nor_relative_fails() {
|
||||
let e = DateSince {
|
||||
fixed: None,
|
||||
relative: None,
|
||||
};
|
||||
assert!(e.validate().is_err());
|
||||
}
|
||||
|
||||
// ── Sliding window tests ──────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn relative_date_calculate_returns_valid_format() {
|
||||
let r = RelativeDate {
|
||||
unit: Unit::Years,
|
||||
value: 1,
|
||||
};
|
||||
let date_str = r.calculate_date().unwrap();
|
||||
// Expect format like "26-May-2025"
|
||||
assert!(date_str.len() > 5);
|
||||
assert!(date_str.contains('-'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn relative_date_one_year_ago_is_before_now() {
|
||||
let r = RelativeDate {
|
||||
unit: Unit::Years,
|
||||
value: 1,
|
||||
};
|
||||
let date_str = r.calculate_date().unwrap();
|
||||
let parsed = chrono::NaiveDate::parse_from_str(&date_str, "%d-%b-%Y").unwrap();
|
||||
let today = chrono::Local::now().date_naive();
|
||||
assert!(parsed < today, "1 year ago ({parsed}) should be before today ({today})");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn relative_date_one_day_ago_is_yesterday() {
|
||||
let r = RelativeDate {
|
||||
unit: Unit::Days,
|
||||
value: 1,
|
||||
};
|
||||
let date_str = r.calculate_date().unwrap();
|
||||
let parsed = chrono::NaiveDate::parse_from_str(&date_str, "%d-%b-%Y").unwrap();
|
||||
let today = chrono::Local::now().date_naive();
|
||||
let yesterday = today - chrono::Duration::days(1);
|
||||
assert_eq!(parsed, yesterday, "1 day ago should be yesterday");
|
||||
}
|
||||
}
|
||||
283
crates/core/src/account/state.rs
Normal file
283
crates/core/src/account/state.rs
Normal file
@@ -0,0 +1,283 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::{
|
||||
database::{delete_impl, find_impl, manager::DB_MANAGER, update_impl, upsert_impl, MemDbModel},
|
||||
error::BichonResult,
|
||||
utc_now,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum DownloadStatus {
|
||||
Running,
|
||||
Success,
|
||||
Failed,
|
||||
#[default]
|
||||
Cancelled,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum TriggerType {
|
||||
Manual,
|
||||
#[default]
|
||||
Scheduled,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum FolderStatus {
|
||||
#[default]
|
||||
Pending,
|
||||
Downloading,
|
||||
Success,
|
||||
Failed,
|
||||
Cancelled,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct FolderProgress {
|
||||
pub folder_name: String,
|
||||
pub planned: u64,
|
||||
pub current: u64,
|
||||
pub status: FolderStatus,
|
||||
pub message: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct DownloadSession {
|
||||
pub start_time: i64,
|
||||
pub end_time: Option<i64>,
|
||||
pub status: DownloadStatus,
|
||||
pub message: Option<String>,
|
||||
pub trigger: TriggerType,
|
||||
pub folder_details: BTreeMap<String, FolderProgress>,
|
||||
pub current_folder: Option<String>,
|
||||
pub errors: Vec<AccountError>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct DownloadState {
|
||||
pub account_id: u64,
|
||||
pub active_session: Option<DownloadSession>,
|
||||
pub history: Vec<DownloadSession>,
|
||||
pub last_trigger_at: i64,
|
||||
pub last_finished_at: Option<i64>,
|
||||
}
|
||||
|
||||
impl MemDbModel for DownloadState {
|
||||
fn collection() -> &'static str {
|
||||
"download_states"
|
||||
}
|
||||
fn key(&self) -> String {
|
||||
self.account_id.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct AccountError {
|
||||
pub error: String,
|
||||
pub at: i64,
|
||||
}
|
||||
|
||||
impl DownloadState {
|
||||
pub fn empty(account_id: u64) -> Self {
|
||||
DownloadState {
|
||||
account_id,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn init(account_id: u64) -> BichonResult<()> {
|
||||
let now = utc_now!();
|
||||
let state = DownloadState {
|
||||
account_id,
|
||||
last_trigger_at: now,
|
||||
active_session: Some(DownloadSession {
|
||||
start_time: now,
|
||||
status: DownloadStatus::Running,
|
||||
trigger: TriggerType::Scheduled,
|
||||
..Default::default()
|
||||
}),
|
||||
history: Default::default(),
|
||||
last_finished_at: Default::default(),
|
||||
};
|
||||
upsert_impl(DB_MANAGER.db(), state)
|
||||
}
|
||||
|
||||
pub fn get(account_id: u64) -> BichonResult<Option<DownloadState>> {
|
||||
find_impl::<DownloadState>(DB_MANAGER.db(), &account_id.to_string())
|
||||
}
|
||||
|
||||
pub fn start_new_session(account_id: u64, trigger: TriggerType) -> BichonResult<()> {
|
||||
Self::update_state(account_id, move |current| {
|
||||
let mut updated = current.clone();
|
||||
updated.last_trigger_at = utc_now!();
|
||||
|
||||
if let Some(mut old_session) = updated.active_session.take() {
|
||||
if old_session.status == DownloadStatus::Running {
|
||||
old_session.status = DownloadStatus::Cancelled;
|
||||
old_session.end_time = Some(utc_now!());
|
||||
old_session.message = Some("Interrupted by a new download session.".into());
|
||||
}
|
||||
updated.history.push(old_session);
|
||||
if updated.history.len() > 30 {
|
||||
updated.history.remove(0);
|
||||
}
|
||||
}
|
||||
|
||||
let new_session = DownloadSession {
|
||||
start_time: utc_now!(),
|
||||
status: DownloadStatus::Running,
|
||||
trigger,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
updated.active_session = Some(new_session);
|
||||
Ok(updated)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn update_session_status(
|
||||
account_id: u64,
|
||||
status: DownloadStatus,
|
||||
message: Option<String>,
|
||||
) -> BichonResult<()> {
|
||||
Self::update_state(account_id, move |current| {
|
||||
let mut updated = current.clone();
|
||||
if let Some(mut session) = updated.active_session.take() {
|
||||
session.status = status.clone();
|
||||
if message.is_some() {
|
||||
session.message = message;
|
||||
}
|
||||
if status == DownloadStatus::Running {
|
||||
updated.active_session = Some(session);
|
||||
} else {
|
||||
let now = utc_now!();
|
||||
session.end_time = Some(now);
|
||||
updated.last_finished_at = Some(now);
|
||||
updated.history.push(session);
|
||||
let to_remove = updated.history.len().saturating_sub(10);
|
||||
if to_remove > 0 {
|
||||
updated.history.drain(0..to_remove);
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(updated)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn update_folder_progress(
|
||||
account_id: u64,
|
||||
folder_name: String,
|
||||
planned: u64,
|
||||
current: u64,
|
||||
status: FolderStatus,
|
||||
message: Option<String>,
|
||||
) -> BichonResult<()> {
|
||||
Self::update_state(account_id, move |state| {
|
||||
let mut updated = state.clone();
|
||||
if let Some(ref mut session) = updated.active_session {
|
||||
session.current_folder = Some(folder_name.clone());
|
||||
|
||||
let progress =
|
||||
session
|
||||
.folder_details
|
||||
.entry(folder_name.clone())
|
||||
.or_insert(FolderProgress {
|
||||
folder_name,
|
||||
..Default::default()
|
||||
});
|
||||
|
||||
progress.planned = planned;
|
||||
progress.current = current;
|
||||
progress.status = status;
|
||||
progress.message = message;
|
||||
}
|
||||
Ok(updated)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn init_folder_details(account_id: u64, folders: Vec<String>) -> BichonResult<()> {
|
||||
Self::update_state(account_id, move |state| {
|
||||
let mut updated = state.clone();
|
||||
if let Some(ref mut session) = updated.active_session {
|
||||
for name in folders {
|
||||
session.folder_details.insert(
|
||||
name.clone(),
|
||||
FolderProgress {
|
||||
folder_name: name,
|
||||
planned: 0,
|
||||
current: 0,
|
||||
status: FolderStatus::Pending,
|
||||
message: None,
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
Ok(updated)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn append_session_error(account_id: u64, error: String) -> BichonResult<()> {
|
||||
Self::update_state(account_id, move |current| {
|
||||
let mut updated = current.clone();
|
||||
let new_error = AccountError {
|
||||
error,
|
||||
at: utc_now!(),
|
||||
};
|
||||
let target = updated
|
||||
.active_session
|
||||
.as_mut()
|
||||
.or_else(|| updated.history.last_mut());
|
||||
if let Some(session) = target {
|
||||
session.errors.push(new_error);
|
||||
let to_remove = session.errors.len().saturating_sub(30);
|
||||
if to_remove > 0 {
|
||||
session.errors.drain(0..to_remove);
|
||||
}
|
||||
}
|
||||
Ok(updated)
|
||||
})
|
||||
}
|
||||
|
||||
fn update_state(
|
||||
account_id: u64,
|
||||
updater: impl FnOnce(DownloadState) -> BichonResult<DownloadState> + Send + 'static,
|
||||
) -> BichonResult<()> {
|
||||
if Self::get(account_id)?.is_some() {
|
||||
update_impl(DB_MANAGER.db(), &account_id.to_string(), updater)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn delete(account_id: u64) -> BichonResult<()> {
|
||||
if Self::get(account_id)?.is_none() {
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
delete_impl::<DownloadState>(DB_MANAGER.db(), &account_id.to_string())
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,9 +16,11 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
#[tokio::test]
|
||||
async fn test() {
|
||||
let config = autoconfig::from_addr("test@gmail.com").await.unwrap();
|
||||
println!("{:#?}", config);
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct AccountStats {
|
||||
pub total_size: u64,
|
||||
pub total_count: u64,
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -17,34 +17,34 @@
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use std::collections::{BTreeSet, HashMap};
|
||||
|
||||
use poem_openapi::Object;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::modules::{
|
||||
use crate::{
|
||||
account::{
|
||||
entity::ImapConfig,
|
||||
migration::{AccountModel, AccountType},
|
||||
migration::{AccountModel, AccountType, QuotaWindow},
|
||||
since::{DateSince, RelativeDate},
|
||||
},
|
||||
users::UserModel,
|
||||
};
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct AccountResp {
|
||||
pub id: u64,
|
||||
pub imap: Option<ImapConfig>,
|
||||
pub enabled: bool,
|
||||
pub email: String,
|
||||
pub name: Option<String>,
|
||||
pub account_name: Option<String>,
|
||||
pub login_name: Option<String>,
|
||||
pub capabilities: Option<Vec<String>>,
|
||||
pub date_since: Option<DateSince>,
|
||||
pub date_before: Option<RelativeDate>,
|
||||
pub folder_limit: Option<u32>,
|
||||
pub sync_folders: Option<Vec<String>>,
|
||||
pub download_folders: Option<Vec<String>>,
|
||||
pub account_type: AccountType,
|
||||
pub sync_interval_min: Option<i64>,
|
||||
pub sync_batch_size: Option<u32>,
|
||||
pub download_interval_min: Option<i64>,
|
||||
pub download_batch_size: Option<u32>,
|
||||
pub max_email_size_bytes: Option<u64>,
|
||||
pub known_folders: Option<BTreeSet<String>>,
|
||||
pub created_at: i64,
|
||||
pub updated_at: i64,
|
||||
@@ -54,6 +54,11 @@ pub struct AccountResp {
|
||||
pub use_proxy: Option<u64>,
|
||||
pub use_dangerous: bool,
|
||||
pub pgp_key: Option<String>,
|
||||
pub imap_quota_bytes: Option<u64>,
|
||||
pub imap_quota_window: Option<QuotaWindow>,
|
||||
pub auto_download_new_mailboxes: Option<bool>,
|
||||
pub download_schedule: Option<String>,
|
||||
pub deleting: bool,
|
||||
}
|
||||
|
||||
impl AccountResp {
|
||||
@@ -64,15 +69,16 @@ impl AccountResp {
|
||||
imap: account.imap,
|
||||
enabled: account.enabled,
|
||||
email: account.email,
|
||||
name: account.name,
|
||||
account_name: account.account_name,
|
||||
login_name: account.login_name,
|
||||
capabilities: account.capabilities,
|
||||
date_since: account.date_since,
|
||||
date_before: account.date_before,
|
||||
folder_limit: account.folder_limit,
|
||||
sync_folders: account.sync_folders,
|
||||
download_folders: account.download_folders,
|
||||
account_type: account.account_type,
|
||||
sync_interval_min: account.sync_interval_min,
|
||||
sync_batch_size: account.sync_batch_size,
|
||||
download_interval_min: account.download_interval_min,
|
||||
download_batch_size: account.download_batch_size,
|
||||
max_email_size_bytes: account.max_email_size_bytes,
|
||||
known_folders: account.known_folders,
|
||||
created_at: account.created_at,
|
||||
updated_at: account.updated_at,
|
||||
@@ -86,6 +92,11 @@ impl AccountResp {
|
||||
use_proxy: account.use_proxy,
|
||||
use_dangerous: account.use_dangerous,
|
||||
pgp_key: account.pgp_key,
|
||||
imap_quota_bytes: account.imap_quota_bytes,
|
||||
imap_quota_window: account.imap_quota_window,
|
||||
auto_download_new_mailboxes: account.auto_download_new_mailboxes,
|
||||
download_schedule: account.download_schedule,
|
||||
deleting: account.deleting,
|
||||
}
|
||||
}
|
||||
}
|
||||
68
crates/core/src/admin/meta.rs
Normal file
68
crates/core/src/admin/meta.rs
Normal file
@@ -0,0 +1,68 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use bichon_memdb::{Durability, MemDb};
|
||||
|
||||
use crate::{
|
||||
database::MemDbModel,
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
raise_error,
|
||||
users::{UserModel, DEFAULT_ADMIN_USER_ID},
|
||||
utils::encrypt::internal_encrypt_string,
|
||||
};
|
||||
|
||||
pub fn open_database(path: impl AsRef<Path>) -> BichonResult<MemDb> {
|
||||
MemDb::open_with(path, Durability::Full).map_err(|e| {
|
||||
raise_error!(
|
||||
format!("Failed to open database: {:?}", e),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn find_admin(db: &MemDb) -> BichonResult<Option<UserModel>> {
|
||||
let key = DEFAULT_ADMIN_USER_ID.to_string();
|
||||
let coll = db.collection(UserModel::collection());
|
||||
coll.get(&key)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
|
||||
pub fn update_admin_password(
|
||||
db: &MemDb,
|
||||
password: String,
|
||||
encrypt_key: &str,
|
||||
) -> BichonResult<()> {
|
||||
let key = DEFAULT_ADMIN_USER_ID.to_string();
|
||||
let coll = db.collection(UserModel::collection());
|
||||
let entity: UserModel = coll
|
||||
.get_required(&key)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
let mut updated = entity.clone();
|
||||
updated.password = Some(
|
||||
internal_encrypt_string(encrypt_key, &password)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?,
|
||||
);
|
||||
|
||||
coll.upsert(&key, &updated)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
20
crates/core/src/admin/mod.rs
Normal file
20
crates/core/src/admin/mod.rs
Normal file
@@ -0,0 +1,20 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
pub mod meta;
|
||||
339
crates/core/src/autoconfig/client.rs
Normal file
339
crates/core/src/autoconfig/client.rs
Normal file
@@ -0,0 +1,339 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use hickory_resolver::name_server::TokioConnectionProvider;
|
||||
use hickory_resolver::proto::rr::RData;
|
||||
use hickory_resolver::proto::rr::RecordType;
|
||||
use hickory_resolver::TokioResolver;
|
||||
use quick_xml::de::from_str;
|
||||
use reqwest::Client;
|
||||
use serde::Deserialize;
|
||||
|
||||
use crate::error::code::ErrorCode;
|
||||
use crate::error::BichonResult;
|
||||
use crate::raise_error;
|
||||
|
||||
/// Parsed result from Thunderbird-style autoconfig XML or DNS SRV fallback.
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||
pub struct MailConfig {
|
||||
pub incoming: Vec<IncomingServer>,
|
||||
pub outgoing: Vec<OutgoingServer>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq, Deserialize)]
|
||||
pub struct IncomingServer {
|
||||
#[serde(rename = "@type")]
|
||||
pub protocol: String,
|
||||
pub hostname: String,
|
||||
#[serde(default)]
|
||||
pub port: u16,
|
||||
#[serde(rename = "socketType")]
|
||||
pub socket_type: String,
|
||||
pub username: String,
|
||||
/// Authentication method from the XML, e.g. "OAuth2", "password-cleartext",
|
||||
/// "password-encrypted", "GSSAPI", "NTLM". Absent in DNS SRV fallback.
|
||||
#[serde(default)]
|
||||
pub authentication: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq, Deserialize)]
|
||||
pub struct OutgoingServer {
|
||||
#[serde(rename = "@type")]
|
||||
pub protocol: String,
|
||||
pub hostname: String,
|
||||
#[serde(default)]
|
||||
pub port: u16,
|
||||
#[serde(rename = "socketType")]
|
||||
pub socket_type: String,
|
||||
pub username: String,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Internal XML wrapper structs matching the Thunderbird config-v1.1 schema:
|
||||
// <clientConfig> → <emailProvider> → <incomingServer> / <outgoingServer>
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
#[serde(rename = "clientConfig")]
|
||||
struct ClientConfig {
|
||||
#[serde(rename = "emailProvider", default)]
|
||||
email_providers: Vec<EmailProvider>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
struct EmailProvider {
|
||||
#[serde(rename = "incomingServer", default)]
|
||||
incoming_servers: Vec<IncomingServer>,
|
||||
#[serde(rename = "outgoingServer", default)]
|
||||
outgoing_servers: Vec<OutgoingServer>,
|
||||
}
|
||||
|
||||
/// Parse Thunderbird autoconfig XML into a `MailConfig`.
|
||||
/// Exposed for unit testing.
|
||||
pub(crate) fn parse_autoconfig_xml(xml: &str) -> Option<MailConfig> {
|
||||
let client_config: ClientConfig = from_str(xml).ok()?;
|
||||
let provider = client_config.email_providers.into_iter().next()?;
|
||||
Some(MailConfig {
|
||||
incoming: provider.incoming_servers,
|
||||
outgoing: provider.outgoing_servers,
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Network helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
async fn fetch_xml(client: &Client, url: &str) -> Option<MailConfig> {
|
||||
let resp = client.get(url).send().await.ok()?;
|
||||
if !resp.status().is_success() {
|
||||
return None;
|
||||
}
|
||||
let text = resp.text().await.ok()?;
|
||||
parse_autoconfig_xml(&text)
|
||||
}
|
||||
|
||||
async fn lookup_srv(domain: &str) -> Option<MailConfig> {
|
||||
let resolver = TokioResolver::builder(TokioConnectionProvider::default())
|
||||
.ok()?
|
||||
.build();
|
||||
|
||||
let imap_srv = format!("_imaps._tcp.{}.", domain);
|
||||
let imap_lookup = resolver.lookup(imap_srv, RecordType::SRV).await.ok()?;
|
||||
let imap_record = imap_lookup.iter().next()?;
|
||||
let (imap_host, imap_port) = match imap_record {
|
||||
RData::SRV(srv) => {
|
||||
let host = srv.target().to_string().trim_end_matches('.').to_string();
|
||||
(host, srv.port())
|
||||
}
|
||||
_ => return None,
|
||||
};
|
||||
|
||||
let smtp_srv = format!("_submission._tcp.{}.", domain);
|
||||
let smtp_lookup = resolver.lookup(smtp_srv, RecordType::SRV).await.ok()?;
|
||||
let smtp_record = smtp_lookup.iter().next()?;
|
||||
let (smtp_host, smtp_port) = match smtp_record {
|
||||
RData::SRV(srv) => {
|
||||
let host = srv.target().to_string().trim_end_matches('.').to_string();
|
||||
(host, srv.port())
|
||||
}
|
||||
_ => return None,
|
||||
};
|
||||
|
||||
Some(MailConfig {
|
||||
incoming: vec![IncomingServer {
|
||||
protocol: "imap".to_string(),
|
||||
hostname: imap_host,
|
||||
port: imap_port,
|
||||
socket_type: "SSL".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: String::new(),
|
||||
}],
|
||||
outgoing: vec![OutgoingServer {
|
||||
protocol: "smtp".to_string(),
|
||||
hostname: smtp_host,
|
||||
port: smtp_port,
|
||||
socket_type: "STARTTLS".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
}],
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Discover mail server configuration for a domain using the Thunderbird
|
||||
/// autoconfig protocol (ISPDB), DNS SRV, MX fallback, and finally guessing.
|
||||
///
|
||||
/// Probe order:
|
||||
/// 1. `https://autoconfig.{domain}/mail/config-v1.1.xml`
|
||||
/// 2. `http://autoconfig.{domain}/mail/config-v1.1.xml`
|
||||
/// 3. `https://{domain}/.well-known/autoconfig/mail/config-v1.1.xml`
|
||||
/// 4. `http://{domain}/.well-known/autoconfig/mail/config-v1.1.xml`
|
||||
/// 5. DNS SRV records (`_imaps._tcp` / `_submission._tcp`)
|
||||
/// 6. Thunderbird central ISPDB (`https://autoconfig.thunderbird.net/v1.1/{domain}`)
|
||||
/// 7. MX lookup → ISPDB for MX domain
|
||||
/// 8. MX lookup → ISP autoconfig for MX domain
|
||||
/// 9. GuessConfig — probe common hostnames + ports
|
||||
pub async fn fetch(domain: &str) -> BichonResult<MailConfig> {
|
||||
let client = Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(10))
|
||||
.build()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
// ── ISP autoconfig (HTTPS, then HTTP) ──────────────────────────
|
||||
if let Some(config) =
|
||||
fetch_xml(&client, &format!("https://autoconfig.{domain}/mail/config-v1.1.xml")).await
|
||||
{
|
||||
return Ok(config);
|
||||
}
|
||||
if let Some(config) =
|
||||
fetch_xml(&client, &format!("http://autoconfig.{domain}/mail/config-v1.1.xml")).await
|
||||
{
|
||||
return Ok(config);
|
||||
}
|
||||
|
||||
// ── Well-known path (HTTPS, then HTTP) ─────────────────────────
|
||||
if let Some(config) = fetch_xml(
|
||||
&client,
|
||||
&format!("https://{domain}/.well-known/autoconfig/mail/config-v1.1.xml"),
|
||||
)
|
||||
.await
|
||||
{
|
||||
return Ok(config);
|
||||
}
|
||||
if let Some(config) = fetch_xml(
|
||||
&client,
|
||||
&format!("http://{domain}/.well-known/autoconfig/mail/config-v1.1.xml"),
|
||||
)
|
||||
.await
|
||||
{
|
||||
return Ok(config);
|
||||
}
|
||||
|
||||
// ── DNS SRV records ────────────────────────────────────────────
|
||||
if let Some(config) = lookup_srv(domain).await {
|
||||
return Ok(config);
|
||||
}
|
||||
|
||||
// ── Thunderbird central ISPDB ──────────────────────────────────
|
||||
if let Some(config) =
|
||||
fetch_xml(&client, &format!("https://autoconfig.thunderbird.net/v1.1/{domain}")).await
|
||||
{
|
||||
return Ok(config);
|
||||
}
|
||||
|
||||
// ── MX fallback ────────────────────────────────────────────────
|
||||
if let Some(config) = fetch_for_mx(&client, domain).await {
|
||||
return Ok(config);
|
||||
}
|
||||
|
||||
// ── GuessConfig ────────────────────────────────────────────────
|
||||
if let Some(config) = crate::autoconfig::guess::guess_config(domain).await {
|
||||
return Ok(config);
|
||||
}
|
||||
|
||||
Err(raise_error!(
|
||||
format!("No autoconfig found for domain: {domain}"),
|
||||
ErrorCode::InternalError
|
||||
))
|
||||
}
|
||||
|
||||
/// DNS MX lookup → retry ISPDB and ISP autoconfig for the MX domain.
|
||||
///
|
||||
/// Many self-hosted domains have their MX pointed at Google, Microsoft, etc.
|
||||
/// The MX domain's ISPDB entry covers the original domain.
|
||||
async fn fetch_for_mx(client: &Client, domain: &str) -> Option<MailConfig> {
|
||||
let mx_domain = lookup_mx_domain(domain).await?;
|
||||
if mx_domain == domain.to_ascii_lowercase() {
|
||||
return None; // same domain, already tried above
|
||||
}
|
||||
|
||||
// Try ISPDB for the MX domain
|
||||
if let Some(config) =
|
||||
fetch_xml(client, &format!("https://autoconfig.thunderbird.net/v1.1/{mx_domain}")).await
|
||||
{
|
||||
return Some(config);
|
||||
}
|
||||
|
||||
// Try ISP autoconfig for the MX domain (HTTPS then HTTP)
|
||||
if let Some(config) =
|
||||
fetch_xml(client, &format!("https://autoconfig.{mx_domain}/mail/config-v1.1.xml")).await
|
||||
{
|
||||
return Some(config);
|
||||
}
|
||||
if let Some(config) =
|
||||
fetch_xml(client, &format!("http://autoconfig.{mx_domain}/mail/config-v1.1.xml")).await
|
||||
{
|
||||
return Some(config);
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
/// DNS MX lookup → extract the second-level domain of the first MX hostname.
|
||||
async fn lookup_mx_domain(domain: &str) -> Option<String> {
|
||||
let resolver = TokioResolver::builder(TokioConnectionProvider::default())
|
||||
.ok()?
|
||||
.build();
|
||||
let lookup = resolver.mx_lookup(domain).await.ok()?;
|
||||
let record = lookup.iter().next()?;
|
||||
let mx_host = record.to_string().trim_end_matches('.').to_string();
|
||||
|
||||
// Extract a reasonable base domain from the MX hostname.
|
||||
// E.g., "aspmx.l.google.com" → "google.com"
|
||||
// "company.mail.protection.outlook.com" → "outlook.com"
|
||||
extract_base_domain(&mx_host)
|
||||
}
|
||||
|
||||
/// Extract the top two labels from a hostname as a rough base domain.
|
||||
fn extract_base_domain(host: &str) -> Option<String> {
|
||||
let parts: Vec<&str> = host.split('.').collect();
|
||||
if parts.len() >= 2 {
|
||||
Some(parts[parts.len() - 2..].join("."))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_fetch_valid_domain() {
|
||||
let domains = vec![
|
||||
// North America
|
||||
("gmail.com", "Google Gmail"),
|
||||
("outlook.com", "Microsoft Outlook"),
|
||||
("hotmail.com", "Microsoft Hotmail"),
|
||||
("yahoo.com", "Yahoo Mail"),
|
||||
("icloud.com", "Apple iCloud"),
|
||||
("aol.com", "AOL Mail"),
|
||||
("protonmail.com", "ProtonMail"),
|
||||
("zoho.com", "Zoho Mail"),
|
||||
("fastmail.com", "FastMail"),
|
||||
// Europe
|
||||
("gmx.de", "GMX Germany"),
|
||||
("gmx.net", "GMX International"),
|
||||
("web.de", "Web.de Germany"),
|
||||
("freenet.de", "Freenet Germany"),
|
||||
("mail.ru", "Mail.ru Russia"),
|
||||
("yandex.ru", "Yandex Russia"),
|
||||
("orange.fr", "Orange France"),
|
||||
("laposte.net", "La Poste France"),
|
||||
("libero.it", "Libero Italy"),
|
||||
("tiscali.it", "Tiscali Italy"),
|
||||
("telenet.be", "Telenet Belgium"),
|
||||
// Asia Pacific
|
||||
("qq.com", "Tencent QQ"),
|
||||
("163.com", "NetEase 163"),
|
||||
("126.com", "NetEase 126"),
|
||||
("sina.com", "Sina Mail"),
|
||||
("naver.com", "Naver Korea"),
|
||||
];
|
||||
|
||||
for (domain, label) in &domains {
|
||||
let result = fetch(domain).await;
|
||||
match result {
|
||||
Ok(config) => println!("✅ [{label}] {domain}: {config:#?}"),
|
||||
Err(e) => println!("⚠️ [{label}] {domain}: {e:?}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,14 +16,12 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use autoconfig::config::OAuth2Config as XOAuth2Config;
|
||||
use poem_openapi::Object;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::modules::account::entity::Encryption;
|
||||
use crate::account::entity::Encryption;
|
||||
|
||||
#[derive(Debug, Clone, Default, Eq, PartialEq, Serialize, Deserialize, Object)]
|
||||
#[derive(Debug, Clone, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct ServerConfig {
|
||||
/// server hostname or IP address
|
||||
pub host: String,
|
||||
@@ -43,7 +41,8 @@ impl ServerConfig {
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, Eq, PartialEq, Serialize, Deserialize, Object)]
|
||||
#[derive(Debug, Clone, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct OAuth2Config {
|
||||
/// The authorization server's issuer identifier URL
|
||||
pub issuer: String,
|
||||
@@ -54,19 +53,8 @@ pub struct OAuth2Config {
|
||||
/// URL of the authorization server's token endpoint
|
||||
pub token_url: String,
|
||||
}
|
||||
|
||||
impl From<&XOAuth2Config> for OAuth2Config {
|
||||
fn from(value: &XOAuth2Config) -> Self {
|
||||
Self {
|
||||
issuer: value.issuer().into(),
|
||||
scope: value.scope().into_iter().map(Into::into).collect(),
|
||||
auth_url: value.auth_url().into(),
|
||||
token_url: value.token_url().into(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default, Eq, PartialEq, Serialize, Deserialize, Object)]
|
||||
#[derive(Debug, Clone, Default, Eq, PartialEq, Serialize, Deserialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct MailServerConfig {
|
||||
/// IMAP server configuration
|
||||
pub imap: ServerConfig,
|
||||
105
crates/core/src/autoconfig/guess.rs
Normal file
105
crates/core/src/autoconfig/guess.rs
Normal file
@@ -0,0 +1,105 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::account::entity::Encryption;
|
||||
use crate::autoconfig::client::{IncomingServer, MailConfig};
|
||||
use crate::imap::client::Client;
|
||||
use tracing::{debug, info};
|
||||
|
||||
/// A single host:port:encryption combination to probe.
|
||||
struct Guess {
|
||||
hostname: String,
|
||||
port: u16,
|
||||
encryption: Encryption,
|
||||
socket_type: &'static str,
|
||||
}
|
||||
|
||||
/// Generate candidates in the same order Thunderbird uses:
|
||||
/// 1. imap.{domain} — most common
|
||||
/// 2. mail.{domain} — fallback
|
||||
/// 3. {domain} — bare domain (rare)
|
||||
fn make_guesses(domain: &str) -> Vec<Guess> {
|
||||
let hosts = [
|
||||
format!("imap.{domain}"),
|
||||
format!("mail.{domain}"),
|
||||
domain.to_string(),
|
||||
];
|
||||
|
||||
let mut guesses = Vec::with_capacity(hosts.len() * 2);
|
||||
for host in &hosts {
|
||||
guesses.push(Guess {
|
||||
hostname: host.clone(),
|
||||
port: 993,
|
||||
encryption: Encryption::Ssl,
|
||||
socket_type: "SSL",
|
||||
});
|
||||
guesses.push(Guess {
|
||||
hostname: host.clone(),
|
||||
port: 143,
|
||||
encryption: Encryption::StartTls,
|
||||
socket_type: "STARTTLS",
|
||||
});
|
||||
}
|
||||
guesses
|
||||
}
|
||||
|
||||
/// Try to open a connection, read the IMAP banner, and close.
|
||||
/// Returns `true` if the server responds with an IMAP greeting.
|
||||
async fn probe(hostname: &str, port: u16, encryption: &Encryption) -> bool {
|
||||
match Client::connection(hostname, encryption, port, None, true).await {
|
||||
Ok(_) => {
|
||||
debug!("GuessConfig probe succeeded: {hostname}:{port} ({encryption:?})");
|
||||
true
|
||||
}
|
||||
Err(e) => {
|
||||
debug!("GuessConfig probe failed for {hostname}:{port}: {e:?}");
|
||||
false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Thunderbird-style guessing: try common hostnames and ports, probing
|
||||
/// each with a real TCP connection.
|
||||
///
|
||||
/// Returns the first working `MailConfig`, or `None` if nothing works.
|
||||
pub async fn guess_config(domain: &str) -> Option<MailConfig> {
|
||||
let guesses = make_guesses(domain);
|
||||
info!("GuessConfig: trying {} candidates for {domain}", guesses.len());
|
||||
|
||||
for g in &guesses {
|
||||
if probe(&g.hostname, g.port, &g.encryption).await {
|
||||
info!(
|
||||
"GuessConfig: found working IMAP at {}:{} ({})",
|
||||
g.hostname, g.port, g.socket_type
|
||||
);
|
||||
return Some(MailConfig {
|
||||
incoming: vec![IncomingServer {
|
||||
protocol: "imap".to_string(),
|
||||
hostname: g.hostname.clone(),
|
||||
port: g.port,
|
||||
socket_type: g.socket_type.to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: String::new(),
|
||||
}],
|
||||
outgoing: vec![],
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
116
crates/core/src/autoconfig/load.rs
Normal file
116
crates/core/src/autoconfig/load.rs
Normal file
@@ -0,0 +1,116 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::account::entity::Encryption;
|
||||
use crate::autoconfig::client::{self, MailConfig};
|
||||
use crate::autoconfig::entity::{MailServerConfig, ServerConfig};
|
||||
use crate::autoconfig::oauth2_providers::lookup_oauth2;
|
||||
use crate::autoconfig::CachedMailSettings;
|
||||
use crate::error::code::ErrorCode;
|
||||
use crate::error::BichonResult;
|
||||
use crate::raise_error;
|
||||
use email_address::EmailAddress;
|
||||
use std::str::FromStr;
|
||||
use tracing::error;
|
||||
|
||||
/// Map an autoconfig XML `socketType` value to our `Encryption` enum.
|
||||
pub(crate) fn socket_type_to_encryption(raw: &str) -> Encryption {
|
||||
match raw.to_ascii_uppercase().as_str() {
|
||||
"SSL" | "TLS" => Encryption::Ssl,
|
||||
"STARTTLS" => Encryption::StartTls,
|
||||
_ => Encryption::None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert the raw `MailConfig` discovered by `client::fetch` into a
|
||||
/// `MailServerConfig` suitable for account provisioning.
|
||||
pub(crate) fn mail_config_to_server_config(config: &MailConfig) -> Option<MailServerConfig> {
|
||||
let imap = config.incoming.iter().find(|s| {
|
||||
let p = s.protocol.to_ascii_lowercase();
|
||||
p == "imap" || p == "imaps"
|
||||
})?;
|
||||
|
||||
let encryption = socket_type_to_encryption(&imap.socket_type);
|
||||
let port = if imap.port != 0 {
|
||||
imap.port
|
||||
} else {
|
||||
match encryption {
|
||||
Encryption::Ssl => 993,
|
||||
_ => 143,
|
||||
}
|
||||
};
|
||||
|
||||
// Detect OAuth2 support: the XML <authentication> field and a known
|
||||
// hostname → issuer mapping determine whether the provider supports OAuth2.
|
||||
let oauth2 = if imap.authentication.eq_ignore_ascii_case("OAuth2") {
|
||||
lookup_oauth2(&imap.hostname)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
Some(MailServerConfig {
|
||||
imap: ServerConfig::new(imap.hostname.clone(), port, encryption),
|
||||
oauth2,
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn resolve_autoconfig(email: impl AsRef<str>) -> BichonResult<Option<MailServerConfig>> {
|
||||
let email = email.as_ref();
|
||||
let email_address = EmailAddress::from_str(email).map_err(|error| {
|
||||
raise_error!(
|
||||
format!("Invalid email address: {email:#?}. {error:#?}"),
|
||||
ErrorCode::InvalidParameter
|
||||
)
|
||||
})?;
|
||||
|
||||
let domain = email_address.domain();
|
||||
// Try local cache first
|
||||
if let Some(cached_entity) = CachedMailSettings::get(domain)? {
|
||||
return Ok(Some(cached_entity.config));
|
||||
}
|
||||
|
||||
let config = client::fetch(domain).await.map_err(|e| {
|
||||
error!(
|
||||
email = %email,
|
||||
domain = %domain,
|
||||
error = ?e,
|
||||
"Autoconfig fetch failed"
|
||||
);
|
||||
raise_error!(
|
||||
format!(
|
||||
"Failed to fetch autoconfig for email '{}': {:#?}",
|
||||
email_address.email(),
|
||||
e
|
||||
),
|
||||
ErrorCode::AutoconfigFetchFailed
|
||||
)
|
||||
})?;
|
||||
|
||||
let result = mail_config_to_server_config(&config).ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!(
|
||||
"No IMAP server found in autoconfig for email: {}",
|
||||
email_address.email()
|
||||
),
|
||||
ErrorCode::ResourceNotFound
|
||||
)
|
||||
})?;
|
||||
|
||||
CachedMailSettings::add(domain.into(), result.clone())?;
|
||||
Ok(Some(result))
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,65 +16,56 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use crate::modules::database::manager::DB_MANAGER;
|
||||
use crate::modules::database::{delete_impl, async_find_impl, upsert_impl};
|
||||
use crate::modules::error::code::ErrorCode;
|
||||
use crate::raise_error;
|
||||
use crate::{
|
||||
modules::autoconfig::entity::MailServerConfig, modules::error::BichonResult, utc_now,
|
||||
};
|
||||
use native_db::*;
|
||||
use native_model::{native_model, Model};
|
||||
use crate::database::manager::DB_MANAGER;
|
||||
use crate::database::{delete_impl, upsert_impl};
|
||||
use crate::database::{find_impl, MemDbModel};
|
||||
use crate::{autoconfig::entity::MailServerConfig, error::BichonResult, utc_now};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
pub mod client;
|
||||
pub mod entity;
|
||||
pub mod guess;
|
||||
pub mod load;
|
||||
mod oauth2_providers;
|
||||
#[cfg(test)]
|
||||
mod tests;
|
||||
|
||||
const EXPIRE_TIME_MS: i64 = 30 * 24 * 60 * 60 * 1000;
|
||||
|
||||
#[derive(Clone, Debug, Default, PartialEq, Eq, Deserialize, Serialize)]
|
||||
#[native_model(id = 3, version = 1)]
|
||||
#[native_db]
|
||||
pub struct CachedMailSettings {
|
||||
#[primary_key]
|
||||
pub domain: String,
|
||||
pub config: MailServerConfig,
|
||||
pub created_at: i64,
|
||||
}
|
||||
|
||||
impl MemDbModel for CachedMailSettings {
|
||||
fn collection() -> &'static str {
|
||||
"autoconfig"
|
||||
}
|
||||
fn key(&self) -> String {
|
||||
self.domain.clone()
|
||||
}
|
||||
}
|
||||
|
||||
impl CachedMailSettings {
|
||||
pub async fn add(domain: String, config: MailServerConfig) -> BichonResult<()> {
|
||||
pub fn add(domain: String, config: MailServerConfig) -> BichonResult<()> {
|
||||
Self {
|
||||
domain,
|
||||
config,
|
||||
created_at: utc_now!(),
|
||||
}
|
||||
.save()
|
||||
.await
|
||||
}
|
||||
|
||||
async fn save(&self) -> BichonResult<()> {
|
||||
upsert_impl(DB_MANAGER.meta_db(), self.to_owned()).await
|
||||
fn save(&self) -> BichonResult<()> {
|
||||
upsert_impl(DB_MANAGER.db(), self.to_owned())
|
||||
}
|
||||
|
||||
pub async fn get(domain: &str) -> BichonResult<Option<CachedMailSettings>> {
|
||||
if let Some(found) =
|
||||
async_find_impl::<CachedMailSettings>(DB_MANAGER.meta_db(), domain.to_string()).await?
|
||||
{
|
||||
pub fn get(domain: &str) -> BichonResult<Option<CachedMailSettings>> {
|
||||
if let Some(found) = find_impl::<CachedMailSettings>(DB_MANAGER.db(), domain)? {
|
||||
if (utc_now!() - found.created_at) > EXPIRE_TIME_MS {
|
||||
let domain = domain.to_string();
|
||||
delete_impl(DB_MANAGER.meta_db(), |rw| {
|
||||
rw.get()
|
||||
.primary::<CachedMailSettings>(domain)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
.ok_or_else(|| {
|
||||
raise_error!("auto config cache miss".into(), ErrorCode::InternalError)
|
||||
})
|
||||
})
|
||||
.await?;
|
||||
delete_impl::<CachedMailSettings>(DB_MANAGER.db(), domain)?;
|
||||
Ok(None)
|
||||
} else {
|
||||
Ok(Some(found))
|
||||
151
crates/core/src/autoconfig/oauth2_providers.rs
Normal file
151
crates/core/src/autoconfig/oauth2_providers.rs
Normal file
@@ -0,0 +1,151 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::autoconfig::entity::OAuth2Config;
|
||||
|
||||
/// Per-provider OAuth2 metadata, mirroring Thunderbird's `OAuth2Providers.sys.mjs`.
|
||||
///
|
||||
/// Each entry maps one or more IMAP hostname suffixes to a well-known OIDC issuer
|
||||
/// and the IMAP-specific OAuth2 scopes.
|
||||
struct Provider {
|
||||
/// Suffixes matched case-insensitively against the end of the IMAP hostname.
|
||||
host_suffixes: &'static [&'static str],
|
||||
/// The OIDC issuer URL used by the provider.
|
||||
issuer: &'static str,
|
||||
/// OAuth2 scope(s) required for IMAP access.
|
||||
scopes: &'static [&'static str],
|
||||
}
|
||||
|
||||
const PROVIDERS: &[Provider] = &[
|
||||
// Google
|
||||
Provider {
|
||||
host_suffixes: &["imap.gmail.com", ".gmail.com", ".googlemail.com"],
|
||||
issuer: "https://accounts.google.com",
|
||||
scopes: &["https://mail.google.com/"],
|
||||
},
|
||||
// Microsoft (Outlook / Office 365 / Hotmail / Live)
|
||||
Provider {
|
||||
host_suffixes: &[
|
||||
"outlook.office365.com",
|
||||
".outlook.com",
|
||||
".hotmail.com",
|
||||
".live.com",
|
||||
".office365.com",
|
||||
],
|
||||
issuer: "https://login.microsoftonline.com/common/v2.0",
|
||||
scopes: &[
|
||||
"https://outlook.office365.com/IMAP.AccessAsUser.All",
|
||||
"offline_access",
|
||||
],
|
||||
},
|
||||
// Yahoo / AOL / ATT / Verizon
|
||||
Provider {
|
||||
host_suffixes: &[
|
||||
"imap.mail.yahoo.com",
|
||||
".yahoo.com",
|
||||
".yahoodns.net",
|
||||
".aol.com",
|
||||
"imap.aol.com",
|
||||
],
|
||||
issuer: "https://login.yahoo.com",
|
||||
scopes: &["mail-w"],
|
||||
},
|
||||
// Yandex
|
||||
Provider {
|
||||
host_suffixes: &["imap.yandex.ru", "imap.yandex.com", ".yandex.ru"],
|
||||
issuer: "https://oauth.yandex.com",
|
||||
scopes: &["imap:all"],
|
||||
},
|
||||
// Mail.ru
|
||||
Provider {
|
||||
host_suffixes: &["imap.mail.ru", ".mail.ru", ".bk.ru", ".list.ru", ".inbox.ru"],
|
||||
issuer: "https://o2.mail.ru",
|
||||
scopes: &["imap"],
|
||||
},
|
||||
// Fastmail
|
||||
Provider {
|
||||
host_suffixes: &["imap.fastmail.com", ".fastmail.com"],
|
||||
issuer: "https://www.fastmail.com",
|
||||
scopes: &[
|
||||
"https://www.fastmail.com/dev/imap",
|
||||
"offline_access",
|
||||
],
|
||||
},
|
||||
// Comcast
|
||||
Provider {
|
||||
host_suffixes: &["imap.comcast.net", ".comcast.net"],
|
||||
issuer: "https://oauth.xfinity.com",
|
||||
scopes: &["https://email.comcast.net/"],
|
||||
},
|
||||
];
|
||||
|
||||
/// Try to find an OAuth2 provider that matches the given IMAP hostname.
|
||||
///
|
||||
/// Matching is case-insensitive and done by suffix: a hostname "imap.gmail.com"
|
||||
/// matches the suffix ".gmail.com".
|
||||
pub fn lookup_oauth2(hostname: &str) -> Option<OAuth2Config> {
|
||||
let host = hostname.to_ascii_lowercase();
|
||||
for provider in PROVIDERS {
|
||||
if provider
|
||||
.host_suffixes
|
||||
.iter()
|
||||
.any(|suffix| host.ends_with(&suffix.to_ascii_lowercase()))
|
||||
{
|
||||
return Some(OAuth2Config {
|
||||
issuer: provider.issuer.to_string(),
|
||||
scope: provider.scopes.iter().map(|s| s.to_string()).collect(),
|
||||
auth_url: String::new(),
|
||||
token_url: String::new(),
|
||||
});
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_known_providers() {
|
||||
let cases = [
|
||||
("imap.gmail.com", Some("https://accounts.google.com")),
|
||||
("imap.gmail.com", Some("https://accounts.google.com")),
|
||||
("outlook.office365.com", Some("https://login.microsoftonline.com/common/v2.0")),
|
||||
("imap.mail.yahoo.com", Some("https://login.yahoo.com")),
|
||||
("imap.aol.com", Some("https://login.yahoo.com")),
|
||||
("imap.yandex.ru", Some("https://oauth.yandex.com")),
|
||||
("imap.mail.ru", Some("https://o2.mail.ru")),
|
||||
("imap.fastmail.com", Some("https://www.fastmail.com")),
|
||||
("imap.comcast.net", Some("https://oauth.xfinity.com")),
|
||||
];
|
||||
for (hostname, expected_issuer) in &cases {
|
||||
let result = lookup_oauth2(hostname);
|
||||
assert_eq!(
|
||||
result.map(|c| c.issuer),
|
||||
expected_issuer.map(|s| s.to_string()),
|
||||
"failed for hostname: {hostname}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_unknown_provider() {
|
||||
assert!(lookup_oauth2("mail.my-company.example").is_none());
|
||||
}
|
||||
}
|
||||
368
crates/core/src/autoconfig/tests.rs
Normal file
368
crates/core/src/autoconfig/tests.rs
Normal file
@@ -0,0 +1,368 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::account::entity::Encryption;
|
||||
use crate::autoconfig::client::{self, IncomingServer, MailConfig};
|
||||
use crate::autoconfig::load::{mail_config_to_server_config, socket_type_to_encryption};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// XML parsing tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn make_valid_xml() -> String {
|
||||
r#"<?xml version="1.0" encoding="UTF-8"?>
|
||||
<clientConfig version="1.1">
|
||||
<emailProvider id="example.com">
|
||||
<domain>example.com</domain>
|
||||
<displayName>Example Mail</displayName>
|
||||
<incomingServer type="imap">
|
||||
<hostname>imap.example.com</hostname>
|
||||
<port>993</port>
|
||||
<socketType>SSL</socketType>
|
||||
<username>%EMAILADDRESS%</username>
|
||||
</incomingServer>
|
||||
<outgoingServer type="smtp">
|
||||
<hostname>smtp.example.com</hostname>
|
||||
<port>587</port>
|
||||
<socketType>STARTTLS</socketType>
|
||||
<username>%EMAILADDRESS%</username>
|
||||
</outgoingServer>
|
||||
</emailProvider>
|
||||
</clientConfig>"#
|
||||
.to_string()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_valid_xml() {
|
||||
let xml = make_valid_xml();
|
||||
let config = client::parse_autoconfig_xml(&xml).expect("should parse valid XML");
|
||||
|
||||
assert_eq!(config.incoming.len(), 1);
|
||||
let imap = &config.incoming[0];
|
||||
assert_eq!(imap.protocol, "imap");
|
||||
assert_eq!(imap.hostname, "imap.example.com");
|
||||
assert_eq!(imap.port, 993);
|
||||
assert_eq!(imap.socket_type, "SSL");
|
||||
assert_eq!(imap.username, "%EMAILADDRESS%");
|
||||
|
||||
assert_eq!(config.outgoing.len(), 1);
|
||||
let smtp = &config.outgoing[0];
|
||||
assert_eq!(smtp.protocol, "smtp");
|
||||
assert_eq!(smtp.hostname, "smtp.example.com");
|
||||
assert_eq!(smtp.port, 587);
|
||||
assert_eq!(smtp.socket_type, "STARTTLS");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_xml_empty_body() {
|
||||
let xml = r#"<?xml version="1.0"?><clientConfig></clientConfig>"#;
|
||||
let config = client::parse_autoconfig_xml(xml);
|
||||
assert!(config.is_none(), "no emailProvider → None");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_xml_no_incoming_servers() {
|
||||
let xml = r#"<?xml version="1.0"?>
|
||||
<clientConfig version="1.1">
|
||||
<emailProvider id="example.com">
|
||||
<domain>example.com</domain>
|
||||
</emailProvider>
|
||||
</clientConfig>"#;
|
||||
let config = client::parse_autoconfig_xml(xml).expect("should parse");
|
||||
assert!(config.incoming.is_empty());
|
||||
assert!(config.outgoing.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_xml_garbage() {
|
||||
let config = client::parse_autoconfig_xml("not xml at all");
|
||||
assert!(config.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_xml_missing_port_defaults_to_zero() {
|
||||
let xml = r#"<?xml version="1.0"?>
|
||||
<clientConfig version="1.1">
|
||||
<emailProvider id="example.com">
|
||||
<incomingServer type="imap">
|
||||
<hostname>imap.example.com</hostname>
|
||||
<socketType>SSL</socketType>
|
||||
<username>%EMAILADDRESS%</username>
|
||||
</incomingServer>
|
||||
</emailProvider>
|
||||
</clientConfig>"#;
|
||||
let config = client::parse_autoconfig_xml(xml).expect("should parse");
|
||||
assert_eq!(config.incoming[0].port, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_xml_multiple_providers_picks_first() {
|
||||
let xml = r#"<?xml version="1.0"?>
|
||||
<clientConfig version="1.1">
|
||||
<emailProvider id="first.example.com">
|
||||
<incomingServer type="imap">
|
||||
<hostname>imap.first.example.com</hostname>
|
||||
<port>993</port>
|
||||
<socketType>SSL</socketType>
|
||||
<username>%EMAILADDRESS%</username>
|
||||
</incomingServer>
|
||||
</emailProvider>
|
||||
<emailProvider id="second.example.com">
|
||||
<incomingServer type="imap">
|
||||
<hostname>imap.second.example.com</hostname>
|
||||
<port>143</port>
|
||||
<socketType>STARTTLS</socketType>
|
||||
<username>%EMAILADDRESS%</username>
|
||||
</incomingServer>
|
||||
</emailProvider>
|
||||
</clientConfig>"#;
|
||||
let config = client::parse_autoconfig_xml(xml).expect("should parse");
|
||||
assert_eq!(config.incoming[0].hostname, "imap.first.example.com");
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// socket_type → Encryption mapping tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn encryption_ssl_uppercase() {
|
||||
assert_eq!(socket_type_to_encryption("SSL"), Encryption::Ssl);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encryption_ssl_lowercase() {
|
||||
assert_eq!(socket_type_to_encryption("ssl"), Encryption::Ssl);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encryption_tls() {
|
||||
assert_eq!(socket_type_to_encryption("TLS"), Encryption::Ssl);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encryption_starttls() {
|
||||
assert_eq!(socket_type_to_encryption("STARTTLS"), Encryption::StartTls);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encryption_starttls_lowercase() {
|
||||
assert_eq!(socket_type_to_encryption("starttls"), Encryption::StartTls);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encryption_starttls_mixed_case() {
|
||||
assert_eq!(socket_type_to_encryption("StartTls"), Encryption::StartTls);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encryption_plain() {
|
||||
assert_eq!(socket_type_to_encryption("plain"), Encryption::None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encryption_empty_string() {
|
||||
assert_eq!(socket_type_to_encryption(""), Encryption::None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encryption_unknown_value() {
|
||||
assert_eq!(socket_type_to_encryption("WPA2-ENTERPRISE"), Encryption::None);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// MailConfig → MailServerConfig conversion tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn make_imap_server(host: &str, port: u16, socket_type: &str) -> IncomingServer {
|
||||
IncomingServer {
|
||||
protocol: "imap".to_string(),
|
||||
hostname: host.to_string(),
|
||||
port,
|
||||
socket_type: socket_type.to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: String::new(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_basic_imap_ssl() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![make_imap_server("imap.example.com", 993, "SSL")],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should convert");
|
||||
assert_eq!(result.imap.host, "imap.example.com");
|
||||
assert_eq!(result.imap.port, 993);
|
||||
assert_eq!(result.imap.encryption, Encryption::Ssl);
|
||||
assert!(result.oauth2.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_imap_starttls_with_default_port() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![make_imap_server("imap.example.com", 0, "STARTTLS")],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should convert");
|
||||
assert_eq!(result.imap.port, 143, "default port for STARTTLS → 143");
|
||||
assert_eq!(result.imap.encryption, Encryption::StartTls);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_imap_ssl_with_default_port() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![make_imap_server("imap.example.com", 0, "SSL")],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should convert");
|
||||
assert_eq!(result.imap.port, 993, "default port for SSL → 993");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_no_imap_only_pop3() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![IncomingServer {
|
||||
protocol: "pop3".to_string(),
|
||||
hostname: "pop.example.com".to_string(),
|
||||
port: 995,
|
||||
socket_type: "SSL".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: String::new(),
|
||||
}],
|
||||
outgoing: vec![],
|
||||
};
|
||||
assert!(mail_config_to_server_config(&config).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_empty_incoming() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![],
|
||||
outgoing: vec![],
|
||||
};
|
||||
assert!(mail_config_to_server_config(&config).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_picks_imap_over_pop3() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![
|
||||
IncomingServer {
|
||||
protocol: "pop3".to_string(),
|
||||
hostname: "pop.example.com".to_string(),
|
||||
port: 995,
|
||||
socket_type: "SSL".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: String::new(),
|
||||
},
|
||||
make_imap_server("imap.example.com", 993, "SSL"),
|
||||
],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should find IMAP");
|
||||
assert_eq!(result.imap.host, "imap.example.com");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_imaps_protocol_variant() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![IncomingServer {
|
||||
protocol: "imaps".to_string(),
|
||||
hostname: "imap.example.com".to_string(),
|
||||
port: 993,
|
||||
socket_type: "SSL".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: String::new(),
|
||||
}],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should recognize 'imaps'");
|
||||
assert_eq!(result.imap.host, "imap.example.com");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_case_insensitive_protocol() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![IncomingServer {
|
||||
protocol: "IMAP".to_string(),
|
||||
hostname: "imap.example.com".to_string(),
|
||||
port: 143,
|
||||
socket_type: "STARTTLS".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: String::new(),
|
||||
}],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should recognize 'IMAP'");
|
||||
assert_eq!(result.imap.host, "imap.example.com");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_gmail_oauth2() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![IncomingServer {
|
||||
protocol: "imap".to_string(),
|
||||
hostname: "imap.gmail.com".to_string(),
|
||||
port: 993,
|
||||
socket_type: "SSL".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: "OAuth2".to_string(),
|
||||
}],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should convert");
|
||||
let oauth2 = result.oauth2.expect("Gmail should have OAuth2");
|
||||
assert_eq!(oauth2.issuer, "https://accounts.google.com");
|
||||
assert!(oauth2.scope.contains(&"https://mail.google.com/".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_outlook_oauth2() {
|
||||
let config = MailConfig {
|
||||
incoming: vec![IncomingServer {
|
||||
protocol: "imap".to_string(),
|
||||
hostname: "outlook.office365.com".to_string(),
|
||||
port: 993,
|
||||
socket_type: "SSL".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: "OAuth2".to_string(),
|
||||
}],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should convert");
|
||||
let oauth2 = result.oauth2.expect("Outlook should have OAuth2");
|
||||
assert!(oauth2.issuer.contains("microsoftonline"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn convert_unknown_host_no_oauth2() {
|
||||
// OAuth2 auth flag on an unknown hostname → no OAuth2 returned
|
||||
let config = MailConfig {
|
||||
incoming: vec![IncomingServer {
|
||||
protocol: "imap".to_string(),
|
||||
hostname: "mail.random-isp.example".to_string(),
|
||||
port: 993,
|
||||
socket_type: "SSL".to_string(),
|
||||
username: "%EMAILADDRESS%".to_string(),
|
||||
authentication: "OAuth2".to_string(),
|
||||
}],
|
||||
outgoing: vec![],
|
||||
};
|
||||
let result = mail_config_to_server_config(&config).expect("should convert");
|
||||
assert!(result.oauth2.is_none(), "unknown hostname → no OAuth2 mapping");
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,27 +16,28 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use std::collections::BTreeSet;
|
||||
|
||||
use crate::{
|
||||
decode_mailbox_name,
|
||||
modules::{
|
||||
decode_mailbox_name, raise_error,
|
||||
{
|
||||
account::migration::{AccountModel, AccountType},
|
||||
cache::imap::mailbox::{AttributeEnum, MailBox},
|
||||
context::executors::MAIL_CONTEXT,
|
||||
cache::imap::mailbox_cache,
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
imap::{executor::ImapExecutor, session::SessionStream},
|
||||
mailbox::list::convert_names_to_mailboxes,
|
||||
},
|
||||
raise_error,
|
||||
};
|
||||
use async_imap::types::Name;
|
||||
use async_imap::{types::Name, Session};
|
||||
use tracing::{debug, info, warn};
|
||||
|
||||
pub async fn get_sync_folders(account: &AccountModel) -> BichonResult<Vec<MailBox>> {
|
||||
pub async fn get_download_folders(
|
||||
account: &AccountModel,
|
||||
session: &mut Session<Box<dyn SessionStream>>,
|
||||
) -> BichonResult<Vec<MailBox>> {
|
||||
assert_eq!(account.account_type, AccountType::IMAP);
|
||||
let executor = MAIL_CONTEXT.imap(account.id).await?;
|
||||
let names = executor.list_all_mailboxes().await?;
|
||||
let names = ImapExecutor::list_all_mailboxes(session).await?;
|
||||
if names.is_empty() {
|
||||
warn!(
|
||||
"Account {}: No mailboxes returned from IMAP server.",
|
||||
@@ -61,8 +62,8 @@ pub async fn get_sync_folders(account: &AccountModel) -> BichonResult<Vec<MailBo
|
||||
mailboxes.iter().map(|(m, _)| m.name.clone()).collect(),
|
||||
)
|
||||
.await?;
|
||||
let account = AccountModel::get(account.id).await?;
|
||||
let subscribed = &account.sync_folders.unwrap_or_default();
|
||||
let account = AccountModel::get(account.id)?;
|
||||
let subscribed = &account.download_folders.unwrap_or_default();
|
||||
let is_noselect = |mailbox: &MailBox| {
|
||||
mailbox
|
||||
.attributes
|
||||
@@ -109,7 +110,7 @@ pub async fn get_sync_folders(account: &AccountModel) -> BichonResult<Vec<MailBo
|
||||
.iter()
|
||||
.map(|n| decode_mailbox_name!(n.name().to_string()))
|
||||
.collect();
|
||||
AccountModel::update_sync_folders(account.id, sync_folders).await?;
|
||||
AccountModel::update_download_folders(account.id, sync_folders)?;
|
||||
} else {
|
||||
warn!(
|
||||
"Account {}: No subscribed mailboxes found. This is unexpected — IMAP server should at least provide INBOX.",
|
||||
@@ -121,7 +122,7 @@ pub async fn get_sync_folders(account: &AccountModel) -> BichonResult<Vec<MailBo
|
||||
), ErrorCode::ImapUnexpectedResult));
|
||||
}
|
||||
}
|
||||
convert_names_to_mailboxes(account.id, matched_mailboxes).await
|
||||
convert_names_to_mailboxes(account.id, session, matched_mailboxes).await
|
||||
}
|
||||
|
||||
pub async fn detect_mailbox_changes(
|
||||
@@ -130,7 +131,7 @@ pub async fn detect_mailbox_changes(
|
||||
) -> BichonResult<()> {
|
||||
if account.known_folders.is_none() {
|
||||
// First time sync: just save without comparing
|
||||
AccountModel::update_known_folders(account.id, all_names).await?;
|
||||
AccountModel::update_known_folders(account.id, all_names)?;
|
||||
return Ok(());
|
||||
}
|
||||
let known_folders = account.known_folders.clone().unwrap_or_default();
|
||||
@@ -139,19 +140,19 @@ pub async fn detect_mailbox_changes(
|
||||
let deleted_folders: Vec<String> = known_folders.difference(&all_names).cloned().collect();
|
||||
|
||||
let has_changes = !new_folders.is_empty() || !deleted_folders.is_empty();
|
||||
let sync_folders = account.sync_folders.as_deref().unwrap_or_default();
|
||||
let download_folders = account.download_folders.as_deref().unwrap_or_default();
|
||||
// Handle deleted folders in sync_folders
|
||||
if !deleted_folders.is_empty() {
|
||||
// Check if any deleted folders are in sync_folders
|
||||
let remaining_sync_folders: Vec<String> = sync_folders
|
||||
let remaining_sync_folders: Vec<String> = download_folders
|
||||
.iter()
|
||||
.filter(|folder| !deleted_folders.contains(folder))
|
||||
.cloned()
|
||||
.collect();
|
||||
|
||||
// If sync_folders changed, update them
|
||||
if remaining_sync_folders.len() != sync_folders.len() {
|
||||
let removed_count = sync_folders.len() - remaining_sync_folders.len();
|
||||
if remaining_sync_folders.len() != download_folders.len() {
|
||||
let removed_count = download_folders.len() - remaining_sync_folders.len();
|
||||
info!(
|
||||
"Account {}: Removed {} deleted folders from sync_folders",
|
||||
account.id, removed_count
|
||||
@@ -159,7 +160,7 @@ pub async fn detect_mailbox_changes(
|
||||
// Note: When all subscribed folders are deleted (remaining_sync_folders empty),
|
||||
// the system's default behavior is to automatically fall back to syncing
|
||||
// only the default folders (INBOX and Sent) in subsequent operations
|
||||
AccountModel::update_sync_folders(account.id, remaining_sync_folders).await?;
|
||||
AccountModel::update_download_folders(account.id, remaining_sync_folders)?;
|
||||
}
|
||||
|
||||
info!(
|
||||
@@ -174,11 +175,22 @@ pub async fn detect_mailbox_changes(
|
||||
"Account {}: New folders detected: {:?}",
|
||||
account.id, new_folders
|
||||
);
|
||||
if account.auto_download_new_mailboxes.unwrap_or(false) {
|
||||
let mut updated: Vec<String> = download_folders.to_vec();
|
||||
updated.extend(new_folders.iter().cloned());
|
||||
AccountModel::update_download_folders(account.id, updated)?;
|
||||
info!(
|
||||
"Account {}: Auto-added {} new folders to download list",
|
||||
account.id,
|
||||
new_folders.len()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Update known folders only if there were changes
|
||||
if has_changes {
|
||||
AccountModel::update_known_folders(account.id, all_names).await?;
|
||||
AccountModel::update_known_folders(account.id, all_names)?;
|
||||
mailbox_cache::invalidate(account.id).await;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
156
crates/core/src/cache/imap/download/download_type.rs
vendored
Normal file
156
crates/core/src/cache/imap/download/download_type.rs
vendored
Normal file
@@ -0,0 +1,156 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use std::str::FromStr;
|
||||
|
||||
use chrono::{DateTime, Local, TimeZone, Utc};
|
||||
use cron::Schedule;
|
||||
|
||||
use crate::{
|
||||
utc_now,
|
||||
{
|
||||
account::{
|
||||
migration::AccountModel,
|
||||
state::{DownloadState, TriggerType},
|
||||
},
|
||||
error::BichonResult,
|
||||
},
|
||||
};
|
||||
|
||||
#[derive(Clone, Debug, Eq, PartialEq)]
|
||||
pub enum DownloadTask {
|
||||
FullFetch,
|
||||
TraceFetch,
|
||||
Idle,
|
||||
}
|
||||
|
||||
pub async fn decide_next_download_task(
|
||||
account: &AccountModel,
|
||||
trigger_type: TriggerType,
|
||||
) -> BichonResult<DownloadTask> {
|
||||
let state = match DownloadState::get(account.id)? {
|
||||
None => {
|
||||
DownloadState::init(account.id).await?;
|
||||
return Ok(DownloadTask::FullFetch);
|
||||
}
|
||||
Some(s) => s,
|
||||
};
|
||||
|
||||
let should_start = match trigger_type {
|
||||
TriggerType::Manual => true,
|
||||
TriggerType::Scheduled => {
|
||||
let now = utc_now!();
|
||||
let cooldown_ok = now - state.last_finished_at.unwrap_or(0) > 60 * 1000;
|
||||
if !cooldown_ok {
|
||||
false
|
||||
} else if let Some(ref schedule) = account.download_schedule {
|
||||
should_trigger_scheduled(schedule, state.last_trigger_at)
|
||||
} else {
|
||||
should_trigger_next_download(
|
||||
state.last_trigger_at,
|
||||
account.download_interval_min.unwrap_or(60),
|
||||
)
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if should_start {
|
||||
DownloadState::start_new_session(account.id, trigger_type)?;
|
||||
Ok(DownloadTask::TraceFetch)
|
||||
} else {
|
||||
Ok(DownloadTask::Idle)
|
||||
}
|
||||
}
|
||||
|
||||
fn should_trigger_next_download(last_trigger_at: i64, sync_interval_min: i64) -> bool {
|
||||
let now = utc_now!();
|
||||
now - last_trigger_at > (sync_interval_min * 60 * 1000)
|
||||
}
|
||||
|
||||
fn should_trigger_scheduled(schedule_str: &str, last_trigger_at: i64) -> bool {
|
||||
let schedule = match Schedule::from_str(schedule_str) {
|
||||
Ok(s) => s,
|
||||
Err(e) => {
|
||||
tracing::warn!(
|
||||
"Invalid cron expression '{}', falling back to no trigger: {}",
|
||||
schedule_str,
|
||||
e
|
||||
);
|
||||
return false;
|
||||
}
|
||||
};
|
||||
// last_trigger_at is a UTC millis timestamp; convert to server local time
|
||||
let last_utc = match Utc.timestamp_millis_opt(last_trigger_at) {
|
||||
chrono::LocalResult::Single(dt) => dt,
|
||||
_ => {
|
||||
tracing::warn!("Invalid last_trigger_at timestamp: {}", last_trigger_at);
|
||||
return false;
|
||||
}
|
||||
};
|
||||
let last_dt: DateTime<Local> = last_utc.with_timezone(&Local);
|
||||
let now = Local::now();
|
||||
schedule
|
||||
.after(&last_dt)
|
||||
.next()
|
||||
.map_or(false, |next| next <= now)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod test {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn cron_every_minute_triggers_after_60s() {
|
||||
// "0 * * * * *" = every minute at second 0. last_trigger 90s ago → should trigger
|
||||
let now = Local::now();
|
||||
let last_trigger = now.timestamp_millis() - 90_000;
|
||||
assert!(should_trigger_scheduled("0 * * * * *", last_trigger));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cron_daily_midnight_triggers_when_missed() {
|
||||
// "0 0 0 * * *" = daily at midnight
|
||||
// last_trigger was 25 hours ago → should trigger (we missed midnight)
|
||||
let now = Local::now();
|
||||
let last_trigger = now.timestamp_millis() - 25 * 60 * 60 * 1000;
|
||||
assert!(should_trigger_scheduled("0 0 0 * * *", last_trigger));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cron_daily_midnight_no_trigger_if_already_fired() {
|
||||
// "0 0 0 * * *" = daily at midnight
|
||||
// last_trigger was 1 minute ago → should NOT trigger
|
||||
let now = Local::now();
|
||||
let last_trigger = now.timestamp_millis() - 60_000;
|
||||
assert!(!should_trigger_scheduled("0 0 0 * * *", last_trigger));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_cron_returns_false() {
|
||||
assert!(!should_trigger_scheduled("invalid cron expression", 0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cron_every_hour_triggers() {
|
||||
// "0 0 * * * *" = every hour at minute 0, second 0
|
||||
// last_trigger was 61 minutes ago → should trigger
|
||||
let now = Local::now();
|
||||
let last_trigger = now.timestamp_millis() - 61 * 60 * 1000;
|
||||
assert!(should_trigger_scheduled("0 0 * * * *", last_trigger));
|
||||
}
|
||||
}
|
||||
1658
crates/core/src/cache/imap/download/flow.rs
vendored
Normal file
1658
crates/core/src/cache/imap/download/flow.rs
vendored
Normal file
File diff suppressed because it is too large
Load Diff
143
crates/core/src/cache/imap/download/mod.rs
vendored
Normal file
143
crates/core/src/cache/imap/download/mod.rs
vendored
Normal file
@@ -0,0 +1,143 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::{
|
||||
account::{
|
||||
migration::{AccountModel, AccountType},
|
||||
state::{DownloadState, DownloadStatus, TriggerType},
|
||||
},
|
||||
cache::imap::{download::flow::FetchDirection, mailbox::MailBox},
|
||||
error::BichonResult,
|
||||
imap::executor::ImapExecutor,
|
||||
};
|
||||
use download_folders::get_download_folders;
|
||||
use download_type::{decide_next_download_task, DownloadTask};
|
||||
use flow::reconcile_mailboxes;
|
||||
use rebuild::{rebuild_cache, rebuild_cache_by_date};
|
||||
use std::time::Instant;
|
||||
use tokio_util::sync::CancellationToken;
|
||||
use tracing::{debug, warn};
|
||||
|
||||
pub mod download_folders;
|
||||
pub mod download_type;
|
||||
pub mod flow;
|
||||
pub mod rebuild;
|
||||
|
||||
pub async fn process_imap_download(
|
||||
account: &AccountModel,
|
||||
token: CancellationToken,
|
||||
trigger_type: TriggerType,
|
||||
) -> BichonResult<()> {
|
||||
assert_eq!(account.account_type, AccountType::IMAP);
|
||||
let start_time = Instant::now();
|
||||
let account_id = account.id;
|
||||
let download_task = decide_next_download_task(account, trigger_type).await?;
|
||||
if matches!(download_task, DownloadTask::Idle) {
|
||||
return Ok(());
|
||||
}
|
||||
let mut session = match ImapExecutor::create_connection(account_id).await {
|
||||
Ok(session) => session,
|
||||
Err(e) => {
|
||||
let err_msg = format!("Failed to connect to IMAP server: {:#?}", e);
|
||||
DownloadState::append_session_error(account_id, err_msg.clone())?;
|
||||
DownloadState::update_session_status(
|
||||
account_id,
|
||||
DownloadStatus::Failed,
|
||||
Some(err_msg),
|
||||
)?;
|
||||
return Err(e);
|
||||
}
|
||||
};
|
||||
let remote_mailboxes = match get_download_folders(account, &mut session).await {
|
||||
Ok(mailboxes) => mailboxes,
|
||||
Err(err) => {
|
||||
let err_msg = format!("Failed to fetch mailboxes: {:#?}", err);
|
||||
warn!(account_id = account.id, error = %err, "{}", err_msg);
|
||||
DownloadState::append_session_error(account_id, err_msg.clone())?;
|
||||
DownloadState::update_session_status(
|
||||
account_id,
|
||||
DownloadStatus::Failed,
|
||||
Some(err_msg),
|
||||
)?;
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
session.logout().await.ok();
|
||||
if matches!(download_task, DownloadTask::FullFetch) {
|
||||
let result = match &account.date_since {
|
||||
Some(date_since) => {
|
||||
rebuild_cache_by_date(
|
||||
account,
|
||||
&remote_mailboxes,
|
||||
&date_since.since_date()?,
|
||||
FetchDirection::Since,
|
||||
token,
|
||||
)
|
||||
.await
|
||||
}
|
||||
None => match &account.date_before {
|
||||
Some(r) => {
|
||||
rebuild_cache_by_date(
|
||||
account,
|
||||
&remote_mailboxes,
|
||||
&r.calculate_date()?,
|
||||
FetchDirection::Before,
|
||||
token,
|
||||
)
|
||||
.await
|
||||
}
|
||||
None => rebuild_cache(account, &remote_mailboxes, token).await,
|
||||
},
|
||||
};
|
||||
match result {
|
||||
Ok(_) => {
|
||||
DownloadState::update_session_status(account_id, DownloadStatus::Success, None)?;
|
||||
}
|
||||
Err(e) => {
|
||||
let err_msg = format!("Email Download interrupted: {:#?}", e);
|
||||
DownloadState::append_session_error(account_id, err_msg.clone())?;
|
||||
DownloadState::update_session_status(
|
||||
account_id,
|
||||
DownloadStatus::Failed,
|
||||
Some(err_msg),
|
||||
)?;
|
||||
}
|
||||
}
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let local_mailboxes = MailBox::list_all(account_id)?;
|
||||
match reconcile_mailboxes(account, &remote_mailboxes, &local_mailboxes, token).await {
|
||||
Ok(_) => DownloadState::update_session_status(account_id, DownloadStatus::Success, None)?,
|
||||
Err(e) => {
|
||||
let err_msg = format!("Email Download interrupted: {:#?}", e);
|
||||
DownloadState::append_session_error(account_id, err_msg.clone())?;
|
||||
DownloadState::update_session_status(
|
||||
account_id,
|
||||
DownloadStatus::Failed,
|
||||
Some(err_msg),
|
||||
)?;
|
||||
}
|
||||
}
|
||||
let elapsed_time = start_time.elapsed().as_secs();
|
||||
debug!(
|
||||
"Account{{{}}} Incremental sync completed: {} seconds elapsed.",
|
||||
account.email, elapsed_time
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
268
crates/core/src/cache/imap/download/rebuild.rs
vendored
Normal file
268
crates/core/src/cache/imap/download/rebuild.rs
vendored
Normal file
@@ -0,0 +1,268 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::{
|
||||
account::{
|
||||
migration::AccountModel,
|
||||
state::{DownloadState, DownloadStatus, FolderStatus},
|
||||
},
|
||||
cache::{
|
||||
imap::{
|
||||
download::flow::{fetch_and_save_by_date, fetch_and_save_full_mailbox, FetchDirection},
|
||||
mailbox::MailBox,
|
||||
},
|
||||
SEMAPHORE,
|
||||
},
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
raise_error,
|
||||
store::tantivy::{attachment::ATTACHMENT_MANAGER, envelope::ENVELOPE_MANAGER},
|
||||
};
|
||||
|
||||
use tokio_util::sync::CancellationToken;
|
||||
use tracing::{error, info};
|
||||
|
||||
pub async fn rebuild_cache(
|
||||
account: &AccountModel,
|
||||
remote_mailboxes: &[MailBox],
|
||||
token: CancellationToken,
|
||||
) -> BichonResult<()> {
|
||||
MailBox::batch_insert(remote_mailboxes)?;
|
||||
DownloadState::init_folder_details(
|
||||
account.id,
|
||||
remote_mailboxes.iter().map(|m| m.name.clone()).collect(),
|
||||
)?;
|
||||
|
||||
let mut has_error = false;
|
||||
let mut last_err = None;
|
||||
|
||||
for mailbox in remote_mailboxes {
|
||||
if token.is_cancelled() {
|
||||
DownloadState::update_session_status(
|
||||
account.id,
|
||||
DownloadStatus::Cancelled,
|
||||
Some("Received termination signal (User stop or System shutdown)".to_string()),
|
||||
)?;
|
||||
break;
|
||||
}
|
||||
if mailbox.exists == 0 {
|
||||
info!(
|
||||
"Account {}: Mailbox '{}' on the remote server has no emails. Skipping fetch for this mailbox.",
|
||||
account.id, &mailbox.name
|
||||
);
|
||||
DownloadState::update_folder_progress(
|
||||
account.id,
|
||||
mailbox.name.clone(),
|
||||
0,
|
||||
0,
|
||||
FolderStatus::Success,
|
||||
None,
|
||||
)?;
|
||||
continue;
|
||||
}
|
||||
let account = account.clone();
|
||||
let mailbox = mailbox.clone();
|
||||
|
||||
let _global_permit = match SEMAPHORE.clone().acquire_owned().await {
|
||||
Ok(permit) => permit,
|
||||
Err(err) => {
|
||||
error!(
|
||||
"Failed to acquire global semaphore permit for account {} mailbox '{}': {:#?}",
|
||||
account.id, &mailbox.name, err
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
|
||||
match fetch_and_save_full_mailbox(&account, &mailbox, token.clone()).await {
|
||||
Ok(new_highest_uid) => {
|
||||
let mut updated = mailbox.clone();
|
||||
updated.highest_uid = new_highest_uid;
|
||||
MailBox::batch_upsert(&[updated])?;
|
||||
}
|
||||
Err(err) => {
|
||||
has_error = true;
|
||||
tracing::error!("Folder sync task failed: {:#?}", err);
|
||||
last_err = Some(err);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if has_error {
|
||||
if let Some(e) = last_err {
|
||||
return Err(e);
|
||||
}
|
||||
return Err(raise_error!(
|
||||
"Some tasks failed".into(),
|
||||
ErrorCode::InternalError
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn rebuild_cache_by_date(
|
||||
account: &AccountModel,
|
||||
remote_mailboxes: &[MailBox],
|
||||
date: &str,
|
||||
direction: FetchDirection,
|
||||
token: CancellationToken,
|
||||
) -> BichonResult<()> {
|
||||
MailBox::batch_insert(remote_mailboxes)?;
|
||||
DownloadState::init_folder_details(
|
||||
account.id,
|
||||
remote_mailboxes.iter().map(|m| m.name.clone()).collect(),
|
||||
)?;
|
||||
|
||||
let mut has_error = false;
|
||||
let mut last_err = None;
|
||||
|
||||
for mailbox in remote_mailboxes {
|
||||
if token.is_cancelled() {
|
||||
DownloadState::update_session_status(
|
||||
account.id,
|
||||
DownloadStatus::Cancelled,
|
||||
Some("Received termination signal (User stop or System shutdown)".to_string()),
|
||||
)?;
|
||||
break;
|
||||
}
|
||||
if mailbox.exists == 0 {
|
||||
info!(
|
||||
"Account {}: Mailbox '{}' on the remote server has no emails. Skipping fetch for this mailbox.",
|
||||
account.id, &mailbox.name
|
||||
);
|
||||
|
||||
DownloadState::update_folder_progress(
|
||||
account.id,
|
||||
mailbox.name.clone(),
|
||||
0,
|
||||
0,
|
||||
FolderStatus::Success,
|
||||
None,
|
||||
)?;
|
||||
continue;
|
||||
}
|
||||
let account = account.clone();
|
||||
let mailbox = mailbox.clone();
|
||||
let date = date.to_string();
|
||||
let direction = direction.clone();
|
||||
|
||||
let _global_permit = match SEMAPHORE.clone().acquire_owned().await {
|
||||
Ok(permit) => permit,
|
||||
Err(err) => {
|
||||
error!(
|
||||
"Failed to acquire global semaphore permit for account {} mailbox '{}': {:#?}",
|
||||
account.id, &mailbox.name, err
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
match fetch_and_save_by_date(&account, date.as_str(), &mailbox, direction, token.clone())
|
||||
.await
|
||||
{
|
||||
Ok(new_highest_uid) => {
|
||||
let mut updated = mailbox.clone();
|
||||
updated.highest_uid = new_highest_uid;
|
||||
MailBox::batch_upsert(&[updated])?;
|
||||
}
|
||||
Err(err) => {
|
||||
has_error = true;
|
||||
tracing::error!("Folder sync task failed: {:#?}", err);
|
||||
last_err = Some(err);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if has_error {
|
||||
if let Some(e) = last_err {
|
||||
return Err(e);
|
||||
}
|
||||
return Err(raise_error!(
|
||||
"Some tasks failed".into(),
|
||||
ErrorCode::InternalError
|
||||
));
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn rebuild_mailbox_cache(
|
||||
account: &AccountModel,
|
||||
local_mailbox: &MailBox,
|
||||
remote_mailbox: &MailBox,
|
||||
token: CancellationToken,
|
||||
) -> BichonResult<Option<u32>> {
|
||||
ENVELOPE_MANAGER
|
||||
.delete_mailbox_envelopes(account.id, vec![local_mailbox.id])
|
||||
.await?;
|
||||
ATTACHMENT_MANAGER
|
||||
.delete_mailbox_attachments(account.id, vec![local_mailbox.id])
|
||||
.await?;
|
||||
if remote_mailbox.exists == 0 {
|
||||
info!(
|
||||
"Account {}: Mailbox '{}' has no emails on the remote server. The mailbox is empty, no envelopes to fetch.",
|
||||
account.id,
|
||||
&local_mailbox.name
|
||||
);
|
||||
DownloadState::update_folder_progress(
|
||||
account.id,
|
||||
remote_mailbox.name.clone(),
|
||||
0,
|
||||
0,
|
||||
FolderStatus::Success,
|
||||
None,
|
||||
)?;
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
let result = fetch_and_save_full_mailbox(account, remote_mailbox, token).await?;
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
pub async fn rebuild_mailbox_cache_by_date(
|
||||
account: &AccountModel,
|
||||
local_mailbox_id: u64,
|
||||
date: &str,
|
||||
remote: &MailBox,
|
||||
direction: FetchDirection,
|
||||
token: CancellationToken,
|
||||
) -> BichonResult<Option<u32>> {
|
||||
ENVELOPE_MANAGER
|
||||
.delete_mailbox_envelopes(account.id, vec![local_mailbox_id])
|
||||
.await?;
|
||||
ATTACHMENT_MANAGER
|
||||
.delete_mailbox_attachments(account.id, vec![local_mailbox_id])
|
||||
.await?;
|
||||
if remote.exists == 0 {
|
||||
info!(
|
||||
"Account {}: Mailbox '{}' has no emails on the remote server. The mailbox is empty, no envelopes to fetch.",
|
||||
account.id,
|
||||
&remote.name
|
||||
);
|
||||
DownloadState::update_folder_progress(
|
||||
account.id,
|
||||
remote.name.clone(),
|
||||
0,
|
||||
0,
|
||||
FolderStatus::Success,
|
||||
None,
|
||||
)?;
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
let result = fetch_and_save_by_date(account, date, remote, direction, token).await?;
|
||||
Ok(result)
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -17,32 +17,24 @@
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::{
|
||||
decode_mailbox_name, encode_mailbox_name,
|
||||
modules::{
|
||||
decode_mailbox_name, encode_mailbox_name, raise_error,
|
||||
{
|
||||
database::{
|
||||
async_find_impl, batch_delete_impl, batch_insert_impl, batch_upsert_impl, delete_impl,
|
||||
filter_by_secondary_key_impl, manager::DB_MANAGER,
|
||||
batch_delete_impl, batch_insert_impl, batch_upsert_impl, delete_impl, filter_impl,
|
||||
find_impl, manager::DB_MANAGER, MemDbModel,
|
||||
},
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
},
|
||||
raise_error,
|
||||
};
|
||||
use async_imap::types::{Name, NameAttribute};
|
||||
use itertools::Itertools;
|
||||
use native_db::*;
|
||||
use native_model::{native_model, Model};
|
||||
use poem_openapi::{Enum, Object};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
#[native_model(id = 1, version = 1)]
|
||||
#[native_db]
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct MailBox {
|
||||
/// The unique identifier for the mailbox
|
||||
#[primary_key]
|
||||
pub id: u64,
|
||||
/// The ID of the account associated with the mailbox
|
||||
#[secondary_key]
|
||||
pub account_id: u64,
|
||||
/// The unique, decoded, human-readable name of the mailbox (e.g., "INBOX", "Sent Items").
|
||||
/// This is the decoded name as presented to users, derived from the IMAP server's mailbox name
|
||||
@@ -64,6 +56,19 @@ pub struct MailBox {
|
||||
/// The validity identifier for UIDs in this mailbox, used to ensure UID consistency across sessions.
|
||||
/// If `None`, the IMAP server has not provided this information.
|
||||
pub uid_validity: Option<u32>,
|
||||
/// The highest UID that has been successfully downloaded and stored locally.
|
||||
/// Used for incremental sync: next fetch starts from `highest_uid + 1`.
|
||||
/// If `None`, a fallback query against the Tantivy index will be performed once.
|
||||
pub highest_uid: Option<u32>,
|
||||
}
|
||||
|
||||
impl MemDbModel for MailBox {
|
||||
fn collection() -> &'static str {
|
||||
"mailboxes"
|
||||
}
|
||||
fn key(&self) -> String {
|
||||
self.id.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
impl MailBox {
|
||||
@@ -71,26 +76,8 @@ impl MailBox {
|
||||
encode_mailbox_name!(&self.name)
|
||||
}
|
||||
|
||||
// pub async fn batch_delete(mailboxes: Vec<MailBox>) -> BichonResult<()> {
|
||||
// batch_delete_impl(DB_MANAGER.envelope_db(), move |rw| {
|
||||
// let mut to_deleted = Vec::new();
|
||||
// for mailbox in mailboxes {
|
||||
// let retrived = rw
|
||||
// .get()
|
||||
// .primary::<MailBox>(mailbox.id)
|
||||
// .map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
// if let Some(retrived) = retrived {
|
||||
// to_deleted.push(retrived);
|
||||
// }
|
||||
// }
|
||||
// Ok(to_deleted)
|
||||
// })
|
||||
// .await?;
|
||||
// Ok(())
|
||||
// }
|
||||
|
||||
pub async fn get(id: u64) -> BichonResult<MailBox> {
|
||||
let result = async_find_impl::<MailBox>(DB_MANAGER.envelope_db(), id).await?;
|
||||
pub fn get(id: u64) -> BichonResult<MailBox> {
|
||||
let result = find_impl::<MailBox>(DB_MANAGER.db(), &id.to_string())?;
|
||||
Ok(result.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("mailbox {} not found", id),
|
||||
@@ -99,47 +86,40 @@ impl MailBox {
|
||||
})?)
|
||||
}
|
||||
|
||||
pub async fn delete(id: u64) -> BichonResult<()> {
|
||||
delete_impl(DB_MANAGER.envelope_db(), move |rw| {
|
||||
rw.get()
|
||||
.primary::<MailBox>(id)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
.ok_or_else(|| raise_error!("mailbox missing".into(), ErrorCode::InternalError))
|
||||
})
|
||||
.await
|
||||
pub fn delete(id: u64) -> BichonResult<()> {
|
||||
delete_impl::<MailBox>(DB_MANAGER.db(), &id.to_string())
|
||||
}
|
||||
|
||||
pub async fn list_all(account_id: u64) -> BichonResult<Vec<MailBox>> {
|
||||
filter_by_secondary_key_impl(DB_MANAGER.envelope_db(), MailBoxKey::account_id, account_id)
|
||||
.await
|
||||
pub fn list_all(account_id: u64) -> BichonResult<Vec<MailBox>> {
|
||||
filter_impl::<MailBox, _>(DB_MANAGER.db(), move |m| m.account_id == account_id)
|
||||
}
|
||||
|
||||
pub async fn batch_insert(mailboxes: &[MailBox]) -> BichonResult<()> {
|
||||
batch_insert_impl(DB_MANAGER.envelope_db(), mailboxes.to_vec()).await
|
||||
pub fn find_mailbox(account_id: u64, mailbox_id: u64) -> BichonResult<Option<MailBox>> {
|
||||
let all = filter_impl::<MailBox, _>(DB_MANAGER.db(), move |m| m.account_id == account_id)?;
|
||||
Ok(all.into_iter().find(|m| m.id == mailbox_id))
|
||||
}
|
||||
|
||||
pub async fn batch_upsert(mailboxes: &[MailBox]) -> BichonResult<()> {
|
||||
batch_upsert_impl(DB_MANAGER.envelope_db(), mailboxes.to_vec()).await
|
||||
pub fn batch_insert(mailboxes: &[MailBox]) -> BichonResult<()> {
|
||||
batch_insert_impl(DB_MANAGER.db(), mailboxes.to_vec())
|
||||
}
|
||||
|
||||
pub async fn clean(account_id: u64) -> BichonResult<()> {
|
||||
batch_delete_impl(DB_MANAGER.envelope_db(), move |rw| {
|
||||
let mailboxes: Vec<MailBox> = rw
|
||||
.scan()
|
||||
.secondary::<MailBox>(MailBoxKey::account_id)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
.start_with(account_id)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?
|
||||
.try_collect()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
Ok(mailboxes)
|
||||
})
|
||||
.await?;
|
||||
pub fn batch_upsert(mailboxes: &[MailBox]) -> BichonResult<()> {
|
||||
batch_upsert_impl(DB_MANAGER.db(), mailboxes.to_vec())
|
||||
}
|
||||
|
||||
pub fn clean(account_id: u64) -> BichonResult<()> {
|
||||
let mailboxes =
|
||||
filter_impl::<MailBox, _>(DB_MANAGER.db(), move |m| m.account_id == account_id)?;
|
||||
let keys: Vec<String> = mailboxes.iter().map(|m| m.id.to_string()).collect();
|
||||
if !keys.is_empty() {
|
||||
batch_delete_impl::<MailBox>(DB_MANAGER.db(), keys)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Eq, PartialEq, Deserialize, Serialize, Object)]
|
||||
#[derive(Clone, Debug, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct Attribute {
|
||||
pub attr: AttributeEnum,
|
||||
pub extension: Option<String>,
|
||||
@@ -151,7 +131,8 @@ impl Attribute {
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Eq, PartialEq, Deserialize, Serialize, Enum)]
|
||||
#[derive(Clone, Debug, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Enum))]
|
||||
pub enum AttributeEnum {
|
||||
NoInferiors,
|
||||
NoSelect,
|
||||
114
crates/core/src/cache/imap/mailbox_cache.rs
vendored
Normal file
114
crates/core/src/cache/imap/mailbox_cache.rs
vendored
Normal file
@@ -0,0 +1,114 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::cache::imap::mailbox::MailBox;
|
||||
use crate::utc_now;
|
||||
use lru::LruCache;
|
||||
use std::collections::HashMap;
|
||||
use std::num::NonZeroUsize;
|
||||
use std::sync::LazyLock;
|
||||
use tokio::sync::Mutex;
|
||||
|
||||
struct CacheEntry {
|
||||
mailboxes: Vec<MailBox>,
|
||||
fetched_at: i64,
|
||||
}
|
||||
|
||||
static CACHE: LazyLock<Mutex<LruCache<u64, CacheEntry>>> = LazyLock::new(|| {
|
||||
Mutex::new(LruCache::new(NonZeroUsize::new(64).unwrap()))
|
||||
});
|
||||
|
||||
const TTL_MS: i64 = 10 * 60 * 1000; // 10 minutes
|
||||
|
||||
pub async fn get(account_id: u64) -> Option<Vec<MailBox>> {
|
||||
let mut guard = CACHE.lock().await;
|
||||
if let Some(entry) = guard.get(&account_id) {
|
||||
if utc_now!() - entry.fetched_at < TTL_MS {
|
||||
return Some(entry.mailboxes.clone());
|
||||
}
|
||||
guard.pop(&account_id);
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
pub async fn set(account_id: u64, mailboxes: Vec<MailBox>) {
|
||||
let mut guard = CACHE.lock().await;
|
||||
guard.put(
|
||||
account_id,
|
||||
CacheEntry {
|
||||
mailboxes,
|
||||
fetched_at: utc_now!(),
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
pub async fn invalidate(account_id: u64) {
|
||||
let mut guard = CACHE.lock().await;
|
||||
guard.pop(&account_id);
|
||||
}
|
||||
|
||||
// Background fetch state tracking
|
||||
#[derive(Clone, Debug)]
|
||||
pub enum FetchStatus {
|
||||
Fetching { examined: usize, total: usize },
|
||||
Ready,
|
||||
Error(String),
|
||||
}
|
||||
|
||||
static FETCH_STATES: LazyLock<Mutex<HashMap<u64, FetchStatus>>> =
|
||||
LazyLock::new(|| Mutex::new(HashMap::new()));
|
||||
|
||||
pub async fn fetch_status(account_id: u64) -> Option<FetchStatus> {
|
||||
FETCH_STATES.lock().await.get(&account_id).cloned()
|
||||
}
|
||||
|
||||
pub async fn set_fetching(account_id: u64) {
|
||||
FETCH_STATES.lock().await.insert(
|
||||
account_id,
|
||||
FetchStatus::Fetching {
|
||||
examined: 0,
|
||||
total: 0,
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
pub async fn update_fetch_progress(account_id: u64, examined: usize, total: usize) {
|
||||
let mut guard = FETCH_STATES.lock().await;
|
||||
guard.insert(
|
||||
account_id,
|
||||
FetchStatus::Fetching { examined, total },
|
||||
);
|
||||
}
|
||||
|
||||
pub async fn set_fetch_ready(account_id: u64) {
|
||||
FETCH_STATES
|
||||
.lock()
|
||||
.await
|
||||
.insert(account_id, FetchStatus::Ready);
|
||||
}
|
||||
|
||||
pub async fn set_fetch_error(account_id: u64, error: String) {
|
||||
FETCH_STATES
|
||||
.lock()
|
||||
.await
|
||||
.insert(account_id, FetchStatus::Error(error));
|
||||
}
|
||||
|
||||
pub async fn clear_fetch_state(account_id: u64) {
|
||||
FETCH_STATES.lock().await.remove(&account_id);
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,30 +16,20 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use std::sync::LazyLock;
|
||||
|
||||
use crate::modules::{account::state::AccountRunningState, database::ModelsAdapter};
|
||||
use ahash::{AHashMap, AHashSet};
|
||||
use mailbox::MailBox;
|
||||
use native_db::Models;
|
||||
|
||||
pub mod download;
|
||||
pub mod mailbox;
|
||||
pub mod sync;
|
||||
pub mod mailbox_cache;
|
||||
pub mod task;
|
||||
|
||||
pub static MAILBOX_MODELS: LazyLock<Models> = LazyLock::new(|| {
|
||||
let mut adapter = ModelsAdapter::new();
|
||||
adapter.register_model::<MailBox>();
|
||||
adapter.register_model::<AccountRunningState>();
|
||||
adapter.models
|
||||
});
|
||||
|
||||
pub fn find_missing_mailboxes(
|
||||
local_mailboxes: &[MailBox],
|
||||
server_mailboxes: &[MailBox],
|
||||
) -> Vec<MailBox> {
|
||||
let local_names: AHashSet<_> = local_mailboxes.iter().map(|m| &m.name).collect();
|
||||
let local_names: HashSet<_> = local_mailboxes.iter().map(|m| &m.name).collect();
|
||||
server_mailboxes
|
||||
.iter()
|
||||
.filter(|m| !local_names.contains(&m.name))
|
||||
@@ -51,7 +41,7 @@ pub fn find_intersecting_mailboxes(
|
||||
local_mailboxes: &[MailBox],
|
||||
remote_mailboxes: &[MailBox],
|
||||
) -> Vec<(MailBox, MailBox)> {
|
||||
let local_map: AHashMap<_, _> = local_mailboxes
|
||||
let local_map: HashMap<_, _> = local_mailboxes
|
||||
.iter()
|
||||
.map(|m| (m.name.clone(), m.clone()))
|
||||
.collect();
|
||||
283
crates/core/src/cache/imap/task.rs
vendored
Normal file
283
crates/core/src/cache/imap/task.rs
vendored
Normal file
@@ -0,0 +1,283 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::account::entity::AuthType;
|
||||
use crate::account::state::{DownloadState, TriggerType};
|
||||
use crate::cache::imap::download::process_imap_download;
|
||||
use crate::common::periodic::{PeriodicTask, TaskHandle};
|
||||
use crate::error::code::ErrorCode;
|
||||
use crate::oauth2::token::OAuth2AccessToken;
|
||||
use crate::{account::migration::AccountModel, error::BichonResult};
|
||||
use crate::{raise_error, utc_now};
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::sync::atomic::{AtomicI64, Ordering};
|
||||
use std::{sync::LazyLock, time::Duration};
|
||||
use tokio::sync::Mutex;
|
||||
use tokio::task::JoinHandle;
|
||||
use tokio_util::sync::CancellationToken;
|
||||
use tracing::{error, info, warn};
|
||||
|
||||
static _DESCRIPTION: &str = "This task periodically synchronizes mailbox data for a specified account, ensuring that all local data is up-to-date.";
|
||||
const TASK_INTERVAL: Duration = Duration::from_secs(10);
|
||||
pub static SYNC_TASKS: LazyLock<AccountDownTask> = LazyLock::new(AccountDownTask::new);
|
||||
static LAST_WARN_TIME: AtomicI64 = AtomicI64::new(0);
|
||||
const WARN_INTERVAL_MS: i64 = 600_000;
|
||||
|
||||
pub struct AccountDownTask {
|
||||
tasks: Mutex<Option<HashMap<u64, (TaskHandle, CancellationToken)>>>,
|
||||
manual_tasks: Mutex<HashMap<u64, (JoinHandle<()>, CancellationToken)>>,
|
||||
busy_accounts: Mutex<HashSet<u64>>,
|
||||
}
|
||||
|
||||
impl AccountDownTask {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
tasks: Mutex::new(Some(HashMap::new())),
|
||||
manual_tasks: Mutex::new(HashMap::new()),
|
||||
busy_accounts: Mutex::new(HashSet::new()),
|
||||
}
|
||||
}
|
||||
|
||||
async fn set_busy(&self, account_id: u64, is_busy: bool) {
|
||||
let mut guard = self.busy_accounts.lock().await;
|
||||
if is_busy {
|
||||
guard.insert(account_id);
|
||||
} else {
|
||||
guard.remove(&account_id);
|
||||
}
|
||||
}
|
||||
|
||||
/// Atomically check and set busy. Returns true if we claimed the slot,
|
||||
/// false if another task is already busy on this account.
|
||||
async fn try_set_busy(&self, account_id: u64) -> bool {
|
||||
let mut guard = self.busy_accounts.lock().await;
|
||||
if guard.contains(&account_id) {
|
||||
false
|
||||
} else {
|
||||
guard.insert(account_id);
|
||||
true
|
||||
}
|
||||
}
|
||||
|
||||
// async fn is_busy(&self, account_id: u64) -> bool {
|
||||
// self.busy_accounts.lock().await.contains(&account_id)
|
||||
// }
|
||||
|
||||
pub async fn start_download_task(&self, account_id: u64, email: String) {
|
||||
let task_name = format!("account-download-task-{}-{}", account_id, &email);
|
||||
let periodic_task = PeriodicTask::new(&task_name);
|
||||
|
||||
let cancel_token = CancellationToken::new();
|
||||
let task_token = cancel_token.clone();
|
||||
|
||||
let task = move |param: Option<u64>| {
|
||||
let account_id = param.unwrap();
|
||||
let internal_token = task_token.clone();
|
||||
Box::pin(async move {
|
||||
if SYNC_TASKS.is_manual_running(account_id).await {
|
||||
info!(
|
||||
"Account {}: Scheduled task skipped (Manual task is running).",
|
||||
account_id
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
if !SYNC_TASKS.try_set_busy(account_id).await {
|
||||
warn!(
|
||||
"Account {}: Scheduled task skipped (Previous sync still active).",
|
||||
account_id
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let _busy_guard = scopeguard::guard(account_id, |id| {
|
||||
tokio::spawn(async move {
|
||||
SYNC_TASKS.set_busy(id, false).await;
|
||||
});
|
||||
});
|
||||
let account = AccountModel::get(account_id).ok();
|
||||
match account {
|
||||
Some(account) => {
|
||||
if account.deleting {
|
||||
return Ok(());
|
||||
}
|
||||
if !account.enabled {
|
||||
let last = LAST_WARN_TIME.load(Ordering::Relaxed);
|
||||
let now = utc_now!();
|
||||
if now - last >= WARN_INTERVAL_MS {
|
||||
LAST_WARN_TIME.store(now, Ordering::Relaxed);
|
||||
warn!(
|
||||
"Account {}: download aborted. Account is currently disabled.",
|
||||
account_id
|
||||
);
|
||||
}
|
||||
} else {
|
||||
if let Some(imap) = &account.imap {
|
||||
if let AuthType::OAuth2 = imap.auth.auth_type {
|
||||
if OAuth2AccessToken::get(account.id)?.is_none() {
|
||||
if utc_now!() % 300_000 == 0 {
|
||||
warn!("Account {}: download aborted. OAuth2 authorization not completed. Please visit the rustmailer admin page to authorize this account.", account_id);
|
||||
}
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
}
|
||||
if let Err(e) = process_imap_download(
|
||||
&account,
|
||||
internal_token,
|
||||
TriggerType::Scheduled,
|
||||
)
|
||||
.await
|
||||
{
|
||||
DownloadState::append_session_error(
|
||||
account.id,
|
||||
format!("error in account download task: {:#?}", e),
|
||||
)?;
|
||||
error!(
|
||||
"Failed to download mailbox data for '{}': {:?}",
|
||||
account_id, e
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
None => {
|
||||
error!(
|
||||
"Account {}: download aborted. Account entity not found.",
|
||||
account_id
|
||||
);
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
};
|
||||
let handler = periodic_task.start(task, Some(account_id), TASK_INTERVAL, true, true);
|
||||
self.add_task(account_id, (handler, cancel_token)).await;
|
||||
}
|
||||
|
||||
pub async fn add_task(&self, account_id: u64, handler: (TaskHandle, CancellationToken)) {
|
||||
let mut guard = self.tasks.lock().await;
|
||||
if let Some(map) = guard.as_mut() {
|
||||
map.insert(account_id, handler);
|
||||
} else {
|
||||
tracing::error!("Failed to add task: HashMap has been taken during shutdown.");
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn stop(&self, account_id: u64) -> BichonResult<()> {
|
||||
let mut guard = self.tasks.lock().await;
|
||||
if let Some(map) = guard.as_mut() {
|
||||
if let Some((handler, token)) = map.remove(&account_id) {
|
||||
drop(guard);
|
||||
token.cancel();
|
||||
handler.cancel().await;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn shutdown(&self) {
|
||||
let mut guard = self.tasks.lock().await;
|
||||
if let Some(map) = guard.take() {
|
||||
drop(guard);
|
||||
for (account_id, (handler, token)) in map {
|
||||
info!(
|
||||
"Shutdown: Sending cancel signal to account {}...",
|
||||
account_id
|
||||
);
|
||||
token.cancel();
|
||||
if let Err(_) = tokio::time::timeout(Duration::from_secs(5), handler.stop()).await {
|
||||
error!(
|
||||
"Shutdown: Account {} download task forced timeout.",
|
||||
account_id
|
||||
);
|
||||
}
|
||||
}
|
||||
info!("Shutdown: All download tasks processed.");
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn start_manual_task(&self, account_id: u64) -> BichonResult<()> {
|
||||
{
|
||||
if self.is_manual_running(account_id).await {
|
||||
return Err(raise_error!(
|
||||
"Manual task already running.".into(),
|
||||
ErrorCode::Forbidden
|
||||
));
|
||||
}
|
||||
if !self.try_set_busy(account_id).await {
|
||||
return Err(raise_error!(
|
||||
"The background synchronization is currently active. Please try again in a few seconds.".into(),
|
||||
ErrorCode::Forbidden
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let cancel_token = CancellationToken::new();
|
||||
let token_clone = cancel_token.clone();
|
||||
let handle = tokio::spawn(async move {
|
||||
// busy already claimed by caller via try_set_busy
|
||||
let _cleanup = scopeguard::guard(account_id, |id| {
|
||||
tokio::spawn(async move {
|
||||
SYNC_TASKS.set_busy(id, false).await;
|
||||
let mut guard = SYNC_TASKS.manual_tasks.lock().await;
|
||||
guard.remove(&id);
|
||||
});
|
||||
});
|
||||
if token_clone.is_cancelled() {
|
||||
return;
|
||||
}
|
||||
let account = match AccountModel::get(account_id) {
|
||||
Ok(acc) => acc,
|
||||
Err(e) => {
|
||||
error!("Failed to fetch account {}: {:?}", account_id, e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
if account.deleting {
|
||||
return;
|
||||
}
|
||||
|
||||
if let Err(e) = process_imap_download(&account, token_clone, TriggerType::Manual).await
|
||||
{
|
||||
error!("Manual download failed for {}: {:?}", account_id, e);
|
||||
let error_msg = format!("error in account download task: {:#?}", e);
|
||||
let _ = DownloadState::append_session_error(account.id, error_msg);
|
||||
}
|
||||
});
|
||||
{
|
||||
let mut guard = self.manual_tasks.lock().await;
|
||||
guard.insert(account_id, (handle, cancel_token));
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn cancel_manual_task(&self, account_id: u64) {
|
||||
let mut guard = self.manual_tasks.lock().await;
|
||||
if let Some((handle, token)) = guard.remove(&account_id) {
|
||||
token.cancel();
|
||||
let _ = handle.await;
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn is_manual_running(&self, account_id: u64) -> bool {
|
||||
let guard = self.manual_tasks.lock().await;
|
||||
guard.contains_key(&account_id)
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -17,7 +17,7 @@
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use crate::modules::settings::cli::SETTINGS;
|
||||
use crate::settings::cli::SETTINGS;
|
||||
use std::sync::{Arc, LazyLock};
|
||||
use tokio::sync::Semaphore;
|
||||
|
||||
143
crates/core/src/common/auth.rs
Normal file
143
crates/core/src/common/auth.rs
Normal file
@@ -0,0 +1,143 @@
|
||||
use std::{
|
||||
collections::{BTreeSet, HashSet},
|
||||
net::IpAddr,
|
||||
};
|
||||
|
||||
use crate::{
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
raise_error,
|
||||
users::{permissions::Permission, role::UserRole, UserModel},
|
||||
};
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct ClientContext {
|
||||
pub ip_addr: Option<IpAddr>,
|
||||
pub user: UserModel,
|
||||
}
|
||||
|
||||
impl ClientContext {
|
||||
pub fn require_any_permission(
|
||||
&self,
|
||||
requirements: Vec<(Option<u64>, &str)>,
|
||||
) -> BichonResult<()> {
|
||||
for (account_id, permission) in requirements {
|
||||
if self.has_permission(account_id, permission) {
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
Err(raise_error!(
|
||||
"Access denied: Insufficient permissions to perform this action.".into(),
|
||||
ErrorCode::Forbidden
|
||||
))
|
||||
}
|
||||
|
||||
pub fn check_has_permission(
|
||||
user: &UserModel,
|
||||
account_id: Option<u64>,
|
||||
permission: &str,
|
||||
) -> bool {
|
||||
if user.is_admin() {
|
||||
return true;
|
||||
}
|
||||
|
||||
let mut global_perms = HashSet::new();
|
||||
for rid in &user.global_roles {
|
||||
if let Some(role) = UserRole::find(*rid).ok().flatten() {
|
||||
global_perms.extend(role.permissions);
|
||||
}
|
||||
}
|
||||
|
||||
if Self::check_global_logic(&global_perms, permission) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if let Some(aid) = account_id {
|
||||
if let Some(role_id) = user.account_access_map.get(&aid) {
|
||||
if let Some(role) = UserRole::find(*role_id).ok().flatten() {
|
||||
if role.permissions.contains(&permission.to_string())
|
||||
|| Self::check_account_logic(&role.permissions, permission)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
pub fn has_permission(&self, account_id: Option<u64>, permission: &str) -> bool {
|
||||
if self.user.is_admin() {
|
||||
return true;
|
||||
}
|
||||
|
||||
let mut global_perms = HashSet::new();
|
||||
for rid in &self.user.global_roles {
|
||||
if let Some(role) = UserRole::find(*rid).ok().flatten() {
|
||||
global_perms.extend(role.permissions);
|
||||
}
|
||||
}
|
||||
|
||||
if Self::check_global_logic(&global_perms, permission) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if let Some(aid) = account_id {
|
||||
if let Some(role_id) = self.user.account_access_map.get(&aid) {
|
||||
if let Some(role) = UserRole::find(*role_id).ok().flatten() {
|
||||
if role.permissions.contains(&permission.to_string())
|
||||
|| Self::check_account_logic(&role.permissions, permission)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
fn check_global_logic(global: &HashSet<String>, perm: &str) -> bool {
|
||||
if global.contains(perm) {
|
||||
return true;
|
||||
}
|
||||
|
||||
match perm {
|
||||
Permission::DATA_READ => global.contains(Permission::DATA_READ_ALL),
|
||||
Permission::DATA_DELETE => global.contains(Permission::DATA_DELETE_ALL),
|
||||
Permission::DATA_RAW_DOWNLOAD => global.contains(Permission::DATA_RAW_DOWNLOAD_ALL),
|
||||
Permission::DATA_EXPORT_BATCH => global.contains(Permission::DATA_EXPORT_BATCH_ALL),
|
||||
Permission::ACCOUNT_MANAGE | Permission::ACCOUNT_READ_DETAILS => {
|
||||
global.contains(Permission::ACCOUNT_MANAGE_ALL)
|
||||
}
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
fn check_account_logic(scoped_perms: &BTreeSet<String>, perm: &str) -> bool {
|
||||
if scoped_perms.contains(perm) {
|
||||
return true;
|
||||
}
|
||||
match perm {
|
||||
Permission::DATA_READ | Permission::ACCOUNT_READ_DETAILS => {
|
||||
scoped_perms.contains(Permission::ACCOUNT_MANAGE)
|
||||
}
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn require_permission(
|
||||
&self,
|
||||
account_id: Option<u64>,
|
||||
permission: &str,
|
||||
) -> BichonResult<()> {
|
||||
if self.has_permission(account_id, permission) {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(raise_error!(
|
||||
format!("Access Denied: Missing permission '{}'", permission),
|
||||
ErrorCode::Forbidden
|
||||
))
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,29 +16,19 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use super::error::code::ErrorCode;
|
||||
use super::error::BichonError;
|
||||
use mail_parser::{Addr as ImapAddr, Address as ImapAddress};
|
||||
use poem::error::ResponseError;
|
||||
use poem::Body;
|
||||
use poem::{http::StatusCode, Error, Response};
|
||||
use poem_openapi::Object;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::ops::Deref;
|
||||
use tracing::error;
|
||||
|
||||
use mail_parser::{Addr as ImapAddr, Address as ImapAddress};
|
||||
use serde::{Deserialize, Serialize};
|
||||
pub mod auth;
|
||||
pub mod error;
|
||||
pub mod log;
|
||||
pub mod paginated;
|
||||
pub mod periodic;
|
||||
pub mod rustls;
|
||||
pub mod signal;
|
||||
pub mod timeout;
|
||||
pub mod tls;
|
||||
#[cfg(feature = "web-api")]
|
||||
pub mod validator;
|
||||
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Serialize, Deserialize, Object)]
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Serialize, Deserialize)]
|
||||
pub struct Addr {
|
||||
/// The optional display name associated with the email address (e.g., "John Doe").
|
||||
/// If `None`, no display name is specified.
|
||||
@@ -91,62 +81,3 @@ impl<'x> From<&ImapAddress<'x>> for AddrVec {
|
||||
AddrVec(vec)
|
||||
}
|
||||
}
|
||||
|
||||
// #[derive(Serialize)]
|
||||
// pub struct ErrorResponse {
|
||||
// pub message: String,
|
||||
// }
|
||||
|
||||
#[inline]
|
||||
fn create_rust_mailer_error(message: &str, code: ErrorCode) -> BichonError {
|
||||
BichonError::Generic {
|
||||
message: message.into(),
|
||||
location: snafu::Location::default(),
|
||||
code,
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn create_api_error_response(message: &str, code: ErrorCode) -> Error {
|
||||
let rust_mailer_error = create_rust_mailer_error(message, code);
|
||||
rust_mailer_error.into()
|
||||
}
|
||||
|
||||
impl ResponseError for BichonError {
|
||||
fn status(&self) -> StatusCode {
|
||||
match self {
|
||||
BichonError::Generic {
|
||||
message: _,
|
||||
location: _,
|
||||
code,
|
||||
} => code.status(),
|
||||
}
|
||||
}
|
||||
|
||||
fn as_response(&self) -> Response
|
||||
where
|
||||
Self: std::error::Error + Send + Sync + 'static,
|
||||
{
|
||||
match self {
|
||||
BichonError::Generic {
|
||||
message,
|
||||
location,
|
||||
code,
|
||||
} => {
|
||||
error!(
|
||||
error_code = *code as u32,
|
||||
error_message = %message,
|
||||
error_location = ?location
|
||||
);
|
||||
|
||||
let body = Body::from_json(serde_json::json!({
|
||||
"code": *code as u32,
|
||||
"message": message.to_string(),
|
||||
}))
|
||||
.unwrap();
|
||||
|
||||
Response::builder().status(self.status()).body(body)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
260
crates/core/src/common/paginated.rs
Normal file
260
crates/core/src/common/paginated.rs
Normal file
@@ -0,0 +1,260 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::{
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
raise_error,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::cmp::min;
|
||||
|
||||
pub fn paginate_vec<T: Clone>(
|
||||
items: &Vec<T>,
|
||||
page: Option<u64>,
|
||||
page_size: Option<u64>,
|
||||
) -> BichonResult<Paginated<T>> {
|
||||
let total_items = items.len() as u64;
|
||||
|
||||
let (offset, total_pages) = match (page, page_size) {
|
||||
(Some(p), Some(s)) if p > 0 && s > 0 => {
|
||||
let offset = (p - 1) * s;
|
||||
let total_pages = if total_items > 0 {
|
||||
(total_items + s - 1) / s
|
||||
} else {
|
||||
0
|
||||
};
|
||||
(Some(offset), Some(total_pages))
|
||||
}
|
||||
(Some(0), _) | (_, Some(0)) => {
|
||||
return Err(raise_error!(
|
||||
"'page' and 'page_size' must be greater than 0.".into(),
|
||||
ErrorCode::InvalidParameter
|
||||
));
|
||||
}
|
||||
_ => (None, None),
|
||||
};
|
||||
|
||||
let data = match offset {
|
||||
Some(offset) if offset >= total_items => vec![],
|
||||
Some(offset) => {
|
||||
let end = min(offset + page_size.unwrap_or(total_items), total_items) as usize;
|
||||
items[offset as usize..end].to_vec()
|
||||
}
|
||||
None => items.clone(),
|
||||
};
|
||||
|
||||
Ok(Paginated::new(
|
||||
page,
|
||||
page_size,
|
||||
total_items,
|
||||
total_pages,
|
||||
data,
|
||||
))
|
||||
}
|
||||
|
||||
|
||||
#[cfg(not(feature = "web-api"))]
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
pub struct DataPage<S>
|
||||
where
|
||||
S: Serialize + std::fmt::Debug + std::marker::Unpin + Send + Sync,
|
||||
{
|
||||
/// The current page number (starting from 1).
|
||||
pub current_page: Option<u64>,
|
||||
/// The number of items per page.
|
||||
pub page_size: Option<u64>,
|
||||
/// The total number of items across all pages.
|
||||
pub total_items: u64,
|
||||
/// The list of items returned on the current page.
|
||||
pub items: Vec<S>,
|
||||
/// The total number of pages. This is optional and may not be set if not calculated.
|
||||
pub total_pages: Option<u64>,
|
||||
}
|
||||
#[cfg(not(feature = "web-api"))]
|
||||
impl<S: Serialize + std::fmt::Debug + std::marker::Unpin + Send + Sync> From<Paginated<S>>
|
||||
for DataPage<S>
|
||||
{
|
||||
fn from(paginated: Paginated<S>) -> Self {
|
||||
DataPage {
|
||||
current_page: paginated.page,
|
||||
page_size: paginated.page_size,
|
||||
total_items: paginated.total_items,
|
||||
total_pages: paginated.total_pages,
|
||||
items: paginated.items,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
#[cfg(feature = "web-api")]
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, poem_openapi::Object)]
|
||||
pub struct DataPage<S>
|
||||
where
|
||||
S: Serialize
|
||||
+ std::fmt::Debug
|
||||
+ std::marker::Unpin
|
||||
+ Send
|
||||
+ Sync
|
||||
+ poem_openapi::types::Type
|
||||
+ poem_openapi::types::ParseFromJSON
|
||||
+ poem_openapi::types::ToJSON,
|
||||
{
|
||||
/// The current page number (starting from 1).
|
||||
pub current_page: Option<u64>,
|
||||
/// The number of items per page.
|
||||
pub page_size: Option<u64>,
|
||||
/// The total number of items across all pages.
|
||||
pub total_items: u64,
|
||||
/// The list of items returned on the current page.
|
||||
pub items: Vec<S>,
|
||||
/// The total number of pages. This is optional and may not be set if not calculated.
|
||||
pub total_pages: Option<u64>,
|
||||
}
|
||||
|
||||
#[cfg(feature = "web-api")]
|
||||
impl<
|
||||
S: Serialize
|
||||
+ std::fmt::Debug
|
||||
+ std::marker::Unpin
|
||||
+ Send
|
||||
+ Sync
|
||||
+ poem_openapi::types::Type
|
||||
+ poem_openapi::types::ParseFromJSON
|
||||
+ poem_openapi::types::ToJSON,
|
||||
> From<Paginated<S>> for DataPage<S>
|
||||
{
|
||||
fn from(paginated: Paginated<S>) -> Self {
|
||||
DataPage {
|
||||
current_page: paginated.page,
|
||||
page_size: paginated.page_size,
|
||||
total_items: paginated.total_items,
|
||||
total_pages: paginated.total_pages,
|
||||
items: paginated.items,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#[derive(Debug)]
|
||||
pub struct Paginated<T> {
|
||||
pub page: Option<u64>,
|
||||
pub page_size: Option<u64>,
|
||||
pub total_items: u64,
|
||||
pub total_pages: Option<u64>,
|
||||
pub items: Vec<T>,
|
||||
}
|
||||
|
||||
impl<T> Paginated<T> {
|
||||
pub fn new(
|
||||
page: Option<u64>,
|
||||
page_size: Option<u64>,
|
||||
total_items: u64,
|
||||
total_pages: Option<u64>,
|
||||
items: Vec<T>,
|
||||
) -> Self {
|
||||
Paginated {
|
||||
page,
|
||||
page_size,
|
||||
total_items,
|
||||
total_pages,
|
||||
items,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_full_list_without_pagination() {
|
||||
let items: Vec<i32> = (1..=10).collect();
|
||||
let result = paginate_vec(&items, None, None).unwrap();
|
||||
assert_eq!(result.items.len(), 10);
|
||||
assert_eq!(result.total_items, 10);
|
||||
assert_eq!(result.page, None);
|
||||
assert_eq!(result.total_pages, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_first_page() {
|
||||
let items: Vec<i32> = (1..=25).collect();
|
||||
let result = paginate_vec(&items, Some(1), Some(10)).unwrap();
|
||||
assert_eq!(result.items, vec![1, 2, 3, 4, 5, 6, 7, 8, 9, 10]);
|
||||
assert_eq!(result.total_items, 25);
|
||||
assert_eq!(result.total_pages, Some(3));
|
||||
assert_eq!(result.page, Some(1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_last_partial_page() {
|
||||
let items: Vec<i32> = (1..=25).collect();
|
||||
let result = paginate_vec(&items, Some(3), Some(10)).unwrap();
|
||||
assert_eq!(result.items, vec![21, 22, 23, 24, 25]);
|
||||
assert_eq!(result.total_items, 25);
|
||||
assert_eq!(result.total_pages, Some(3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_page_beyond_range_returns_empty() {
|
||||
let items: Vec<i32> = (1..=10).collect();
|
||||
let result = paginate_vec(&items, Some(5), Some(10)).unwrap();
|
||||
assert_eq!(result.items.len(), 0);
|
||||
assert_eq!(result.total_items, 10);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_empty_list() {
|
||||
let items: Vec<i32> = vec![];
|
||||
let result = paginate_vec(&items, Some(1), Some(10)).unwrap();
|
||||
assert_eq!(result.items.len(), 0);
|
||||
assert_eq!(result.total_items, 0);
|
||||
assert_eq!(result.total_pages, Some(0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_zero_page_returns_error() {
|
||||
let items: Vec<i32> = (1..=10).collect();
|
||||
assert!(paginate_vec(&items, Some(0), Some(10)).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_zero_page_size_returns_error() {
|
||||
let items: Vec<i32> = (1..=10).collect();
|
||||
assert!(paginate_vec(&items, Some(1), Some(0)).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_single_item() {
|
||||
let items = vec![42];
|
||||
let result = paginate_vec(&items, Some(1), Some(10)).unwrap();
|
||||
assert_eq!(result.items, vec![42]);
|
||||
assert_eq!(result.total_items, 1);
|
||||
assert_eq!(result.total_pages, Some(1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paginate_vec_exact_page_boundary() {
|
||||
let items: Vec<i32> = (1..=20).collect();
|
||||
let result = paginate_vec(&items, Some(2), Some(10)).unwrap();
|
||||
assert_eq!(result.items, vec![11, 12, 13, 14, 15, 16, 17, 18, 19, 20]);
|
||||
assert_eq!(result.total_pages, Some(2));
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,10 +16,9 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use crate::modules::{common::signal::SIGNAL_MANAGER, error::BichonResult};
|
||||
use crate::{common::signal::SIGNAL_MANAGER, error::BichonResult};
|
||||
use std::{future::Future, time::Duration};
|
||||
use tokio::{sync::oneshot, time::MissedTickBehavior};
|
||||
use tokio::{sync::oneshot, task::JoinHandle, time::MissedTickBehavior};
|
||||
use tracing::{info, warn};
|
||||
|
||||
pub struct PeriodicTask {
|
||||
@@ -28,7 +27,7 @@ pub struct PeriodicTask {
|
||||
|
||||
pub struct TaskHandle {
|
||||
cancel_sender: Option<oneshot::Sender<()>>,
|
||||
join_handle: tokio::task::JoinHandle<()>,
|
||||
join_handle: JoinHandle<()>,
|
||||
}
|
||||
|
||||
impl TaskHandle {
|
||||
@@ -38,6 +37,10 @@ impl TaskHandle {
|
||||
}
|
||||
let _ = self.join_handle.await;
|
||||
}
|
||||
|
||||
pub async fn stop(self) {
|
||||
let _ = self.join_handle.await;
|
||||
}
|
||||
}
|
||||
|
||||
impl PeriodicTask {
|
||||
@@ -82,6 +85,14 @@ impl PeriodicTask {
|
||||
let mut cancel_receiver = cancel_receiver_opt;
|
||||
|
||||
loop {
|
||||
let cancel_fut = async {
|
||||
if let Some(ref mut rx) = cancel_receiver {
|
||||
rx.await.ok();
|
||||
} else {
|
||||
std::future::pending::<()>().await;
|
||||
}
|
||||
};
|
||||
|
||||
tokio::select! {
|
||||
_ = interval.tick() => {
|
||||
match task(param).await {
|
||||
@@ -92,13 +103,7 @@ impl PeriodicTask {
|
||||
}
|
||||
}
|
||||
// only enabled if cancel_receiver is Some
|
||||
_ = async {
|
||||
if let Some(ref mut rx) = cancel_receiver {
|
||||
rx.await.ok()
|
||||
} else {
|
||||
futures::future::pending().await
|
||||
}
|
||||
} => {
|
||||
_ = cancel_fut => {
|
||||
info!("Task '{}' received cancellation signal", name_clone);
|
||||
break;
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,18 +16,17 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use crate::{
|
||||
modules::{
|
||||
{
|
||||
context::Initialize,
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
},
|
||||
raise_error,
|
||||
};
|
||||
|
||||
pub struct RustMailerTls;
|
||||
pub struct BichonTls;
|
||||
|
||||
impl Initialize for RustMailerTls {
|
||||
impl Initialize for BichonTls {
|
||||
async fn initialize() -> BichonResult<()> {
|
||||
rustls::crypto::CryptoProvider::install_default(rustls::crypto::ring::default_provider())
|
||||
.map_err(|_| {
|
||||
@@ -38,4 +37,3 @@ impl Initialize for RustMailerTls {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,12 +16,9 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use std::sync::LazyLock;
|
||||
|
||||
use crate::modules::{
|
||||
context::Initialize, error::BichonResult, utils::shutdown::shutdown_signal,
|
||||
};
|
||||
use crate::{context::Initialize, error::BichonResult, utils::shutdown::shutdown_signal};
|
||||
use tokio::sync::broadcast;
|
||||
|
||||
pub static SIGNAL_MANAGER: LazyLock<SignalManager> = LazyLock::new(SignalManager::new);
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -23,6 +23,7 @@ use std::{
|
||||
};
|
||||
|
||||
use email_address::EmailAddress;
|
||||
|
||||
use poem_openapi::Validator;
|
||||
|
||||
pub struct EmailValidator;
|
||||
@@ -33,6 +34,7 @@ impl Display for EmailValidator {
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
impl Validator<String> for EmailValidator {
|
||||
fn check(&self, value: &String) -> bool {
|
||||
match EmailAddress::from_str(value) {
|
||||
@@ -40,4 +42,4 @@ impl Validator<String> for EmailValidator {
|
||||
Err(_) => false,
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,30 +16,29 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
use crate::modules::{cache::imap::task::SYNC_TASKS, error::BichonResult};
|
||||
use crate::{cache::imap::task::SYNC_TASKS, error::BichonResult};
|
||||
use std::{sync::LazyLock, time::Duration};
|
||||
use tokio::sync::mpsc;
|
||||
use tracing::{error, info};
|
||||
|
||||
pub static SYNC_CONTROLLER: LazyLock<SyncController> = LazyLock::new(SyncController::new);
|
||||
pub static DOWNLOAD_CONTROLLER: LazyLock<DownloadController> =
|
||||
LazyLock::new(DownloadController::new);
|
||||
|
||||
pub struct SyncController {
|
||||
channel: mpsc::Sender<(u64, String)>, // Channel to trigger account sync by account ID
|
||||
pub struct DownloadController {
|
||||
channel: mpsc::Sender<(u64, String)>, // Channel to trigger account download by account ID
|
||||
}
|
||||
|
||||
impl SyncController {
|
||||
impl DownloadController {
|
||||
pub fn new() -> Self {
|
||||
let (tx, mut rx) = mpsc::channel::<(u64, String)>(100);
|
||||
|
||||
tokio::spawn(async move {
|
||||
while let Some((account_id, email)) = rx.recv().await {
|
||||
match Self::start_syncer(account_id, email.clone()).await {
|
||||
Ok(Some(_)) => {}
|
||||
Ok(None) => {}
|
||||
match Self::start_download(account_id, email.clone()).await {
|
||||
Ok(_) => {}
|
||||
Err(err) => {
|
||||
error!(
|
||||
"Failed to prepare and start syncer of account {{{}-{}}}, error: {:#?}",
|
||||
"Failed to prepare and start scheduled download of account {{{}-{}}}, error: {:#?}",
|
||||
&account_id, &email, err
|
||||
);
|
||||
}
|
||||
@@ -47,26 +46,26 @@ impl SyncController {
|
||||
}
|
||||
});
|
||||
|
||||
SyncController { channel: tx }
|
||||
DownloadController { channel: tx }
|
||||
}
|
||||
|
||||
/// Trigger synchronization for a specific account
|
||||
pub async fn trigger_start(&self, account_id: u64, email: String) {
|
||||
pub async fn trigger_schedule(&self, account_id: u64, email: String) {
|
||||
if let Err(e) = self.channel.send((account_id, email)).await {
|
||||
error!(
|
||||
"Failed to trigger synchronization for account={{{}}}, error: {:?}",
|
||||
"Failed to trigger download for account={{{}}}, error: {:?}",
|
||||
account_id, e
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
async fn start_syncer(account_id: u64, email: String) -> BichonResult<Option<()>> {
|
||||
async fn start_download(account_id: u64, email: String) -> BichonResult<()> {
|
||||
info!(
|
||||
"Account syncer starting for account: {}-{}.",
|
||||
"Account download starting for account: {}-{}.",
|
||||
account_id, email
|
||||
);
|
||||
SYNC_TASKS.start_account_sync_task(account_id, email).await;
|
||||
SYNC_TASKS.start_download_task(account_id, email).await;
|
||||
tokio::time::sleep(Duration::from_millis(100)).await;
|
||||
Ok(Some(()))
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
75
crates/core/src/context/executors.rs
Normal file
75
crates/core/src/context/executors.rs
Normal file
@@ -0,0 +1,75 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::account::migration::AccountType;
|
||||
use crate::context::Initialize;
|
||||
use crate::{
|
||||
{
|
||||
account::migration::AccountModel, context::controller::DOWNLOAD_CONTROLLER, error::BichonResult,
|
||||
},
|
||||
utc_now,
|
||||
};
|
||||
use std::sync::LazyLock;
|
||||
use tracing::info;
|
||||
|
||||
pub static BICHON_CONTEXT: LazyLock<BichonContext> = LazyLock::new(BichonContext::new);
|
||||
|
||||
pub struct BichonContext {
|
||||
start_at: i64,
|
||||
}
|
||||
|
||||
impl Initialize for BichonContext {
|
||||
async fn initialize() -> BichonResult<()> {
|
||||
BICHON_CONTEXT.start_account_downloader().await
|
||||
}
|
||||
}
|
||||
|
||||
impl BichonContext {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
start_at: utc_now!(),
|
||||
}
|
||||
}
|
||||
pub fn uptime_ms(&self) -> i64 {
|
||||
utc_now!() - self.start_at
|
||||
}
|
||||
|
||||
pub async fn start_account_downloader(&self) -> BichonResult<()> {
|
||||
let accounts = AccountModel::list_all()?;
|
||||
let active_accounts: Vec<AccountModel> = accounts
|
||||
.into_iter()
|
||||
.filter(|a| a.enabled && matches!(a.account_type, AccountType::IMAP))
|
||||
.collect();
|
||||
|
||||
if active_accounts.is_empty() {
|
||||
info!("No active accounts found for account initialization.");
|
||||
return Ok(());
|
||||
}
|
||||
info!(
|
||||
"System has {} active IMAP accounts to initialize.",
|
||||
active_accounts.len()
|
||||
);
|
||||
for account in active_accounts {
|
||||
DOWNLOAD_CONTROLLER
|
||||
.trigger_schedule(account.id, account.email)
|
||||
.await
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,17 +16,16 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::modules::error::BichonResult;
|
||||
use crate::{common::periodic::TaskHandle, error::BichonResult};
|
||||
|
||||
pub mod controller;
|
||||
pub mod executors;
|
||||
pub mod status;
|
||||
|
||||
#[allow(async_fn_in_trait)]
|
||||
pub trait Initialize {
|
||||
async fn initialize() -> BichonResult<()>;
|
||||
}
|
||||
|
||||
pub trait RustMailTask {
|
||||
fn start();
|
||||
pub trait BichonTask {
|
||||
fn start() -> TaskHandle;
|
||||
}
|
||||
227
crates/core/src/dashboard/mod.rs
Normal file
227
crates/core/src/dashboard/mod.rs
Normal file
@@ -0,0 +1,227 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::{
|
||||
store::tantivy::{
|
||||
attachment::ATTACHMENT_MANAGER,
|
||||
envelope::ENVELOPE_MANAGER,
|
||||
fields::{F_CONTENT_HASH, F_ID},
|
||||
schema::SchemaTools,
|
||||
},
|
||||
users::permissions::Permission,
|
||||
};
|
||||
//use poem_openapi::Object;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::HashSet;
|
||||
use tantivy::{schema::Value, TantivyDocument};
|
||||
|
||||
use crate::{
|
||||
bichon_version, raise_error,
|
||||
{
|
||||
account::migration::AccountModel,
|
||||
common::auth::ClientContext,
|
||||
error::{code::ErrorCode, BichonResult},
|
||||
settings::dir::DATA_DIR_MANAGER,
|
||||
utils::get_total_size,
|
||||
},
|
||||
};
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct DashboardStats {
|
||||
pub account_count: usize, // Number of accounts
|
||||
pub email_count: u64, // Total number of emails
|
||||
pub attachment_count: u64, // Total number of attachments
|
||||
pub total_size_bytes: u64, // Total size of all emails (in bytes)
|
||||
pub storage_usage_bytes: u64, // Actual storage used (in bytes)
|
||||
pub index_usage_bytes: u64, // Index storage size (in bytes)
|
||||
pub recent_activity: Vec<TimeBucket>, // Email activity over recent days
|
||||
pub top_senders: Vec<Group>, // Top 10 senders
|
||||
pub top_accounts: Vec<Group>, // Top 10 accounts
|
||||
pub with_attachment_count: u64, // Emails with attachments
|
||||
pub without_attachment_count: u64, // Emails without attachments
|
||||
pub top_largest_emails: Vec<LargestEmail>, // Top 10 largest emails
|
||||
pub top_largest_attachments: Vec<LargestAttachment>, // Top 10 largest attachments
|
||||
pub system_version: String, // The semantic version string of the currently running backend service
|
||||
}
|
||||
|
||||
impl DashboardStats {
|
||||
pub async fn get(context: ClientContext) -> BichonResult<Self> {
|
||||
let has_all_accounts = context.has_permission(None, Permission::ACCOUNT_MANAGE_ALL);
|
||||
let authorized_ids: Option<HashSet<u64>> = if has_all_accounts {
|
||||
None
|
||||
} else {
|
||||
Some(context.user.account_access_map.keys().cloned().collect())
|
||||
};
|
||||
|
||||
let mut stat = ENVELOPE_MANAGER.get_dashboard_stats(&authorized_ids)?;
|
||||
|
||||
stat.top_largest_emails = ENVELOPE_MANAGER.top_10_largest_emails(&authorized_ids)?;
|
||||
stat.top_largest_attachments =
|
||||
ATTACHMENT_MANAGER.top_10_largest_attachments(&authorized_ids)?;
|
||||
|
||||
stat.account_count = if has_all_accounts {
|
||||
AccountModel::count()?
|
||||
} else {
|
||||
authorized_ids.as_ref().map(|ids| ids.len()).unwrap_or(0)
|
||||
};
|
||||
|
||||
stat.email_count = ENVELOPE_MANAGER.total_emails(&authorized_ids)?;
|
||||
stat.attachment_count = ATTACHMENT_MANAGER.total_attachments(&authorized_ids)?;
|
||||
stat.storage_usage_bytes = get_total_size(&DATA_DIR_MANAGER.storage_dir)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
stat.index_usage_bytes = get_total_size(&&DATA_DIR_MANAGER.envelope_dir)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
stat.system_version = bichon_version!().to_string();
|
||||
|
||||
Ok(stat)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct TimeBucket {
|
||||
pub timestamp_ms: i64, // Timestamp in milliseconds
|
||||
pub count: u64, // Number of emails in this time bucket
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct Group {
|
||||
pub key: String,
|
||||
pub count: u64, // Number of emails from this sender
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct LargestEmail {
|
||||
pub subject: String, // Email subject
|
||||
pub size_bytes: u64, // Email size in bytes
|
||||
pub id: String,
|
||||
}
|
||||
|
||||
impl LargestEmail {
|
||||
pub fn from_tantivy_doc(document: &TantivyDocument) -> BichonResult<Self> {
|
||||
let fields = SchemaTools::email_fields();
|
||||
let value = document.get_first(fields.f_size).ok_or_else(|| {
|
||||
raise_error!(
|
||||
"miss 'size' field in tantivy document".into(),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
let size_bytes = value.as_u64().ok_or_else(|| {
|
||||
raise_error!("'size' field is not a u64".into(), ErrorCode::InternalError)
|
||||
})?;
|
||||
let value = document.get_first(fields.f_subject).ok_or_else(|| {
|
||||
raise_error!("'subject' field not found".into(), ErrorCode::InternalError)
|
||||
})?;
|
||||
let subject = value.as_str().map(|s| s.to_string()).ok_or_else(|| {
|
||||
raise_error!(
|
||||
"'subject' field is not a string".into(),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
|
||||
let value = document.get_first(fields.f_id).ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("'{}' field not found", F_ID),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
let id = value.as_str().map(|s| s.to_string()).ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("'{}' field is not a string", F_ID),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
|
||||
let envelope = LargestEmail {
|
||||
subject,
|
||||
size_bytes,
|
||||
id,
|
||||
};
|
||||
|
||||
Ok(envelope)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
#[cfg_attr(feature = "web-api", derive(poem_openapi::Object))]
|
||||
pub struct LargestAttachment {
|
||||
pub name: String, // Attachment name
|
||||
pub size_bytes: u64, // Attachment size in bytes
|
||||
pub id: String,
|
||||
pub content_hash: String,
|
||||
}
|
||||
|
||||
impl LargestAttachment {
|
||||
pub fn from_tantivy_doc(document: &TantivyDocument) -> BichonResult<Self> {
|
||||
let fields = SchemaTools::attachment_fields();
|
||||
let value = document.get_first(fields.f_size).ok_or_else(|| {
|
||||
raise_error!(
|
||||
"miss 'size' field in tantivy document".into(),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
let size_bytes = value.as_u64().ok_or_else(|| {
|
||||
raise_error!("'size' field is not a u64".into(), ErrorCode::InternalError)
|
||||
})?;
|
||||
let name = document
|
||||
.get_first(fields.f_name_exact)
|
||||
.and_then(|v| v.as_str())
|
||||
.map(|s| s.to_string())
|
||||
.unwrap_or_else(|| "Unknown".to_string());
|
||||
|
||||
let value = document.get_first(fields.f_id).ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("'{}' field not found", F_ID),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
let id = value.as_str().map(|s| s.to_string()).ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("'{}' field is not a string", F_ID),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
|
||||
let value = document.get_first(fields.f_content_hash).ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("'{}' field not found", F_CONTENT_HASH),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
let content_hash = value.as_str().map(|s| s.to_string()).ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!("'{}' field is not a string", F_CONTENT_HASH),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
|
||||
let attachment = LargestAttachment {
|
||||
name,
|
||||
size_bytes,
|
||||
id,
|
||||
content_hash,
|
||||
};
|
||||
|
||||
Ok(attachment)
|
||||
}
|
||||
}
|
||||
60
crates/core/src/database/manager.rs
Normal file
60
crates/core/src/database/manager.rs
Normal file
@@ -0,0 +1,60 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::settings::dir::DATA_DIR_MANAGER;
|
||||
use bichon_memdb::{Durability, MemDb};
|
||||
use std::sync::LazyLock;
|
||||
use std::time::Duration;
|
||||
|
||||
pub static DB_MANAGER: LazyLock<DatabaseManager> = LazyLock::new(DatabaseManager::new);
|
||||
|
||||
pub struct DatabaseManager {
|
||||
db: MemDb,
|
||||
}
|
||||
|
||||
impl DatabaseManager {
|
||||
fn new() -> Self {
|
||||
let db_path = &DATA_DIR_MANAGER.memdb_dir;
|
||||
std::fs::create_dir_all(db_path).expect("Failed to create memdb data directory");
|
||||
|
||||
let db = MemDb::open_with(db_path, Durability::Batch { max_ops: 100 })
|
||||
.expect("Failed to open memdb database");
|
||||
|
||||
// Start periodic snapshot worker (every 5 minutes)
|
||||
db.start_snapshot_worker(Duration::from_secs(300));
|
||||
|
||||
// Start periodic flush worker (every 10 seconds) so buffered writes
|
||||
// are flushed regularly and not only at the batch threshold.
|
||||
db.start_flush_worker(Duration::from_secs(10));
|
||||
|
||||
DatabaseManager { db }
|
||||
}
|
||||
|
||||
/// Get a reference to the MemDb instance.
|
||||
pub fn db(&self) -> &MemDb {
|
||||
&self.db
|
||||
}
|
||||
|
||||
/// Flush any buffered WAL entries to disk. Must be called before shutdown
|
||||
/// to avoid losing writes that haven't hit the batch threshold yet.
|
||||
pub fn flush(&self) {
|
||||
if let Err(e) = self.db.flush() {
|
||||
eprintln!("[memdb] flush error on shutdown: {e}");
|
||||
}
|
||||
}
|
||||
}
|
||||
225
crates/core/src/database/mod.rs
Normal file
225
crates/core/src/database/mod.rs
Normal file
@@ -0,0 +1,225 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::common::paginated::Paginated;
|
||||
use crate::error::code::ErrorCode;
|
||||
use crate::error::BichonResult;
|
||||
use crate::raise_error;
|
||||
use bichon_memdb::{MemDb, Transaction};
|
||||
use serde::de::DeserializeOwned;
|
||||
use serde::Serialize;
|
||||
|
||||
pub mod manager;
|
||||
|
||||
/// Trait for models that can be stored in MemDb collections.
|
||||
pub trait MemDbModel: Serialize + DeserializeOwned + Clone + Send + 'static {
|
||||
/// The collection name this model is stored under.
|
||||
fn collection() -> &'static str;
|
||||
/// The primary key as a string for MemDb storage.
|
||||
fn key(&self) -> String;
|
||||
}
|
||||
|
||||
// ─── Insert ───────────────────────────────────────────────────────────────
|
||||
|
||||
pub fn insert_impl<M: MemDbModel>(db: &MemDb, item: M) -> BichonResult<()> {
|
||||
let coll = db.collection(M::collection());
|
||||
let key = item.key();
|
||||
coll.insert(key, &item)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
|
||||
pub fn batch_insert_impl<M: MemDbModel>(db: &MemDb, items: Vec<M>) -> BichonResult<()> {
|
||||
let txn = db.transaction();
|
||||
let mut txn = txn;
|
||||
for item in &items {
|
||||
txn = txn
|
||||
.insert(M::collection(), item.key(), item)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
|
||||
// ─── Upsert ────────────────────────────────────────────────────────────────
|
||||
|
||||
pub fn upsert_impl<M: MemDbModel>(db: &MemDb, item: M) -> BichonResult<()> {
|
||||
let coll = db.collection(M::collection());
|
||||
coll.upsert(item.key(), &item)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
|
||||
pub fn batch_upsert_impl<M: MemDbModel>(db: &MemDb, items: Vec<M>) -> BichonResult<()> {
|
||||
let txn = db.transaction();
|
||||
let mut txn = txn;
|
||||
for item in &items {
|
||||
txn = txn
|
||||
.upsert(M::collection(), item.key(), item)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
|
||||
// ─── Find ──────────────────────────────────────────────────────────────────
|
||||
|
||||
pub fn find_impl<M: MemDbModel>(db: &MemDb, key: &str) -> BichonResult<Option<M>> {
|
||||
let coll = db.collection(M::collection());
|
||||
coll.get(key)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
|
||||
// ─── Filter (replaces secondary key queries) ──────────────────────────────
|
||||
|
||||
pub fn filter_impl<M, F>(db: &MemDb, predicate: F) -> BichonResult<Vec<M>>
|
||||
where
|
||||
M: MemDbModel,
|
||||
F: Fn(&M) -> bool + Send + 'static,
|
||||
{
|
||||
let coll = db.collection(M::collection());
|
||||
coll.filter(predicate)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
|
||||
// ─── Update (read-modify-write under a single spawn_blocking) ─────────────
|
||||
|
||||
pub fn update_impl<M: MemDbModel>(
|
||||
db: &MemDb,
|
||||
key: &str,
|
||||
update_fn: impl FnOnce(M) -> BichonResult<M> + Send + 'static,
|
||||
) -> BichonResult<M> {
|
||||
let coll = db.collection(M::collection());
|
||||
let current: M = coll
|
||||
.get_required(key)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
let updated = update_fn(current)?;
|
||||
coll.upsert(key, &updated)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
Ok(updated)
|
||||
}
|
||||
|
||||
// ─── Delete ────────────────────────────────────────────────────────────────
|
||||
|
||||
pub fn delete_impl<M: MemDbModel>(db: &MemDb, key: &str) -> BichonResult<()> {
|
||||
let coll = db.collection(M::collection());
|
||||
let existed = coll
|
||||
.delete(key)
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
if !existed {
|
||||
return Err(raise_error!(
|
||||
format!("{} '{}' not found for deletion", M::collection(), key),
|
||||
ErrorCode::ResourceNotFound
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn batch_delete_impl<M: MemDbModel>(db: &MemDb, keys: Vec<String>) -> BichonResult<usize> {
|
||||
let txn = db.transaction();
|
||||
let mut txn = txn;
|
||||
let mut count = 0usize;
|
||||
for key in &keys {
|
||||
txn = txn.delete(M::collection(), key.clone());
|
||||
count += 1;
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
Ok(count)
|
||||
}
|
||||
|
||||
// ─── List / Count ──────────────────────────────────────────────────────────
|
||||
|
||||
pub fn list_all_impl<M: MemDbModel>(db: &MemDb) -> BichonResult<Vec<M>> {
|
||||
let coll = db.collection(M::collection());
|
||||
coll.list_all()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
|
||||
pub fn count_impl<M: MemDbModel>(db: &MemDb) -> BichonResult<usize> {
|
||||
let coll = db.collection(M::collection());
|
||||
Ok(coll.count())
|
||||
}
|
||||
|
||||
// ─── Paginate ──────────────────────────────────────────────────────────────
|
||||
|
||||
pub fn paginate_impl<M: MemDbModel>(
|
||||
db: &MemDb,
|
||||
page: Option<u64>,
|
||||
page_size: Option<u64>,
|
||||
desc: Option<bool>,
|
||||
) -> BichonResult<Paginated<M>> {
|
||||
let coll = db.collection(M::collection());
|
||||
let total_items = coll.count() as u64;
|
||||
|
||||
let (offset, total_pages) = match (page, page_size) {
|
||||
(Some(p), Some(s)) if p > 0 && s > 0 => {
|
||||
let offset = (p - 1) * s;
|
||||
let total_pages = if total_items > 0 {
|
||||
(total_items as f64 / s as f64).ceil() as u64
|
||||
} else {
|
||||
0
|
||||
};
|
||||
(Some(offset), Some(total_pages))
|
||||
}
|
||||
(Some(0), _) | (_, Some(0)) => {
|
||||
return Err(raise_error!(
|
||||
"'page' and 'page_size' must be greater than 0.".into(),
|
||||
ErrorCode::InvalidParameter
|
||||
));
|
||||
}
|
||||
_ => (None, None),
|
||||
};
|
||||
|
||||
let all: Vec<M> = coll
|
||||
.list_all()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))?;
|
||||
|
||||
let items: Vec<M> = match desc {
|
||||
Some(true) => {
|
||||
let iter: Vec<M> = all.into_iter().rev().collect();
|
||||
let skip = offset.unwrap_or(0) as usize;
|
||||
let take = page_size.unwrap_or(total_items) as usize;
|
||||
iter.into_iter().skip(skip).take(take).collect()
|
||||
}
|
||||
_ => {
|
||||
let skip = offset.unwrap_or(0) as usize;
|
||||
let take = page_size.unwrap_or(total_items) as usize;
|
||||
all.into_iter().skip(skip).take(take).collect()
|
||||
}
|
||||
};
|
||||
|
||||
Ok(Paginated::new(
|
||||
page,
|
||||
page_size,
|
||||
total_items,
|
||||
total_pages,
|
||||
items,
|
||||
))
|
||||
}
|
||||
|
||||
// ─── Transaction ───────────────────────────────────────────────────────────
|
||||
|
||||
/// Execute operations within a single atomic transaction (one WAL entry).
|
||||
pub fn with_transaction(
|
||||
db: &MemDb,
|
||||
f: impl FnOnce(Transaction) -> BichonResult<Transaction> + Send + 'static,
|
||||
) -> BichonResult<()> {
|
||||
let txn = db.transaction();
|
||||
let txn = f(txn)?;
|
||||
txn.commit()
|
||||
.map_err(|e| raise_error!(format!("{:#?}", e), ErrorCode::InternalError))
|
||||
}
|
||||
902
crates/core/src/envelope/extractor.rs
Normal file
902
crates/core/src/envelope/extractor.rs
Normal file
@@ -0,0 +1,902 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::account::migration::AccountModel;
|
||||
use crate::cache::imap::mailbox::MailBox;
|
||||
use crate::common::AddrVec;
|
||||
use crate::envelope::meta::parse_bichon_metadata;
|
||||
use crate::envelope::utils::normalize_subject;
|
||||
use crate::error::code::ErrorCode;
|
||||
use crate::error::BichonResult;
|
||||
use crate::imap::executor::ImapExecutor;
|
||||
use crate::message::content::AttachmentInfo;
|
||||
use crate::store::blob::{DetachedEmail, BLOB_MANAGER};
|
||||
use crate::store::tantivy::attachment::ATTACHMENT_MANAGER;
|
||||
use crate::store::tantivy::dedup_cache::DEDUP_CACHE;
|
||||
use crate::store::tantivy::envelope::ENVELOPE_MANAGER;
|
||||
use crate::store::tantivy::model::{AttachmentModel, EnvelopeWithAttachments};
|
||||
use crate::utils::html::extract_text;
|
||||
use crate::utils::{compute_content_hash, hex_hash};
|
||||
use crate::{id, store::envelope::Envelope};
|
||||
use crate::{raise_error, utc_now};
|
||||
use async_imap::types::Fetch;
|
||||
use bytes::Bytes;
|
||||
use mail_parser::{Address, HeaderName, Message, MessageParser, MimeHeaders};
|
||||
use tantivy::TantivyDocument;
|
||||
use tantivy::schema::Facet;
|
||||
use tracing::error;
|
||||
use uuid::Uuid;
|
||||
|
||||
pub async fn extract_envelope_and_store_it(
|
||||
fetch: Fetch,
|
||||
account_id: u64,
|
||||
mailbox_id: u64,
|
||||
) -> BichonResult<()> {
|
||||
let internal_date = fetch
|
||||
.internal_date()
|
||||
.map(|d| d.timestamp_millis())
|
||||
.unwrap_or(0);
|
||||
let uid = fetch.uid.unwrap_or(0);
|
||||
let body = match fetch.body() {
|
||||
Some(b) => b,
|
||||
None => {
|
||||
tracing::warn!(
|
||||
account_id,
|
||||
uid = fetch.uid,
|
||||
"FETCH response has no body, skipping message"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
let size = fetch.size.unwrap_or(body.len() as u32);
|
||||
extract_envelope_core(body, uid, size, internal_date, account_id, mailbox_id).await
|
||||
}
|
||||
|
||||
pub async fn extract_envelope_from_eml(
|
||||
body: &[u8],
|
||||
account_id: u64,
|
||||
mailbox_id: u64,
|
||||
) -> BichonResult<()> {
|
||||
extract_envelope_core(body, 0, body.len() as u32, 0, account_id, mailbox_id).await
|
||||
}
|
||||
|
||||
pub async fn extract_envelope_from_smtp(
|
||||
body: &[u8],
|
||||
account_id: u64,
|
||||
mailbox_id: u64,
|
||||
) -> BichonResult<()> {
|
||||
extract_envelope_core(
|
||||
body,
|
||||
0,
|
||||
body.len() as u32,
|
||||
utc_now!(),
|
||||
account_id,
|
||||
mailbox_id,
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn extract_envelope_core(
|
||||
body: &[u8],
|
||||
uid: u32,
|
||||
size: u32,
|
||||
internal_date: i64,
|
||||
account_id: u64,
|
||||
mailbox_id: u64,
|
||||
) -> BichonResult<()> {
|
||||
//The content hash of the original raw EML
|
||||
let email_content_hash = compute_content_hash(body);
|
||||
if DEDUP_CACHE.contains(account_id, mailbox_id, &email_content_hash) {
|
||||
tracing::debug!("Duplicate email detected");
|
||||
//println!("Duplicate email detected");
|
||||
return Ok(());
|
||||
}
|
||||
let message: Message<'_> = MessageParser::new().parse(body).ok_or_else(|| {
|
||||
raise_error!(
|
||||
"Email header parse result is not available".into(),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
|
||||
if let Ok(account) = AccountModel::get(account_id) {
|
||||
if let Some(ref rules) = account.archive_rules {
|
||||
let sender = message.from().and_then(|addr| {
|
||||
AddrVec::from(addr).0.into_iter().next().and_then(|a| a.address)
|
||||
});
|
||||
let subject = message.subject().map(|s| s.to_string());
|
||||
|
||||
let is_spam = !rules.spam_headers.is_empty()
|
||||
&& rules.spam_headers.iter().any(|h| {
|
||||
message
|
||||
.header_raw(h.clone())
|
||||
.map(|v| matches!(v.trim().to_lowercase().as_str(), "yes" | "true"))
|
||||
.unwrap_or(false)
|
||||
});
|
||||
|
||||
if !rules.should_archive(sender.as_deref(), subject.as_deref(), size, is_spam) {
|
||||
tracing::debug!(
|
||||
account_id,
|
||||
uid,
|
||||
sender = sender.as_deref().unwrap_or("?"),
|
||||
subject = subject.as_deref().unwrap_or("?"),
|
||||
"Email filtered out by archive rules"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let preview_limit = 100;
|
||||
let text = if let Some(text) = message.body_text(0).map(|cow| cow.into_owned()) {
|
||||
text
|
||||
} else if let Some(html) = message.body_html(0).map(|cow| cow.into_owned()) {
|
||||
extract_text(html)
|
||||
} else {
|
||||
String::new()
|
||||
};
|
||||
|
||||
let text = text.split_whitespace().collect::<Vec<_>>().join(" ");
|
||||
|
||||
let preview = if text.chars().count() > preview_limit {
|
||||
text.chars().take(preview_limit).collect::<String>() + "..."
|
||||
} else {
|
||||
text.clone()
|
||||
};
|
||||
|
||||
let body_text = text;
|
||||
|
||||
let message_id = message
|
||||
.message_id()
|
||||
.map(String::from)
|
||||
.unwrap_or_else(generate_message_id);
|
||||
|
||||
let in_reply_to = message.in_reply_to().as_text().map(String::from);
|
||||
let references = extract_references(&message);
|
||||
let thread_id = compute_thread_id(in_reply_to, references, &message_id);
|
||||
|
||||
let mut subject = message.subject().map(String::from).unwrap_or_default();
|
||||
if subject.contains('\u{FFFD}') {
|
||||
subject = normalize_subject(message.header_raw(HeaderName::Subject));
|
||||
}
|
||||
|
||||
let date = message.date().map(|d| d.to_timestamp() * 1000).unwrap_or(0);
|
||||
let internal_date = if internal_date == 0 {
|
||||
date
|
||||
} else {
|
||||
internal_date
|
||||
};
|
||||
let parse_addrs = |addrs: Option<&Address<'_>>| {
|
||||
addrs
|
||||
.map(|addr| {
|
||||
AddrVec::from(addr)
|
||||
.0
|
||||
.into_iter()
|
||||
.filter_map(|a| a.address)
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default()
|
||||
};
|
||||
|
||||
let bcc = parse_addrs(message.bcc());
|
||||
let cc = parse_addrs(message.cc());
|
||||
let to = parse_addrs(message.to());
|
||||
|
||||
let from = message
|
||||
.from()
|
||||
.and_then(|addr| AddrVec::from(addr).0.into_iter().next())
|
||||
.and_then(|add| add.address)
|
||||
.unwrap_or_else(|| "unknown".to_string());
|
||||
let attachment_count = message.attachment_count();
|
||||
let attachments = detach_and_store_attachments(body, &message, &email_content_hash, account_id, mailbox_id).await;
|
||||
|
||||
let envelope_id = Uuid::new_v4().to_string();
|
||||
let now = utc_now!();
|
||||
|
||||
|
||||
let mut final_tags = Vec::new();
|
||||
|
||||
if let Some(meta_header) = message.header_raw("X-Bichon-Metadata") {
|
||||
if let Some(bmd) = parse_bichon_metadata(meta_header) {
|
||||
if let Some(tags) = bmd.tags {
|
||||
let validated_tags: Result<Vec<String>, _> = tags
|
||||
.iter()
|
||||
.map(|tag| {
|
||||
Facet::from_text(tag)
|
||||
.map(|_| tag.clone())
|
||||
.map_err(|e| e)
|
||||
})
|
||||
.collect();
|
||||
|
||||
match validated_tags {
|
||||
Ok(valid_list) => {
|
||||
final_tags = valid_list;
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
"Tag validation failed, ignoring all tags: {:#?}",
|
||||
e
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let attachment_docs: Vec<TantivyDocument> = attachments
|
||||
.iter()
|
||||
.filter(|a| !a.inline || a.content_id.is_none())
|
||||
.map(|a| {
|
||||
let has_text = a.extracted_text.is_some();
|
||||
AttachmentModel {
|
||||
id: Uuid::new_v4().to_string(),
|
||||
envelope_id: envelope_id.clone(),
|
||||
account_id,
|
||||
account_email: None,
|
||||
mailbox_id,
|
||||
mailbox_name: None,
|
||||
subject: subject.clone(),
|
||||
content_hash: a.content_hash.clone(),
|
||||
from: from.clone(),
|
||||
date,
|
||||
ingest_at: now,
|
||||
size: a.size as u64,
|
||||
ext: a.get_extension(),
|
||||
category: a.get_category().to_string(),
|
||||
content_type: a.file_type.clone(),
|
||||
shard_id: 0,
|
||||
text: a.extracted_text.clone(),
|
||||
has_text,
|
||||
is_ocr: a.extracted_is_ocr,
|
||||
page_count: a.extracted_page_count.map(|n| n as u64),
|
||||
is_indexed: has_text,
|
||||
is_message: a.is_message,
|
||||
name: a.filename.clone(),
|
||||
tags: None,
|
||||
auto_tags: None,
|
||||
}
|
||||
})
|
||||
.map(|a| a.into_document())
|
||||
.collect();
|
||||
|
||||
let envelope = Envelope {
|
||||
id: envelope_id,
|
||||
message_id,
|
||||
account_id,
|
||||
mailbox_id,
|
||||
uid,
|
||||
subject,
|
||||
preview,
|
||||
from,
|
||||
to,
|
||||
cc,
|
||||
bcc,
|
||||
date,
|
||||
internal_date,
|
||||
ingest_at: now,
|
||||
size,
|
||||
thread_id,
|
||||
attachment_count,
|
||||
regular_attachment_count: attachment_docs.len(),
|
||||
tags: (!final_tags.is_empty()).then_some(final_tags),
|
||||
account_email: None,
|
||||
mailbox_name: None,
|
||||
content_hash: email_content_hash.clone(),
|
||||
account_name: None,
|
||||
};
|
||||
// 'attachments' contains both regular and inline attachments
|
||||
let ea = EnvelopeWithAttachments {
|
||||
envelope,
|
||||
attachments: Some(attachments),
|
||||
};
|
||||
let doc = ea.to_document(&body_text, 0)?;
|
||||
tracing::debug!(
|
||||
"[account {}][mailbox {}] extract: uid={} msg_id={} content_hash={}",
|
||||
account_id,
|
||||
mailbox_id,
|
||||
uid,
|
||||
&ea.envelope.message_id,
|
||||
&ea.envelope.content_hash,
|
||||
);
|
||||
ENVELOPE_MANAGER.queue(doc).await;
|
||||
DEDUP_CACHE.insert(account_id, mailbox_id, &email_content_hash);
|
||||
for doc in attachment_docs {
|
||||
ATTACHMENT_MANAGER.queue(doc).await;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn extract_envelope_from_nested_message(
|
||||
message: Message<'_>,
|
||||
account_id: u64,
|
||||
) -> BichonResult<Envelope> {
|
||||
let text = if let Some(text) = message.body_text(0).map(|cow| cow.into_owned()) {
|
||||
text
|
||||
} else if let Some(html) = message.body_html(0).map(|cow| cow.into_owned()) {
|
||||
extract_text(html)
|
||||
} else {
|
||||
String::new()
|
||||
};
|
||||
|
||||
let message_id = message
|
||||
.message_id()
|
||||
.map(String::from)
|
||||
.unwrap_or_else(generate_message_id);
|
||||
|
||||
let in_reply_to = message.in_reply_to().as_text().map(String::from);
|
||||
let references = extract_references(&message);
|
||||
let thread_id = compute_thread_id(in_reply_to, references, &message_id);
|
||||
|
||||
let mut subject = message.subject().map(String::from).unwrap_or_default();
|
||||
if subject.contains('\u{FFFD}') {
|
||||
subject = normalize_subject(message.header_raw(HeaderName::Subject));
|
||||
}
|
||||
|
||||
let date = message.date().map(|d| d.to_timestamp() * 1000).unwrap_or(0);
|
||||
|
||||
let parse_addrs = |addrs: Option<&Address<'_>>| {
|
||||
addrs
|
||||
.map(|addr| {
|
||||
AddrVec::from(addr)
|
||||
.0
|
||||
.into_iter()
|
||||
.filter_map(|a| a.address)
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default()
|
||||
};
|
||||
|
||||
let bcc = parse_addrs(message.bcc());
|
||||
let cc = parse_addrs(message.cc());
|
||||
let to = parse_addrs(message.to());
|
||||
|
||||
let from = message
|
||||
.from()
|
||||
.and_then(|addr| AddrVec::from(addr).0.into_iter().next())
|
||||
.and_then(|add| add.address)
|
||||
.unwrap_or_else(|| "unknown".to_string());
|
||||
|
||||
let envelope = Envelope {
|
||||
id: Default::default(),
|
||||
message_id,
|
||||
account_id,
|
||||
mailbox_id: Default::default(),
|
||||
uid: Default::default(),
|
||||
subject,
|
||||
preview: text,
|
||||
from,
|
||||
to,
|
||||
cc,
|
||||
bcc,
|
||||
date,
|
||||
internal_date: Default::default(),
|
||||
ingest_at: Default::default(),
|
||||
size: Default::default(),
|
||||
thread_id,
|
||||
attachment_count: Default::default(),
|
||||
regular_attachment_count: Default::default(),
|
||||
tags: Default::default(),
|
||||
account_email: Default::default(),
|
||||
account_name: Default::default(),
|
||||
mailbox_name: Default::default(),
|
||||
content_hash: Default::default(),
|
||||
};
|
||||
|
||||
Ok(envelope)
|
||||
}
|
||||
|
||||
pub fn compute_thread_id(
|
||||
in_reply_to: Option<String>,
|
||||
references: Option<Vec<String>>,
|
||||
message_id: &str,
|
||||
) -> String {
|
||||
if in_reply_to.is_some() && references.as_ref().map_or(false, |r| !r.is_empty()) {
|
||||
return hex_hash(&references.as_ref().unwrap()[0]);
|
||||
}
|
||||
hex_hash(message_id)
|
||||
}
|
||||
|
||||
pub fn generate_message_id() -> String {
|
||||
let ts = utc_now!();
|
||||
let pid = std::process::id();
|
||||
format!("<{:016x}.{}.{}@{}>", id!(128), ts, pid, "bichon")
|
||||
}
|
||||
|
||||
pub fn extract_references(message: &Message<'_>) -> Option<Vec<String>> {
|
||||
match message.references() {
|
||||
mail_parser::HeaderValue::Text(cow) => Some(vec![cow.to_string()]),
|
||||
mail_parser::HeaderValue::TextList(vec) => {
|
||||
Some(vec.iter().map(|cow| cow.to_string()).collect())
|
||||
}
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn detach_and_store_attachments(
|
||||
original_body: &[u8],
|
||||
message: &Message<'_>,
|
||||
eml_content_hash: &str,
|
||||
account_id: u64,
|
||||
mailbox_id: u64,
|
||||
) -> Vec<AttachmentInfo> {
|
||||
let rules = if account_id > 0 {
|
||||
AccountModel::get(account_id)
|
||||
.ok()
|
||||
.and_then(|a| a.extraction_rules)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let mailbox_name = match rules.as_ref().map(|r| !r.folders.is_empty()) {
|
||||
Some(true) => MailBox::get(mailbox_id).ok().map(|mb| mb.name),
|
||||
_ => None,
|
||||
};
|
||||
|
||||
let sender = message
|
||||
.from()
|
||||
.and_then(|addr| AddrVec::from(addr).0.into_iter().next())
|
||||
.and_then(|add| add.address);
|
||||
|
||||
let mut stripped_eml = original_body.to_vec();
|
||||
let mut attachment_infos = Vec::new();
|
||||
// Step 1: Collect and sort attachment ranges in reverse to maintain offset integrity
|
||||
let mut ranges: Vec<_> = message
|
||||
.attachments()
|
||||
.map(|att| {
|
||||
(
|
||||
att.raw_body_offset() as usize,
|
||||
att.raw_end_offset() as usize,
|
||||
att,
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
|
||||
ranges.sort_by(|a, b| b.0.cmp(&a.0));
|
||||
let mut attachments = Vec::with_capacity(ranges.len());
|
||||
|
||||
// Collect candidates for text extraction (non-inline, known document types).
|
||||
struct TextCandidate {
|
||||
content_hash: String,
|
||||
file_type: String,
|
||||
ext: String,
|
||||
bytes: Vec<u8>,
|
||||
}
|
||||
let mut text_candidates: Vec<TextCandidate> = Vec::new();
|
||||
|
||||
for (raw_start, raw_end, att) in ranges {
|
||||
// mail-parser may report attachment offsets past the body end for
|
||||
// malformed messages; clamp the range to avoid a slice panic.
|
||||
let body_len = original_body.len();
|
||||
let raw_start = raw_start.min(body_len);
|
||||
let raw_end = raw_end.min(body_len);
|
||||
let range_valid = raw_start < raw_end;
|
||||
|
||||
// content hash is computed from the decoded attachment contents,
|
||||
// which is always available regardless of raw offset validity.
|
||||
let content_hash = compute_content_hash(att.contents());
|
||||
|
||||
if range_valid {
|
||||
let raw_bytes = &original_body[raw_start..raw_end];
|
||||
// The actual content stored in the blob is the raw undecoded data.
|
||||
attachments.push((content_hash.clone(), Bytes::copy_from_slice(raw_bytes)));
|
||||
|
||||
// Replace raw attachment content with a hash-based placeholder
|
||||
let placeholder = format!("<<BICHON_DETACH_HASH:{}>>", &content_hash);
|
||||
stripped_eml.splice(raw_start..raw_end, placeholder.as_bytes().iter().cloned());
|
||||
} else {
|
||||
// Invalid range: store a zero-length blob so the consistency
|
||||
// check passes; reattachment will log a warning for the missing
|
||||
// blob data but won't panic.
|
||||
attachments.push((content_hash.clone(), Bytes::new()));
|
||||
}
|
||||
|
||||
let inline = att
|
||||
.content_disposition()
|
||||
.map(|d| d.is_inline())
|
||||
.unwrap_or_else(|| att.content_id().is_some());
|
||||
let file_type = att
|
||||
.content_type()
|
||||
.map(|ct| {
|
||||
format!(
|
||||
"{}/{}",
|
||||
ct.c_type.as_ref(),
|
||||
ct.c_subtype.as_deref().unwrap_or("")
|
||||
)
|
||||
})
|
||||
.unwrap_or_else(|| "application/octet-stream".to_string());
|
||||
let has_cid = att.content_id().is_some();
|
||||
let att_name = att.attachment_name().map(|n| n.to_string());
|
||||
let ext = att_name
|
||||
.as_deref()
|
||||
.and_then(|n| {
|
||||
std::path::Path::new(n)
|
||||
.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.map(|s| s.to_ascii_lowercase())
|
||||
})
|
||||
.unwrap_or_default();
|
||||
|
||||
let should_extract = rules.as_ref().map_or(true, |r| {
|
||||
r.should_extract(
|
||||
&ext,
|
||||
mailbox_name.as_deref(),
|
||||
att_name.as_deref(),
|
||||
sender.as_deref(),
|
||||
)
|
||||
});
|
||||
|
||||
if !inline || !has_cid {
|
||||
let decoded_len = att.contents().len();
|
||||
if should_extract
|
||||
&& decoded_len <= crate::ext::text_extractor::MAX_EXTRACT_BYTES
|
||||
&& crate::ext::text_extractor::should_try_extract(&file_type, &ext)
|
||||
{
|
||||
text_candidates.push(TextCandidate {
|
||||
content_hash: content_hash.clone(),
|
||||
file_type: file_type.clone(),
|
||||
ext: ext.clone(),
|
||||
bytes: att.contents().to_vec(),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
let info = AttachmentInfo {
|
||||
filename: att.attachment_name().map(|n| n.to_string()),
|
||||
size: att.contents().len(),
|
||||
inline,
|
||||
file_type,
|
||||
content_id: att.content_id().map(|id| id.to_string()),
|
||||
content_hash: content_hash.clone(),
|
||||
is_message: att.is_message(),
|
||||
extracted_text: None,
|
||||
extracted_page_count: None,
|
||||
extracted_is_ocr: false,
|
||||
};
|
||||
|
||||
attachment_infos.push(info);
|
||||
}
|
||||
|
||||
// Run text extraction in a single spawn_blocking batch.
|
||||
if !text_candidates.is_empty() {
|
||||
if let Ok(mut extracted_map) = tokio::task::spawn_blocking(move || {
|
||||
let mut map: std::collections::HashMap<
|
||||
String,
|
||||
(String, Option<u32>, bool),
|
||||
> = std::collections::HashMap::new();
|
||||
for c in text_candidates {
|
||||
if let Some(r) =
|
||||
crate::ext::text_extractor::extract_text(&c.file_type, &c.ext, &c.bytes)
|
||||
{
|
||||
map.insert(c.content_hash, (r.text, r.page_count, r.is_ocr));
|
||||
}
|
||||
}
|
||||
map
|
||||
})
|
||||
.await
|
||||
{
|
||||
for info in &mut attachment_infos {
|
||||
if let Some((text, pages, is_ocr)) = extracted_map.remove(&info.content_hash) {
|
||||
info.extracted_text = Some(text);
|
||||
info.extracted_page_count = pages;
|
||||
info.extracted_is_ocr = is_ocr;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Step 4: Store the final stripped EML content
|
||||
BLOB_MANAGER
|
||||
.queue(DetachedEmail {
|
||||
email: (eml_content_hash.to_string(), Bytes::from(stripped_eml)),
|
||||
attachments: Some(attachments),
|
||||
})
|
||||
.await;
|
||||
|
||||
attachment_infos
|
||||
}
|
||||
|
||||
pub fn reattach_eml_content(
|
||||
account_id: u64,
|
||||
envelope_id: String,
|
||||
) -> BichonResult<(Envelope, Bytes)> {
|
||||
let e = ENVELOPE_MANAGER
|
||||
.get_envelope_by_id(account_id, &envelope_id)
|
||||
?
|
||||
.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!(
|
||||
"Envelope not found: account_id={} envelope_id={}",
|
||||
account_id, &envelope_id
|
||||
),
|
||||
ErrorCode::ResourceNotFound
|
||||
)
|
||||
})?;
|
||||
|
||||
let restored_eml = BLOB_MANAGER
|
||||
.get_email(&e.envelope.content_hash)?
|
||||
.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!(
|
||||
"Original email content not found: account_id={} envelope_id={} content_hash={}",
|
||||
account_id, &envelope_id, &e.envelope.content_hash
|
||||
),
|
||||
ErrorCode::ResourceNotFound
|
||||
)
|
||||
})?;
|
||||
|
||||
if !e.envelope.has_any_attachments() {
|
||||
return Ok((e.envelope, restored_eml));
|
||||
}
|
||||
|
||||
let mut restored_eml = restored_eml.to_vec();
|
||||
let actual_count = e.attachments.as_ref().map(|a| a.len()).unwrap_or(0);
|
||||
if e.envelope.attachment_count != actual_count {
|
||||
return Err(raise_error!(
|
||||
format!(
|
||||
"Consistency check failed: envelope.attachment_count ({}) does not match attachments.len ({})",
|
||||
e.envelope.attachment_count,
|
||||
actual_count
|
||||
),
|
||||
ErrorCode::InternalError
|
||||
));
|
||||
}
|
||||
|
||||
let mut tasks = Vec::new();
|
||||
for detail in e.attachments.unwrap() {
|
||||
let placeholder_str = format!("<<BICHON_DETACH_HASH:{}>>", &detail.content_hash);
|
||||
let pattern = placeholder_str.as_bytes();
|
||||
let pattern_len = pattern.len();
|
||||
|
||||
let mut search_cursor = 0;
|
||||
while let Some(pos) = restored_eml[search_cursor..]
|
||||
.windows(pattern_len)
|
||||
.position(|window| window == pattern)
|
||||
{
|
||||
let absolute_start = search_cursor + pos;
|
||||
let absolute_end = absolute_start + pattern_len;
|
||||
|
||||
tasks.push((
|
||||
absolute_start,
|
||||
absolute_end,
|
||||
detail.content_hash.clone(),
|
||||
));
|
||||
search_cursor = absolute_end;
|
||||
}
|
||||
}
|
||||
|
||||
tasks.sort_by(|a, b| b.0.cmp(&a.0));
|
||||
|
||||
for (start, end, hash) in tasks {
|
||||
if let Some(original_data) = BLOB_MANAGER.get_attachment(&hash)? {
|
||||
restored_eml.splice(start..end, original_data.iter().cloned());
|
||||
} else {
|
||||
error!("[ERROR] Missing attachment blob for hash: {}", hash);
|
||||
}
|
||||
}
|
||||
|
||||
Ok((e.envelope, Bytes::from(restored_eml)))
|
||||
}
|
||||
|
||||
/// Returns the raw EML for an indexed message, self-healing a missing content blob.
|
||||
///
|
||||
/// Behaves like [`reattach_eml_content`], but when the message's content blob is
|
||||
/// absent from the blob store it fetches that single message on demand from the
|
||||
/// IMAP server (`UID FETCH <uid> (BODY.PEEK[])`), persists it for future requests,
|
||||
/// and returns it. If the on-demand fetch itself fails, the original "content not
|
||||
/// found" error from [`reattach_eml_content`] is surfaced unchanged so the caller
|
||||
/// still produces its 404.
|
||||
pub async fn reattach_eml_content_self_healing(
|
||||
account_id: u64,
|
||||
envelope_id: String,
|
||||
) -> BichonResult<(Envelope, Bytes)> {
|
||||
let envelope = ENVELOPE_MANAGER
|
||||
.get_envelope_by_id(account_id, &envelope_id)?
|
||||
.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!(
|
||||
"Envelope not found: account_id={} envelope_id={}",
|
||||
account_id, &envelope_id
|
||||
),
|
||||
ErrorCode::ResourceNotFound
|
||||
)
|
||||
})?
|
||||
.envelope;
|
||||
|
||||
// Fast path: the content blob is present, reuse the regular reattach logic.
|
||||
if BLOB_MANAGER.get_email(&envelope.content_hash)?.is_some() {
|
||||
return reattach_eml_content(account_id, envelope_id);
|
||||
}
|
||||
|
||||
// The blob is missing. Try to recover it directly from the IMAP server.
|
||||
match recover_message_blob(&envelope).await {
|
||||
Ok(raw_body) => {
|
||||
tracing::info!(
|
||||
account_id,
|
||||
envelope_id = %envelope_id,
|
||||
uid = envelope.uid,
|
||||
"Self-healed missing email content blob via on-demand IMAP fetch"
|
||||
);
|
||||
Ok((envelope, raw_body))
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!(
|
||||
account_id,
|
||||
envelope_id = %envelope_id,
|
||||
uid = envelope.uid,
|
||||
error = %e,
|
||||
"On-demand IMAP fetch for missing content blob failed; returning not-found"
|
||||
);
|
||||
Err(e)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Fetches one message from IMAP and re-stores its detached blob.
|
||||
///
|
||||
/// On success the freshly fetched raw RFC822 body is returned; it is also queued
|
||||
/// (in detached form) into the blob store so subsequent requests hit the cache.
|
||||
/// Fails if the message cannot be fetched, or if the fetched bytes do not match
|
||||
/// the archived `content_hash` (the server-side message no longer matches what
|
||||
/// Bichon archived, so it cannot be treated as a recovery of that blob).
|
||||
async fn recover_message_blob(envelope: &Envelope) -> BichonResult<Bytes> {
|
||||
let mailbox = MailBox::find_mailbox(envelope.account_id, envelope.mailbox_id)?
|
||||
.ok_or_else(|| {
|
||||
raise_error!(
|
||||
format!(
|
||||
"Mailbox not found: account_id={} mailbox_id={}",
|
||||
envelope.account_id, envelope.mailbox_id
|
||||
),
|
||||
ErrorCode::ResourceNotFound
|
||||
)
|
||||
})?;
|
||||
|
||||
let mut session = ImapExecutor::create_connection(envelope.account_id).await?;
|
||||
let result = ImapExecutor::fetch_single_message_body(
|
||||
&mut session,
|
||||
&mailbox.encoded_name(),
|
||||
envelope.uid,
|
||||
)
|
||||
.await;
|
||||
session.logout().await.ok();
|
||||
let raw_body = result?;
|
||||
|
||||
let fetched_hash = compute_content_hash(&raw_body);
|
||||
if fetched_hash != envelope.content_hash {
|
||||
return Err(raise_error!(
|
||||
format!(
|
||||
"Fetched message does not match archived content: expected content_hash={} got={}",
|
||||
envelope.content_hash, fetched_hash
|
||||
),
|
||||
ErrorCode::ImapUnexpectedResult
|
||||
));
|
||||
}
|
||||
|
||||
// Re-create the detached blob (stripped EML + attachments) so the missing
|
||||
// blob is repopulated for future requests. The detached EML is queued under
|
||||
// `fetched_hash`, which equals `envelope.content_hash`.
|
||||
let message = MessageParser::new().parse(raw_body.as_slice()).ok_or_else(|| {
|
||||
raise_error!(
|
||||
"Failed to parse fetched email content".into(),
|
||||
ErrorCode::InternalError
|
||||
)
|
||||
})?;
|
||||
detach_and_store_attachments(&raw_body, &message, &fetched_hash, envelope.account_id, envelope.mailbox_id).await;
|
||||
|
||||
Ok(Bytes::from(raw_body))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod test {
|
||||
use html2text::config;
|
||||
|
||||
#[test]
|
||||
fn test_various_html_with_overflow_enabled() {
|
||||
let cases = [
|
||||
("<p>Hello World</p>", "Simple paragraph"),
|
||||
("<h1>Title</h1><p>Content</p>", "Heading + paragraph"),
|
||||
("<ul><li>Item1</li><li>Item2</li></ul>", "Unordered list"),
|
||||
(
|
||||
"<strong>Bold</strong> and <em>italic</em>",
|
||||
"Inline formatting",
|
||||
),
|
||||
(
|
||||
"<div><span>Nested</span> elements</div>",
|
||||
"Nested inline elements inside block",
|
||||
),
|
||||
(
|
||||
"<table><tr><td>A</td><td>B</td></tr></table>",
|
||||
"Simple table",
|
||||
),
|
||||
(
|
||||
"<pre> preformatted text\n line2</pre>",
|
||||
"Preformatted block",
|
||||
),
|
||||
("😃 emoji test", "Wide emoji"),
|
||||
("<a href=\"#\">link</a>", "Anchor tag"),
|
||||
(
|
||||
"<blockquote><p>Quoted text</p></blockquote>",
|
||||
"Blockquote with paragraph",
|
||||
),
|
||||
];
|
||||
|
||||
for (html, desc) in cases {
|
||||
let result = config::plain()
|
||||
.allow_width_overflow()
|
||||
.string_from_read(html.as_bytes(), 100);
|
||||
|
||||
match result {
|
||||
Ok(output) => {
|
||||
println!("✓ Rendered ({}) =>\n{}", desc, output);
|
||||
}
|
||||
Err(e) => panic!("Unexpected error for {}: {:?}", desc, e),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Verifies that [`super::detach_and_store_attachments`] does not panic
|
||||
/// when mail-parser reports attachment offsets past the raw body length.
|
||||
///
|
||||
/// Regression test for: "range end index X out of range for slice of
|
||||
/// length Y" panic caused by a malformed email whose attachment
|
||||
/// `raw_end_offset` exceeded the actual body size.
|
||||
#[tokio::test]
|
||||
async fn detach_attachments_bounds_check() {
|
||||
let raw = concat!(
|
||||
"From: sender@example.com\r\n",
|
||||
"To: recipient@example.com\r\n",
|
||||
"Subject: Test\r\n",
|
||||
"MIME-Version: 1.0\r\n",
|
||||
"Content-Type: multipart/mixed; boundary=\"bnd\"\r\n",
|
||||
"\r\n",
|
||||
"--bnd\r\n",
|
||||
"Content-Type: text/plain\r\n",
|
||||
"\r\n",
|
||||
"Hello\r\n",
|
||||
"--bnd\r\n",
|
||||
"Content-Type: application/octet-stream\r\n",
|
||||
"Content-Disposition: attachment; filename=\"test.bin\"\r\n",
|
||||
"\r\n",
|
||||
"AAAAABBBBBCCCCCDDDDDEEEEEAAAAABBBBBCCCCCDDDDDEEEEE\r\n",
|
||||
"--bnd--\r\n",
|
||||
)
|
||||
.as_bytes()
|
||||
.to_vec();
|
||||
|
||||
let message = mail_parser::MessageParser::new()
|
||||
.parse(&raw)
|
||||
.expect("parse valid MIME message");
|
||||
assert_eq!(message.attachment_count(), 1);
|
||||
|
||||
// Truncate the raw body so the attachment's raw_end_offset lies
|
||||
// past the body end — exactly the scenario reported by users.
|
||||
let truncated = &raw[..raw.len() - 20];
|
||||
assert!(truncated.len() < raw.len());
|
||||
|
||||
// Must not panic.
|
||||
let infos = super::detach_and_store_attachments(
|
||||
truncated,
|
||||
&message,
|
||||
"test_content_hash",
|
||||
0,
|
||||
0,
|
||||
)
|
||||
.await;
|
||||
|
||||
// The attachment count must still match so the consistency check
|
||||
// in reattach_eml_content doesn't fail later.
|
||||
assert_eq!(infos.len(), 1);
|
||||
}
|
||||
}
|
||||
15
crates/core/src/envelope/meta.rs
Normal file
15
crates/core/src/envelope/meta.rs
Normal file
@@ -0,0 +1,15 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::base64_decode;
|
||||
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Deserialize, Serialize)]
|
||||
pub struct BichonMetadata {
|
||||
pub account_email: Option<String>,
|
||||
pub mailbox_name: Option<String>,
|
||||
pub tags: Option<Vec<String>>,
|
||||
}
|
||||
|
||||
pub fn parse_bichon_metadata(header_value: &str) -> Option<BichonMetadata> {
|
||||
let decoded = base64_decode!(header_value.trim());
|
||||
serde_json::from_slice(&decoded).ok()
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -17,4 +17,5 @@
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
pub mod extractor;
|
||||
pub mod meta;
|
||||
pub mod utils;
|
||||
@@ -1,3 +1,21 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use mail_parser::parsers::MessageStream;
|
||||
use regex::{Captures, Regex};
|
||||
|
||||
@@ -72,43 +90,86 @@ pub fn normalize_subject(raw_subject: Option<&str>) -> String {
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::modules::envelope::utils::merge_contiguous_encoded_words;
|
||||
use crate::envelope::utils::{merge_contiguous_encoded_words, normalize_subject};
|
||||
|
||||
// ── merge_contiguous_encoded_words ──────────────────────────────
|
||||
|
||||
#[tokio::test]
|
||||
async fn test3() {
|
||||
#[test]
|
||||
fn merge_basic_utf8_b() {
|
||||
let s = "Hello =?UTF-8?B?SGVsbG8=?= =?UTF-8?B?V29ybGQ=?= !!!";
|
||||
assert_eq!(
|
||||
merge_contiguous_encoded_words(s),
|
||||
"Hello =?UTF-8?B?SGVsbG8=V29ybGQ=?= !!!"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_three_blocks() {
|
||||
let s = "=?UTF-8?B?QQ==?= =?UTF-8?B?Qg==?= =?UTF-8?B?Qw==?=";
|
||||
assert_eq!(
|
||||
merge_contiguous_encoded_words(s),
|
||||
"=?UTF-8?B?QQ==Qg==Qw==?="
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_noncontiguous_blocks() {
|
||||
let s = "=?UTF-8?B?QQ==?= =?UTF-8?B?Qg==?= test =?UTF-8?B?Qw==?= =?UTF-8?B?RA==?=";
|
||||
assert_eq!(
|
||||
merge_contiguous_encoded_words(s),
|
||||
"=?UTF-8?B?QQ==Qg==?= test =?UTF-8?B?Qw==RA==?="
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reject_different_charsets() {
|
||||
let s = "=?UTF-8?B?QQ==?= =?GBK?B?Qg==?=";
|
||||
assert_eq!(merge_contiguous_encoded_words(s), s);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reject_different_encodings() {
|
||||
let s = "=?UTF-8?B?QQ==?= =?UTF-8?Q?Qg?=";
|
||||
assert_eq!(merge_contiguous_encoded_words(s), s);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_case_insensitive_encoding() {
|
||||
let s = "=?UTF-8?b?QQ==?= =?UTF-8?B?Qg==?=";
|
||||
assert_eq!(merge_contiguous_encoded_words(s), "=?UTF-8?B?QQ==Qg==?=");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn single_encoded_word_unchanged() {
|
||||
let s = "Hello =?UTF-8?B?SGVsbG8=?= !!!";
|
||||
assert_eq!(merge_contiguous_encoded_words(s), s);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiple_spaces_between_words() {
|
||||
let s = "=?UTF-8?B?QQ==?= =?UTF-8?B?Qg==?=";
|
||||
assert_eq!(merge_contiguous_encoded_words(s), "=?UTF-8?B?QQ==Qg==?=");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn plain_subject_line() {
|
||||
let s = "Just a normal subject line";
|
||||
assert_eq!(merge_contiguous_encoded_words(s), s);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_quoted_printable() {
|
||||
let s = "=?UTF-8?Q?Hello_?= =?UTF-8?Q?World?=";
|
||||
assert_eq!(merge_contiguous_encoded_words(s), "=?UTF-8?Q?Hello_World?=");
|
||||
assert_eq!(
|
||||
merge_contiguous_encoded_words(s),
|
||||
"=?UTF-8?Q?Hello_World?="
|
||||
);
|
||||
}
|
||||
|
||||
// ── normalize_subject ───────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn normalize_subject_none() {
|
||||
assert_eq!(normalize_subject(None), "");
|
||||
}
|
||||
}
|
||||
44
crates/core/src/error/code.rs
Normal file
44
crates/core/src/error/code.rs
Normal file
@@ -0,0 +1,44 @@
|
||||
#[derive(Copy, Clone, Debug, Eq, PartialEq)]
|
||||
#[repr(u32)]
|
||||
pub enum ErrorCode {
|
||||
// Client-side errors (10000–10999)
|
||||
InvalidParameter = 10000,
|
||||
MissingConfiguration = 10020,
|
||||
Incompatible = 10030,
|
||||
PayloadTooLarge = 10070,
|
||||
RequestTimeout = 10080,
|
||||
MethodNotAllowed = 10090,
|
||||
|
||||
// Authentication and authorization errors (20000–20999)
|
||||
PermissionDenied = 20000,
|
||||
AccountDisabled = 20010,
|
||||
Forbidden = 20020,
|
||||
OAuth2ItemDisabled = 20050,
|
||||
MissingRefreshToken = 20060,
|
||||
|
||||
// Resource errors (30000–30999)
|
||||
ResourceNotFound = 30000,
|
||||
TooManyRequest = 30020,
|
||||
AlreadyExists = 30030,
|
||||
|
||||
// Network connection errors (40000–40999)
|
||||
NetworkError = 40000,
|
||||
ConnectionTimeout = 40010,
|
||||
ConnectionPoolTimeout = 40020,
|
||||
HttpResponseError = 40030,
|
||||
|
||||
// Mail service errors (50000–50999)
|
||||
ImapCommandFailed = 50000,
|
||||
ImapAuthenticationFailed = 50010,
|
||||
ImapUnexpectedResult = 50020,
|
||||
AutoconfigFetchFailed = 50060,
|
||||
// Internal system errors (70000–70999)
|
||||
InternalError = 70000,
|
||||
UnhandledPoemError = 70010,
|
||||
}
|
||||
|
||||
impl ErrorCode {
|
||||
pub fn to_u32(&self) -> u32 {
|
||||
*self as u32
|
||||
}
|
||||
}
|
||||
27
crates/core/src/error/mod.rs
Normal file
27
crates/core/src/error/mod.rs
Normal file
@@ -0,0 +1,27 @@
|
||||
use snafu::{Location, Snafu};
|
||||
|
||||
use crate::error::code::ErrorCode;
|
||||
|
||||
pub mod code;
|
||||
|
||||
#[derive(Debug, Snafu)]
|
||||
#[snafu(visibility(pub))]
|
||||
pub enum BichonError {
|
||||
#[snafu(display("{message}"))]
|
||||
Generic {
|
||||
message: String,
|
||||
#[snafu(implicit)]
|
||||
location: Location,
|
||||
code: ErrorCode,
|
||||
},
|
||||
}
|
||||
|
||||
impl BichonError {
|
||||
pub fn code(&self) -> ErrorCode {
|
||||
match self {
|
||||
BichonError::Generic { code, .. } => *code,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub type BichonResult<T, E = BichonError> = std::result::Result<T, E>;
|
||||
86
crates/core/src/ext/event_bus.rs
Normal file
86
crates/core/src/ext/event_bus.rs
Normal file
@@ -0,0 +1,86 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
// Event bus extension point.
|
||||
//
|
||||
// Community edition: NoopEventBus — all events are discarded.
|
||||
// Pro edition: AuditEventBus — events are persisted to audit database.
|
||||
// Enterprise edition: adds SIEM webhook to the same trait impl.
|
||||
//
|
||||
// The open-source server emits events at key points (login, view, delete, search).
|
||||
// It never reads from the event bus — events are fire-and-forget.
|
||||
|
||||
use std::net::IpAddr;
|
||||
use std::sync::{LazyLock, RwLock};
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum Event {
|
||||
EmailViewed {
|
||||
email_id: String,
|
||||
user: String,
|
||||
ip: IpAddr,
|
||||
},
|
||||
EmailDeleted {
|
||||
email_id: String,
|
||||
user: String,
|
||||
},
|
||||
UserLoggedIn {
|
||||
user: String,
|
||||
ip: IpAddr,
|
||||
},
|
||||
UserCreated {
|
||||
created_by: String,
|
||||
new_user: String,
|
||||
},
|
||||
SearchPerformed {
|
||||
query: String,
|
||||
user: String,
|
||||
},
|
||||
SettingsChanged {
|
||||
key: String,
|
||||
user: String,
|
||||
},
|
||||
AttachmentDownloaded {
|
||||
email_id: String,
|
||||
content_hash: String,
|
||||
user: String,
|
||||
},
|
||||
}
|
||||
|
||||
pub trait EventBus: Send + Sync {
|
||||
fn emit(&self, event: Event);
|
||||
}
|
||||
|
||||
/// Default — all events are discarded.
|
||||
struct NoopEventBus;
|
||||
impl EventBus for NoopEventBus {
|
||||
fn emit(&self, _event: Event) {}
|
||||
}
|
||||
|
||||
static EVENT_BUS: LazyLock<RwLock<Box<dyn EventBus>>> =
|
||||
LazyLock::new(|| RwLock::new(Box::new(NoopEventBus)));
|
||||
|
||||
/// Called by Pro/Enterprise at startup to replace the noop default.
|
||||
pub fn set_event_bus(bus: Box<dyn EventBus>) {
|
||||
*EVENT_BUS.write().unwrap() = bus;
|
||||
}
|
||||
|
||||
/// Fire-and-forget. Called by the server at key points.
|
||||
pub fn emit(event: Event) {
|
||||
EVENT_BUS.read().unwrap().emit(event);
|
||||
}
|
||||
29
crates/core/src/ext/mod.rs
Normal file
29
crates/core/src/ext/mod.rs
Normal file
@@ -0,0 +1,29 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
// Event bus extension point.
|
||||
//
|
||||
// Community edition: NoopEventBus — all events are discarded.
|
||||
// Pro edition: AuditEventBus — events are persisted to audit database.
|
||||
// Enterprise edition: adds SIEM webhook to the same trait impl.
|
||||
//
|
||||
// The open-source server emits events at key points (login, view, delete, search).
|
||||
// It never reads from the event bus — events are fire-and-forget.
|
||||
|
||||
pub mod event_bus;
|
||||
pub mod text_extractor;
|
||||
85
crates/core/src/ext/text_extractor.rs
Normal file
85
crates/core/src/ext/text_extractor.rs
Normal file
@@ -0,0 +1,85 @@
|
||||
//
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU Affero General Public License as published by
|
||||
// the Free Software Foundation, either version 3 of the License, or
|
||||
// (at your option) any later version.
|
||||
//
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU Affero General Public License for more details.
|
||||
//
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
// Attachment text extraction extension point.
|
||||
//
|
||||
// Community edition: NoopExtractor — no attachments are text-indexed.
|
||||
// Pro edition: PdfExtractor — extracts text from PDF, Word, etc.
|
||||
//
|
||||
// Used in: crates/core/src/envelope/extractor.rs
|
||||
|
||||
use std::sync::{LazyLock, RwLock};
|
||||
|
||||
pub struct ExtractedText {
|
||||
pub text: String,
|
||||
pub page_count: Option<u32>,
|
||||
pub is_ocr: bool,
|
||||
}
|
||||
|
||||
pub trait AttachmentTextExtractor: Send + Sync {
|
||||
/// Returns None if this extractor doesn't handle the file type.
|
||||
/// Returns Some(ExtractedText) if text was successfully extracted.
|
||||
fn extract(&self, content_type: &str, ext: &str, bytes: &[u8]) -> Option<ExtractedText>;
|
||||
}
|
||||
|
||||
/// Default — all attachments are skipped.
|
||||
struct NoopExtractor;
|
||||
impl AttachmentTextExtractor for NoopExtractor {
|
||||
fn extract(&self, _ct: &str, _ext: &str, _bytes: &[u8]) -> Option<ExtractedText> {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
static EXTRACTOR: LazyLock<RwLock<Box<dyn AttachmentTextExtractor>>> =
|
||||
LazyLock::new(|| RwLock::new(Box::new(NoopExtractor)));
|
||||
|
||||
/// Called by Pro/Enterprise at startup to replace the noop default.
|
||||
pub fn set_extractor(extractor: Box<dyn AttachmentTextExtractor>) {
|
||||
*EXTRACTOR.write().unwrap() = extractor;
|
||||
}
|
||||
|
||||
/// Attachments larger than this are skipped (10 MiB). Avoids excessive memory
|
||||
/// and CPU cost for huge files whose text is rarely useful for search.
|
||||
pub const MAX_EXTRACT_BYTES: usize = 10 * 1024 * 1024;
|
||||
|
||||
/// Quick pre-filter: returns true for file types where text extraction may
|
||||
/// produce useful results. Avoids cloning attachment bytes for images, videos,
|
||||
/// archives, etc. when no registered extractor would handle them.
|
||||
pub fn should_try_extract(content_type: &str, ext: &str) -> bool {
|
||||
matches!(
|
||||
ext,
|
||||
"pdf"
|
||||
| "doc"
|
||||
| "docx"
|
||||
| "xls"
|
||||
| "xlsx"
|
||||
| "ppt"
|
||||
| "pptx"
|
||||
| "txt"
|
||||
| "rtf"
|
||||
| "odt"
|
||||
| "ods"
|
||||
| "odp"
|
||||
) || content_type.starts_with("text/")
|
||||
}
|
||||
|
||||
/// Called by the attachment pipeline during IMAP sync.
|
||||
/// The caller should wrap this in spawn_blocking for CPU-bound extraction.
|
||||
pub fn extract_text(content_type: &str, ext: &str, bytes: &[u8]) -> Option<ExtractedText> {
|
||||
EXTRACTOR.read().unwrap().extract(content_type, ext, bytes)
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
//
|
||||
// Copyright (c) 2025 rustmailer.com (https://rustmailer.com)
|
||||
// Copyright (c) 2025-2026 rustmailer.com (https://rustmailer.com)
|
||||
//
|
||||
// This file is part of the Bichon Email Archiving Project
|
||||
//
|
||||
@@ -16,9 +16,9 @@
|
||||
// You should have received a copy of the GNU Affero General Public License
|
||||
// along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
use crate::modules::error::code::ErrorCode;
|
||||
use crate::modules::imap::session::SessionStream;
|
||||
use crate::{modules::error::BichonResult, raise_error};
|
||||
use crate::error::code::ErrorCode;
|
||||
use crate::imap::session::SessionStream;
|
||||
use crate::{error::BichonResult, raise_error};
|
||||
use async_imap::types::Capability;
|
||||
use async_imap::{types::Capabilities, Session};
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user