diff --git a/CHANGELOG.md b/CHANGELOG.md index 4eae0c5..87ee752 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,25 @@ and this project adheres to [Semantic Versioning](https://semver.org/). ## [Unreleased] +## [0.8.1] - 2026-10-02 + +### Added + +- Non-ASCII search terms also work on servers that advertise `LITERAL-` instead of `LITERAL+`, such as Gmail, for terms up to 4096 bytes + +### Changed + +- Non-ASCII folder names are shown decoded (`[Gmail]/Messages envoyés` instead of `[Gmail]/Messages envoy&AOk-s`), and folder options and config settings accept either form; `--json` output and `export` file names use the server's listed name (so `--folder inbox` exports `INBOX_.eml`) +- `--limit` fetches headers only for recent matches when they suffice instead of for every match, on servers without SORT, such as Gmail, and in each folder of `--all-folders` on every server (`search --limit 5` on a 7,480-message Gmail INBOX: 22.9 s to 1.6 s) +- On Proton Mail Bridge, `--all-folders` skips the label views (`Labels/...` and `Starred`), so a labelled message is listed and counted once, from its regular folder (on a 150,000-message account, `search --all-folders --limit 5` went from 33.9 s to 3.1 s and the `count --all-folders` total from 305,131 to 151,285) + +### Fixed + +- `--all-folders` and `status` skip containers that cannot be opened (`\Noselect`, such as Gmail's `[Gmail]`) instead of warning about or listing them +- On Gmail, `--all-folders` lists a message once even when it has several labels (the INBOX copy if there is one), and actions apply to that copy; `count --all-folders` totals count it once +- `--all-folders` also skips folders inside Trash, Spam/Junk, and All Mail (such as `[Gmail]/Trash/Old`) +- When slashmail orders messages itself (servers without SORT, `--all-folders`, `--all-accounts`), a Date header more than a day after a message arrived counts as arrival plus one day, so a wrong or forged future date no longer pins a message to the top or pushes newer messages out of a `--limit` + ## [0.8.0] - 2026-09-28 ### Added @@ -180,7 +199,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/). - Plaintext connection warning for non-loopback hosts - Passwords securely zeroed from memory after login -[Unreleased]: https://github.com/mwmdev/slashmail/compare/v0.8.0...HEAD +[Unreleased]: https://github.com/mwmdev/slashmail/compare/v0.8.1...HEAD +[0.8.1]: https://github.com/mwmdev/slashmail/compare/v0.8.0...v0.8.1 [0.8.0]: https://github.com/mwmdev/slashmail/compare/v0.7.0...v0.8.0 [0.7.0]: https://github.com/mwmdev/slashmail/compare/v0.6.0...v0.7.0 [0.6.0]: https://github.com/mwmdev/slashmail/compare/v0.5.0...v0.6.0 diff --git a/Cargo.lock b/Cargo.lock index 6cdd322..b68a74b 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1509,7 +1509,7 @@ checksum = "b2aa850e253778c88a04c3d7323b043aeda9d3e30d5971937c1855769763678e" [[package]] name = "slashmail" -version = "0.8.0" +version = "0.8.1" dependencies = [ "anyhow", "clap", diff --git a/Cargo.toml b/Cargo.toml index 71e70f3..539e5ae 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "slashmail" -version = "0.8.0" +version = "0.8.1" edition = "2021" rust-version = "1.88" description = "CLI for searching, managing, drafting, and bulk-operating on emails via IMAP" diff --git a/README.md b/README.md index bc5ec17..bca7860 100644 --- a/README.md +++ b/README.md @@ -60,7 +60,7 @@ Commands: move Search + move matching messages to a folder export Search + export matching messages as .eml files mark Search + set/unset flags on matching messages - count Count matching messages (no FETCH) + count Count matching messages (no header or body fetch) quota Show mailbox quota usage status Show per-folder message statistics ``` @@ -266,7 +266,7 @@ Search, read, count, and bulk message commands share these filter options: ``` -f, --folder Folder to search [default: INBOX] - --all-folders Search across all folders (excludes Trash, Junk/Spam, All Mail) + --all-folders Search across all folders (excludes Trash, Junk/Spam, All Mail, and folders inside them) --subject Subject contains --from From address contains --to To address contains @@ -288,7 +288,13 @@ Search, read, count, and bulk message commands share these filter options: All filter criteria are AND'd together. Omitting all criteria matches all messages. Text filters (`--subject`, `--from`, `--to`, `--cc`, `--body`, `--text`) must not be empty, so an unset shell variable cannot turn a filter into "match everything". `--folder` and `--all-folders` cannot be combined. -`--all-folders` skips mailboxes the server marks `\All`, `\Trash`, or `\Junk` (for example `Deleted Items` and `Junk Email`), plus folders named `Trash`, `Spam`, `Junk`, or `All Mail` for servers without those markers. `delete` and `move` also never search their destination folder, and naming the destination as the only source folder is an error. +`--all-folders` skips mailboxes the server marks `\All`, `\Trash`, or `\Junk` (for example `Deleted Items` and `Junk Email`), plus folders named `Trash`, `Spam`, `Junk`, or `All Mail` for servers without those markers, and every folder inside one of those (such as `Deleted Items/2024` or `[Gmail]/Trash/Old`). It also skips containers that cannot be opened (`\Noselect`, such as Gmail's `[Gmail]`) but still searches the folders inside them. `delete` and `move` also never search their destination folder, and naming the destination as the only source folder is an error. + +On Gmail, where every label is a folder, `--all-folders` lists a message once even when it has several labels: the INBOX copy if there is one, otherwise the copy in the first folder listed. `read`, `export`, `mark`, `move`, and `delete` act on that copy. `count --all-folders` still shows each folder's own count, but its total counts each message once. + +On Proton Mail Bridge, every message sits in exactly one regular folder (INBOX, Archive, Sent, `Folders/...`), and its `Labels/...` folders and `Starred` are views of those messages. `--all-folders` skips those views there, so each message is listed and counted once, from its regular folder. + +Servers list non-ASCII folder names in IMAP's modified UTF-7 (`[Gmail]/Messages envoy&AOk-s`). The terminal shows them decoded (`[Gmail]/Messages envoyés`), and every folder option and config setting accepts either form. `--json` output keeps the server's form. ### Action options @@ -305,11 +311,11 @@ Commands that modify messages (`delete`, `move`, `mark`) support: `delete` and `move` require the server to advertise `MOVE` or `UIDPLUS`. Without `MOVE`, messages are copied, flagged `\Deleted`, and removed with `UID EXPUNGE` of exactly those UIDs; other messages already flagged `\Deleted` are never expunged. This fallback is not atomic: if a step fails, slashmail stops and reports it without retrying, including how many messages were already moved or updated. Every mutating command and `export`/`read` refuse to act if the folder's `UIDVALIDITY` changed since the search. Immediately before `delete`, `move`, and `mark` act, slashmail asks the server which searched messages still exist; their receipts count only those and report any another client removed in the meantime. -`export` supports `--yes`, `--force` (replace existing files), and `-o, --output-dir`. Files are named `_.eml`, where the folder name is percent-encoded: ASCII letters, digits, and `-` are kept and every other byte becomes `%XX` (so on Linux and macOS `Work/Projects` is `Work%2FProjects_1.eml` and `Work_Projects` is `Work%5FProjects_1.eml`). On Windows, lowercase letters are also encoded so folders differing only by case stay distinct (`Work/P` is `W%6F%72%6B%2FP_1.eml`). Without `--force`, an existing file is skipped only when it already holds the same message (identical bytes or the same Message-ID). UIDs restart when a mailbox is recreated or migrated, so an existing file holding a different message is left unchanged and reported as an error after the other messages are exported; use `--force` or a new output directory. `--force` replaces only a regular file or symlink entry and never follows symlinks. New exports and saved attachments are created owner-only (`0600`) on Unix. +`export` supports `--yes`, `--force` (replace existing files), and `-o, --output-dir`. Files are named `_.eml`, where `` is the server's listed name (`INBOX` even for `--folder inbox`, and non-ASCII names in their encoded form), percent-encoded: ASCII letters, digits, and `-` are kept and every other byte becomes `%XX` (so on Linux and macOS `Work/Projects` is `Work%2FProjects_1.eml` and `Work_Projects` is `Work%5FProjects_1.eml`). On Windows, lowercase letters are also encoded so folders differing only by case stay distinct (`Work/P` is `W%6F%72%6B%2FP_1.eml`). Without `--force`, an existing file is skipped only when it already holds the same message (identical bytes or the same Message-ID). UIDs restart when a mailbox is recreated or migrated, so an existing file holding a different message is left unchanged and reported as an error after the other messages are exported; use `--force` or a new output directory. `--force` replaces only a regular file or symlink entry and never follows symlinks. New exports and saved attachments are created owner-only (`0600`) on Unix. `mark` takes one or more actions: `--read`, `--unread`, `--set-flagged`, `--clear-flagged`. -Search terms containing non-ASCII text are sent as UTF-8 literals and require the server to advertise `LITERAL+`; otherwise the search fails before any mailbox is searched. +Search terms containing non-ASCII text are sent as UTF-8 literals and require the server to advertise `LITERAL+`, or `LITERAL-` (as Gmail does) for terms up to 4096 bytes; otherwise the search fails before any mailbox is searched. Proton Mail Bridge advertises neither, so non-ASCII search fails there. Bridge also matches `--subject` and `--text` against the raw message as stored, so words inside an encoded subject (common when it has non-ASCII characters, often base64) may not match, accented or not. To find such messages there, narrow with other filters (`--from`, `--since`) and check the decoded `subject` in `search --json`. ## Examples @@ -376,7 +382,7 @@ slashmail mark -u user@example.com --from "notifications" --read # Flag important messages slashmail mark -u user@example.com --subject "urgent" --set-flagged -# Count matching messages (fast, no FETCH) +# Count matching messages (fast, no header or body fetch) slashmail count -u user@example.com --from "newsletter" # Show folder statistics @@ -455,17 +461,19 @@ Destructive operations always dry-run first and ask for confirmation. ## Tested with -- Gmail (via `--tls --host imap.gmail.com`) -- Fastmail (via `--tls --host imap.fastmail.com`) -- Dovecot -- Any standard IMAP4rev1 server +- Gmail (`--tls --host imap.gmail.com`, with an app password) +- Proton Mail Bridge 3.23 (`127.0.0.1:1143`, read-only checks) +- Dovecot 2.3 +- GreenMail (automated test suite) + +Other IMAP4rev1 servers should work. `delete` and `move` need `MOVE` or `UIDPLUS`, and non-ASCII search needs `LITERAL+` or `LITERAL-`. ## How it works - All filtering runs server-side via IMAP SEARCH -- Uses IMAP SORT extension (RFC 5256) when available; falls back to client-side sort -- With SORT, `--limit` truncates results before fetching (fewer bytes over the wire) -- `search`, `delete`, `move`, `mark`, `count` only fetch headers and size -- never full messages +- Uses IMAP SORT extension (RFC 5256) when available for a single folder. Otherwise, and for every folder of `--all-folders` and every account of `--all-accounts`, slashmail orders by the Date header, but never later than one day after a message arrived, so a wrong or forged future date cannot keep a message at the top +- `--limit` keeps header fetches small. With SORT, a single-folder search is truncated before fetching. Otherwise (no SORT, as on Gmail and Proton Bridge, and each folder or account being merged), a large search is first narrowed to recently arrived messages with `SINCE`, and every match is fetched only when those hold too few. The result is the same as fetching everything +- `search`, `delete`, `move`, `mark` only fetch headers and size -- never full messages; `count` fetches neither (only message IDs with `--all-folders` on Gmail) - `export` fetches full message bodies via `BODY.PEEK[]` - Uses `BODY.PEEK` to avoid marking messages as read, and opens folders read-only (`EXAMINE`) for `search`, `read`, `count`, `export`, and `--dry-run`, so they do not clear the `\Recent` flag - UID sets are compressed into ranges and chunked to stay within IMAP command length limits @@ -497,7 +505,7 @@ All errors print to stderr. Combine `--yes` with cron or scripts for unattended ### Folder not found - Run `slashmail status` to list all available folders and their names -- Folder names are case-sensitive on most IMAP servers +- `search`, `read`, `export`, `mark`, `move`, and `delete` need a name `slashmail status` lists, decoded as shown or as the server sends it (INBOX in any letter case). `count`, `reply`, and `attachments` also try a name exactly as typed, for mailboxes a server opens but does not list - Gmail uses `[Gmail]/Trash`, `[Gmail]/All Mail`, etc. — use `--trash-folder` with `delete` if needed - Exchange/Outlook uses `Deleted Items` instead of `Trash` diff --git a/skills/slashmail/SKILL.md b/skills/slashmail/SKILL.md index 8ef9031..7688273 100644 --- a/skills/slashmail/SKILL.md +++ b/skills/slashmail/SKILL.md @@ -33,7 +33,7 @@ Optional `sender` and `drafts_folder` values can be set at the top level or per | Flag | Description | |------|-------------| | `-f, --folder FOLDER` | Target folder (default: INBOX); cannot be combined with `--all-folders` | -| `--all-folders` | Search all folders (excludes Trash, Junk/Spam, All Mail, and the move/delete destination) | +| `--all-folders` | Search all folders (excludes Trash, Junk/Spam, All Mail and folders inside them, and the move/delete destination; on Proton Bridge also `Labels/...` and `Starred`; on Gmail each message is one row, from INBOX when it is there) | | `--subject TEXT` | Filter by subject | | `--from TEXT` | Filter by sender | | `--to TEXT` | Filter by recipient | @@ -143,7 +143,10 @@ With `--json`, the receipt is one object with `account`, `folder`, `uid`, `messa - Never pass an empty text filter (for example an unset variable as `--from`); slashmail rejects it rather than matching every message. - If `export` reports existing files that hold a different message, do not add `--force` without the user's approval; suggest a new output directory instead. - If an action on searched messages fails because the folder's UIDVALIDITY changed, the mailbox was rebuilt since the search: run the search again and act on the new UIDs. Never reuse the old ones. -- `delete` and `move` require server support for `MOVE` or `UIDPLUS` and fail before changing anything otherwise. Non-ASCII search terms require `LITERAL+`; without it the search fails rather than matching nothing. Report these failures instead of working around them. +- `delete` and `move` require server support for `MOVE` or `UIDPLUS` and fail before changing anything otherwise. Non-ASCII search terms require `LITERAL+`, or `LITERAL-` for terms up to 4096 bytes; without it the search fails rather than matching nothing. Report these failures instead of working around them. +- `--json` `folder` values are the server's names (non-ASCII names in modified UTF-7, such as `[Gmail]/Messages envoy&AOk-s`). Pass them back unchanged; folder options also accept the decoded name the terminal shows. +- With `--all-folders`, a Gmail message with several labels is one row, from INBOX when it is there, and commands act on that copy. Gmail keeps flags per message, so `mark` changes it in every label. On Proton Bridge, label folders are skipped: to work with a label, name it with `--folder` (for example `--folder "Labels/Clients"`). +- On Proton Bridge, server search cannot see words inside an encoded subject (common when it has non-ASCII characters): `--subject` and `--text` both match the raw message. Narrow with other filters (`--from`, `--since`) and check the decoded `subject` field of `search --json` yourself, and tell the user before reporting a Bridge subject search as complete. ## Received Attachment Rules diff --git a/src/connection.rs b/src/connection.rs index 656511b..ae75d5a 100644 --- a/src/connection.rs +++ b/src/connection.rs @@ -2,7 +2,7 @@ use crate::draft::{AppendAttempt, DraftMailboxSession, HeaderFetch, MailboxListi use crate::search; use anyhow::{Context, Result}; use imap::Session; -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use std::io::{self, ErrorKind}; use std::net::{TcpStream, ToSocketAddrs}; use std::time::{Duration, Instant}; @@ -21,6 +21,10 @@ enum Inner { pub struct ImapSession { inner: Inner, capabilities: HashSet, + /// The server greeted as Proton Mail Bridge, whose `Labels/...` folders + /// and `Starred` are views of messages that also sit in exactly one + /// regular folder. + proton_bridge: bool, } impl ImapSession { @@ -53,14 +57,13 @@ impl ImapSession { } /// Run `UID SEARCH`. Queries containing UTF-8 literals declare - /// `CHARSET UTF-8` and require LITERAL+ so no continuation is needed. + /// `CHARSET UTF-8` and must pass `search::ensure_query_supported`, so no + /// continuation is needed. pub fn uid_search(&mut self, query: &str) -> Result> { + search::ensure_query_supported(self, query)?; let command = if query.is_ascii() { std::borrow::Cow::Borrowed(query) } else { - if !self.has_capability("LITERAL+") { - anyhow::bail!(search::LITERAL_PLUS_REQUIRED); - } std::borrow::Cow::Owned(format!("CHARSET UTF-8 {query}")) }; let uids = match &mut self.inner { @@ -137,6 +140,37 @@ impl ImapSession { parse_status_response(&data) } + /// Gmail's `X-GM-MSGID` for each UID of `uid_set` in the selected + /// mailbox. Gmail lists a message once per label, and the ID is the same + /// in every label folder. UIDs the server omits are absent. + pub fn gmail_message_ids(&mut self, uid_set: &str) -> Result> { + let data = + self.run_command_and_read_response(&format!("UID FETCH {uid_set} (UID X-GM-MSGID)"))?; + let mut ids = HashMap::new(); + parse_responses(&data, |_, response| { + if let imap_proto::Response::Fetch(_, attributes) = response { + let mut uid = None; + let mut id = None; + for attribute in attributes { + match attribute { + imap_proto::AttributeValue::Uid(value) => uid = Some(value), + imap_proto::AttributeValue::GmailMsgId(value) => id = Some(value), + _ => {} + } + } + if let (Some(uid), Some(id)) = (uid, id) { + ids.insert(uid, id); + } + } + true + })?; + Ok(ids) + } + + pub fn is_proton_bridge(&self) -> bool { + self.proton_bridge + } + pub fn get_quota_root( &mut self, mailbox: &str, @@ -320,13 +354,15 @@ fn parse_list_response(data: &[u8]) -> Result> { parse_responses(data, |raw, response| { if let imap_proto::Response::MailboxData(imap_proto::MailboxDatum::List { name_attributes, + delimiter, name, - .. }) = response { listed.push(MailboxListing { name: listed_mailbox_name(raw, &name), attributes: name_attributes.iter().map(name_attribute_text).collect(), + // A delimiter is a quoted char or NIL, never a literal. + delimiter: delimiter.map(|delimiter| unescape_quoted(&delimiter)), }); } true @@ -353,8 +389,14 @@ fn listed_mailbox_name(raw: &[u8], name: &str) -> String { if is_literal || !line.ends_with(b"\"") { return name.to_string(); } - let mut decoded = String::with_capacity(name.len()); - let mut characters = name.chars(); + unescape_quoted(name) +} + +/// Remove the `\` of each quoted-pair (`\\`, `\"`) that imap-proto leaves in +/// a quoted string. +fn unescape_quoted(text: &str) -> String { + let mut decoded = String::with_capacity(text.len()); + let mut characters = text.chars(); while let Some(character) = characters.next() { match character { '\\' => decoded.extend(characters.next()), @@ -422,7 +464,7 @@ pub fn ensure_secure_transport(host: &str, tls: bool) -> Result<()> { pub fn connect(host: &str, port: u16, tls: bool, user: &str, pass: &str) -> Result { ensure_secure_transport(host, tls)?; - let mut session = if tls { + let (mut session, greeting) = if tls { let tls_connector = native_tls::TlsConnector::builder() .min_protocol_version(Some(native_tls::Protocol::Tlsv12)) .danger_accept_invalid_certs(false) @@ -435,26 +477,26 @@ pub fn connect(host: &str, port: u16, tls: bool, user: &str, pass: &str) -> Resu .connect(host, tcp) .with_context(|| format!("Failed to TLS-connect to {host}:{port}"))?; let mut client = imap::Client::new(tls); - client + let greeting = client .read_greeting() .context("Failed to read the IMAP greeting")?; let s = client .login(user, pass) .map_err(|e| e.0) .context("IMAP login failed")?; - Inner::Tls(s) + (Inner::Tls(s), greeting) } else { let tcp = connect_tcp(host, port, IMAP_CONNECT_TIMEOUT, IMAP_IO_TIMEOUT) .with_context(|| format!("Failed to connect to {host}:{port}"))?; let mut client = imap::Client::new(tcp); - client + let greeting = client .read_greeting() .context("Failed to read the IMAP greeting")?; let s = client .login(user, pass) .map_err(|e| e.0) .context("IMAP login failed")?; - Inner::Plain(s) + (Inner::Plain(s), greeting) }; let caps = match &mut session { @@ -462,19 +504,36 @@ pub fn connect(host: &str, port: u16, tls: bool, user: &str, pass: &str) -> Resu Inner::Tls(s) => s.capabilities(), } .context("Failed to fetch capabilities")?; - let capabilities = ["SORT", "MOVE", "QUOTA", "UIDPLUS", "LITERAL+"] - .iter() - .filter(|c| caps.has_str(**c)) - .map(|c| c.to_string()) - .collect(); + let capabilities = [ + "SORT", + "MOVE", + "QUOTA", + "UIDPLUS", + "LITERAL+", + "LITERAL-", + "X-GM-EXT-1", + ] + .iter() + .filter(|c| caps.has_str(**c)) + .map(|c| c.to_string()) + .collect(); drop(caps); Ok(ImapSession { inner: session, capabilities, + proton_bridge: is_proton_bridge_greeting(&greeting), }) } +/// Proton Mail Bridge greets with `* OK [...] ProtonMailBridge 03.23.01 - +/// gluon session ID 1`. +fn is_proton_bridge_greeting(greeting: &[u8]) -> bool { + greeting + .windows(b"ProtonMailBridge".len()) + .any(|window| window == b"ProtonMailBridge") +} + fn connect_tcp( host: &str, port: u16, diff --git a/src/delete.rs b/src/delete.rs index 63ef293..197d2e4 100644 --- a/src/delete.rs +++ b/src/delete.rs @@ -3,7 +3,7 @@ use indicatif::{ProgressBar, ProgressStyle}; use std::time::Duration; use crate::connection::ImapSession; -use crate::display::{display_messages, sanitize_terminal_field}; +use crate::display::{display_messages, sanitize_folder_name, sanitize_terminal_field}; use crate::search::{self, SearchCriteria}; fn spinner(msg: &str) -> ProgressBar { @@ -20,7 +20,7 @@ fn spinner(msg: &str) -> ProgressBar { pub fn search_and_move( session: &mut ImapSession, - criteria: &SearchCriteria, + criteria: &mut SearchCriteria, dest: &str, yes: bool, dry_run: bool, @@ -30,14 +30,25 @@ pub fn search_and_move( pub fn search_and_move_with_account( session: &mut ImapSession, - criteria: &SearchCriteria, + criteria: &mut SearchCriteria, dest: &str, yes: bool, dry_run: bool, account_name: Option<&str>, ) -> Result<()> { + // Resolve the destination first so it is excluded from the search. A + // missing destination still allows a dry run and fails before any move. + let listed_dest = search::lookup_folder(session, dest)?; + let safe_dest = match &listed_dest { + Some(name) => sanitize_folder_name(name), + None => sanitize_terminal_field(dest), + }; let sp = spinner("Searching..."); - let mut messages = search::search_excluding(session, criteria, Some(dest))?; + let mut messages = search::search_excluding( + session, + criteria, + Some(listed_dest.as_deref().unwrap_or(dest)), + )?; sp.finish_and_clear(); if let Some(account) = account_name { for msg in &mut messages { @@ -52,7 +63,6 @@ pub fn search_and_move_with_account( display_messages(&messages); - let safe_dest = sanitize_terminal_field(dest); if dry_run { println!( "Dry run: {} message(s) would be moved to {safe_dest}.", @@ -62,7 +72,7 @@ pub fn search_and_move_with_account( } session.ensure_safe_move_supported()?; - search::ensure_folder_exists(session, dest)?; + let dest = listed_dest.ok_or_else(|| search::missing_folder(dest))?; // Validate every row's mailbox identity before any mutation. let groups = search::group_message_uids(&messages, &criteria.folder)?; @@ -89,7 +99,7 @@ pub fn search_and_move_with_account( let failed = |total: usize| { format!( "Failed to move messages from '{}' to '{safe_dest}' ({total} already moved)", - sanitize_terminal_field(folder) + sanitize_folder_name(folder) ) }; search::select_verified(session, folder, *uid_validity) @@ -99,7 +109,7 @@ pub fn search_and_move_with_account( search::existing_uids(session, chunk).with_context(|| failed(total))?; for set in &search::build_uid_set(&present) { session - .uid_move_or_fallback(set, dest) + .uid_move_or_fallback(set, &dest) .with_context(|| failed(total))?; total += search::uid_set_len(set); } @@ -127,7 +137,7 @@ pub fn report_vanished(searched: usize, acted: usize) { pub fn delete( session: &mut ImapSession, - criteria: &SearchCriteria, + criteria: &mut SearchCriteria, trash_folder: &str, yes: bool, dry_run: bool, @@ -137,7 +147,7 @@ pub fn delete( pub fn delete_with_account( session: &mut ImapSession, - criteria: &SearchCriteria, + criteria: &mut SearchCriteria, trash_folder: &str, yes: bool, dry_run: bool, diff --git a/src/display.rs b/src/display.rs index 179e1a9..49d4edc 100644 --- a/src/display.rs +++ b/src/display.rs @@ -21,6 +21,30 @@ pub struct MessageRow { /// UID actions refuse to run if the mailbox identity has changed. #[serde(skip)] pub uid_validity: Option, + /// Gmail's `X-GM-MSGID`, fetched for multi-folder results so a message + /// listed under several labels is shown once. Internal only. + #[serde(skip)] + pub gmail_msgid: Option, + /// INTERNALDATE (arrival) as Unix seconds. Internal only. + #[serde(skip)] + pub arrival: Option, +} + +/// How far a Date header may run ahead of arrival before it is capped. +pub const ARRIVAL_SLACK_SECS: i64 = 86_400; + +impl MessageRow { + /// The time messages are ordered by when slashmail sorts them: the Date + /// header, but never more than a day after the message arrived, so a + /// wrong or forged future date cannot pin a message to the top (and + /// date-narrowed searches stay exact on servers that filter by + /// arrival). An undated message sorts last. + pub fn sort_time(&self) -> i64 { + match self.arrival { + Some(arrival) => self.timestamp.min(arrival + ARRIVAL_SLACK_SECS), + None => self.timestamp, + } + } } pub fn format_size(bytes: u64) -> String { @@ -41,6 +65,12 @@ pub fn sanitize_terminal_field(value: &str) -> String { sanitize_terminal(value, false) } +/// A mailbox name for the terminal: modified UTF-7 decoded, then sanitized +/// like any other untrusted field. +pub fn sanitize_folder_name(name: &str) -> String { + sanitize_terminal_field(&crate::utf7::display(name)) +} + /// Like [`sanitize_terminal_field`], but keeps multi-line structure: LF and /// HTAB are preserved, CRLF or lone CR become LF, and invisible formatting /// characters (soft hyphens, zero-width spaces) are dropped rather than @@ -79,6 +109,41 @@ fn is_tag_character(character: char) -> bool { matches!(character, '\u{e0000}'..='\u{e007f}') } +/// Characters that show as nothing or as blank space in a terminal field: +/// whitespace, controls, the sanitizer's invisible and bidi formatting +/// characters, every Unicode Default_Ignorable_Code_Point (joiners, +/// variation selectors, Hangul fillers, tags, ...), and the blank Braille +/// pattern. +pub(crate) fn renders_blank(character: char) -> bool { + character.is_whitespace() + || character.is_control() + || is_invisible_character(character) + || is_unsafe_format_character(character) + || matches!( + character, + '\u{00ad}' + | '\u{034f}' + | '\u{061c}' + | '\u{115f}' + | '\u{1160}' + | '\u{17b4}' + | '\u{17b5}' + | '\u{180b}'..='\u{180f}' + | '\u{200b}'..='\u{200f}' + | '\u{202a}'..='\u{202e}' + | '\u{2060}'..='\u{206f}' + | '\u{2800}' + | '\u{3164}' + | '\u{fe00}'..='\u{fe0f}' + | '\u{feff}' + | '\u{ffa0}' + | '\u{fff0}'..='\u{fff8}' + | '\u{1bca0}'..='\u{1bca3}' + | '\u{1d173}'..='\u{1d17a}' + | '\u{e0000}'..='\u{e0fff}' + ) +} + fn sanitize_terminal(value: &str, multiline: bool) -> String { #[derive(Clone, Copy)] enum State { @@ -241,7 +306,7 @@ pub fn display_messages(messages: &[MessageRow]) { ))); } if has_folder { - row.push(Cell::new(sanitize_terminal_field( + row.push(Cell::new(sanitize_folder_name( msg.folder.as_deref().unwrap_or(""), ))); } @@ -435,6 +500,8 @@ mod tests { answered: false, flagged: false, uid_validity: None, + gmail_msgid: None, + arrival: None, } } diff --git a/src/draft.rs b/src/draft.rs index fc96871..31b7a34 100644 --- a/src/draft.rs +++ b/src/draft.rs @@ -55,6 +55,51 @@ pub struct ComposedDraft { pub struct MailboxListing { pub name: String, pub attributes: Vec, + /// Hierarchy delimiter from LIST (`/` or `.`); `None` for a flat + /// namespace. + pub delimiter: Option, +} + +impl MailboxListing { + /// `\Noselect` and `\NonExistent` (RFC 5258) mailboxes cannot be opened, + /// such as Gmail's `[Gmail]` container. + pub fn is_selectable(&self) -> bool { + !self.has_attribute("\\Noselect") && !self.has_attribute("\\NonExistent") + } + + pub fn has_attribute(&self, expected: &str) -> bool { + self.attributes + .iter() + .any(|attribute| attribute.eq_ignore_ascii_case(expected)) + } + + /// Whether `other` is nested inside this mailbox (`Trash/Old` in + /// `Trash`). + pub fn contains(&self, other: &MailboxListing) -> bool { + self.delimiter + .as_deref() + .filter(|delimiter| !delimiter.is_empty()) + .is_some_and(|delimiter| { + other + .name + .strip_prefix(&self.name) + .is_some_and(|rest| rest.starts_with(delimiter)) + }) + } + + /// The listed mailbox a user-supplied name refers to: an exact match + /// (INBOX case-insensitively), else the one listed under the name's + /// modified UTF-7 encoding. `Envoyés`, `Envoy&AOk-s`, `R&D` (listed as + /// `R&-D`), and names a server lists in raw UTF-8 all resolve. + pub fn find<'a>(mailboxes: &'a [MailboxListing], requested: &str) -> Option<&'a Self> { + mailboxes + .iter() + .find(|mailbox| crate::search::same_mailbox(&mailbox.name, requested)) + .or_else(|| { + let encoded = crate::utf7::encode(requested); + mailboxes.iter().find(|mailbox| mailbox.name == encoded) + }) + } } #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -146,19 +191,15 @@ pub fn resolve_destination( let requested = command_override.or(configured); if let Some(requested) = requested { validate_header_text(requested, "Drafts folder")?; - let matches = mailboxes - .iter() - .filter(|mailbox| mailbox.name == requested && is_selectable(mailbox)) - .collect::>(); - return match matches.as_slice() { - [mailbox] => Ok(mailbox.name.clone()), + return match MailboxListing::find(mailboxes, requested) { + Some(mailbox) if mailbox.is_selectable() => Ok(mailbox.name.clone()), _ => bail!("Drafts folder '{requested}' does not exist or is not selectable"), }; } let candidates = mailboxes .iter() - .filter(|mailbox| is_selectable(mailbox) && has_attribute(mailbox, "\\Drafts")) + .filter(|mailbox| mailbox.is_selectable() && mailbox.has_attribute("\\Drafts")) .collect::>(); match candidates.as_slice() { [mailbox] => Ok(mailbox.name.clone()), @@ -171,17 +212,6 @@ pub fn resolve_destination( } } -fn is_selectable(mailbox: &MailboxListing) -> bool { - !has_attribute(mailbox, "\\Noselect") -} - -fn has_attribute(mailbox: &MailboxListing, expected: &str) -> bool { - mailbox - .attributes - .iter() - .any(|attribute| attribute.eq_ignore_ascii_case(expected)) -} - pub fn require_exact_source(fetches: Vec, requested_uid: u32) -> Result> { let mut matching = fetches .into_iter() @@ -278,7 +308,7 @@ pub fn render_receipt(receipt: &DraftReceipt) -> String { format!( "Draft saved: Account={} | Folder={} | UID={} | To={} | Cc={} | Bcc={} | Subject={}", sanitize_receipt_field(&receipt.account), - sanitize_receipt_field(&receipt.folder), + crate::display::sanitize_folder_name(&receipt.folder), receipt.uid, render_receipt_list(&receipt.to), render_receipt_list(&receipt.cc), @@ -1491,18 +1521,22 @@ Content-Transfer-Encoding: base64\r\n\r\n\ MailboxListing { name: "Nested/Entwürfe".to_string(), attributes: vec!["\\dRaFtS".to_string(), "\\HasNoChildren".to_string()], + delimiter: None, }, MailboxListing { name: "Disabled".to_string(), attributes: vec!["\\DRAFTS".to_string(), "\\NOSELECT".to_string()], + delimiter: None, }, MailboxListing { name: "Configured".to_string(), attributes: Vec::new(), + delimiter: None, }, MailboxListing { name: "Command".to_string(), attributes: Vec::new(), + delimiter: None, }, ]; @@ -1529,10 +1563,12 @@ Content-Transfer-Encoding: base64\r\n\r\n\ MailboxListing { name: "Drafts".to_string(), attributes: vec!["\\Drafts".to_string()], + delimiter: None, }, MailboxListing { name: "Other/Drafts".to_string(), attributes: vec!["\\drafts".to_string()], + delimiter: None, }, ]; assert!(resolve_destination(None, None, &mailboxes).is_err()); diff --git a/src/export.rs b/src/export.rs index 80bf18e..fd060f2 100644 --- a/src/export.rs +++ b/src/export.rs @@ -5,7 +5,7 @@ use std::path::{Path, PathBuf}; use crate::attachment::{create_output_file, write_created_file}; use crate::connection::ImapSession; -use crate::display::{sanitize_terminal_field, MessageRow}; +use crate::display::{sanitize_folder_name, sanitize_terminal_field, MessageRow}; use crate::search; /// Longest file name component accepted by common filesystems. @@ -39,7 +39,7 @@ fn export_filename(folder: &str, uid: u32) -> Result { if filename.len() > MAX_FILENAME_BYTES { bail!( "Export file name for UID {uid} in '{}' exceeds {MAX_FILENAME_BYTES} bytes", - sanitize_terminal_field(folder) + sanitize_folder_name(folder) ); } Ok(filename) @@ -299,7 +299,7 @@ pub fn export_messages( let fetches = session.uid_fetch(chunk, "BODY.PEEK[]").with_context(|| { format!( "Failed to fetch messages from '{}'", - sanitize_terminal_field(folder) + sanitize_folder_name(folder) ) })?; diff --git a/src/lib.rs b/src/lib.rs index 4cf8ce7..1b3d4c2 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -7,3 +7,4 @@ pub mod draft; pub mod export; pub mod read; pub mod search; +pub mod utf7; diff --git a/src/main.rs b/src/main.rs index 9b0df9a..ac0e88a 100644 --- a/src/main.rs +++ b/src/main.rs @@ -83,7 +83,7 @@ enum Commands { Export(ExportArgs), /// Search + set/unset flags on matching messages Mark(MarkArgs), - /// Count matching messages (no FETCH) + /// Count matching messages (no header or body fetch) Count(CountArgs), /// Show mailbox quota usage Quota, @@ -203,7 +203,7 @@ struct FilterArgs { #[arg(short, long, conflicts_with = "all_folders")] folder: Option, - /// Search across all folders (excludes Trash, Junk/Spam, and All Mail) + /// Search across all folders (excludes Trash, Junk/Spam, All Mail, and folders inside them) #[arg(long)] all_folders: bool, @@ -440,6 +440,7 @@ impl FilterArgs { answered: self.answered, draft: self.draft, limit, + client_order: false, } } } @@ -718,15 +719,35 @@ fn resolve_drafts_folder( ) } +/// The folder `count`, `reply`, and `attachments` open: the listed name when +/// the server lists the requested one (in either form), else the name as +/// typed, since servers may open mailboxes they do not list or match names +/// case-insensitively. Returns the name to send and its terminal form. +fn source_folder( + session: &mut connection::ImapSession, + requested: &str, +) -> Result<(String, String)> { + Ok(match search::lookup_folder(session, requested)? { + Some(listed) => { + let shown = display::sanitize_folder_name(&listed); + (listed, shown) + } + None => ( + requested.to_string(), + display::sanitize_terminal_field(requested), + ), + }) +} + +/// The source message's raw bytes and its folder's terminal form. fn fetch_source_message( session: &mut connection::ImapSession, folder: &str, uid: u32, -) -> Result> { - search::validate_folder_name(folder)?; - let safe_folder = display::sanitize_terminal_field(folder); +) -> Result<(String, Vec)> { + let (folder, safe_folder) = source_folder(session, folder)?; session - .examine(folder) + .examine(&folder) .map_err(|_| anyhow::anyhow!("Failed to examine source folder '{safe_folder}'"))?; let fetches = session .uid_fetch(&uid.to_string(), "BODY.PEEK[]") @@ -740,7 +761,7 @@ fn fetch_source_message( body: fetch.body().map(<[u8]>::to_vec), }) .collect(); - draft::require_exact_source(messages, uid) + Ok((safe_folder, draft::require_exact_source(messages, uid)?)) } fn draft_receipt( @@ -866,7 +887,7 @@ fn cmd_reply(account: &config::ResolvedAccount, args: &ReplyArgs) -> Result<()> let mut session = factory.connect()?; let folder = resolve_drafts_folder(&mut session, account, args.drafts_folder.as_deref())?; let source_folder = args.folder.as_deref().unwrap_or(&account.default_folder); - let source = fetch_source_message(&mut session, source_folder, args.uid)?; + let (_, source) = fetch_source_message(&mut session, source_folder, args.uid)?; let composed = draft::compose_reply_draft_with_attachments( draft::ReplyDraftInput { sender, @@ -892,14 +913,13 @@ fn cmd_attachments( default_folder: &str, ) -> Result<()> { let folder = args.folder.as_deref().unwrap_or(default_folder); - let safe_folder = display::sanitize_terminal_field(folder); let sp = spinner(&format!("Inspecting attachments for UID {}...", args.uid)); - let inspected = fetch_source_message(session, folder, args.uid).and_then(|raw| { + let inspected = fetch_source_message(session, folder, args.uid).and_then(|(shown, raw)| { attachment::attachments_from_message(&raw, if args.save { &args.part } else { &[] }) .with_context(|| { format!( "Failed to inspect attachments for UID {} in '{}'", - args.uid, safe_folder + args.uid, shown ) }) }); @@ -974,7 +994,7 @@ fn tag_messages(messages: &mut [display::MessageRow], account: &config::Resolved } fn sort_and_limit_messages(messages: &mut Vec, limit: Option) { - messages.sort_by_key(|message| std::cmp::Reverse(message.timestamp)); + messages.sort_by_key(|message| std::cmp::Reverse(message.sort_time())); if let Some(n) = limit { messages.truncate(n); } @@ -990,8 +1010,9 @@ fn search_accounts( for account in accounts { let mut messages = with_account_session(account, |session| { let sp = spinner(&format!("Searching {}...", safe_label(account))); - let criteria = filter.to_criteria(limit, &account.default_folder); - let result = search::search(session, &criteria); + let mut criteria = filter.to_criteria(limit, &account.default_folder); + criteria.client_order = accounts.len() > 1; + let result = search::search(session, &mut criteria); sp.finish_and_clear(); result })?; @@ -1149,8 +1170,10 @@ fn fetch_status_rows( ) -> Result> { let folders = session.list_all().context("Failed to list folders")?; + // Containers such as Gmail's `[Gmail]` hold no messages and refuse STATUS. Ok(folders .into_iter() + .filter(|mailbox| mailbox.is_selectable()) .map(|mailbox| { // A failed or malformed STATUS leaves counts unknown rather than zero. let counts = session.status_counts(&mailbox.name).unwrap_or_default(); @@ -1193,7 +1216,7 @@ fn display_status_rows(rows: &[StatusRow], include_account: bool) { row.account.as_deref().unwrap_or(""), ))); } - cells.push(Cell::new(display::sanitize_terminal_field(&row.folder))); + cells.push(Cell::new(display::sanitize_folder_name(&row.folder))); cells.push(status_cell(row.messages)); cells.push(status_cell(row.unseen)); cells.push(status_cell(row.recent)); @@ -1242,9 +1265,9 @@ fn cmd_export( default_folder: &str, account_name: Option<&str>, ) -> Result<()> { - let criteria = args.filter.to_criteria(args.limit, default_folder); + let mut criteria = args.filter.to_criteria(args.limit, default_folder); let sp = spinner("Searching..."); - let mut messages = search::search(session, &criteria)?; + let mut messages = search::search(session, &mut criteria)?; sp.finish_and_clear(); if let Some(account) = account_name { for msg in &mut messages { @@ -1351,9 +1374,9 @@ fn cmd_mark( ) -> Result<()> { validate_mark_flags(args.read, args.unread, args.set_flagged, args.clear_flagged)?; - let criteria = args.filter.to_criteria(args.limit, default_folder); + let mut criteria = args.filter.to_criteria(args.limit, default_folder); let sp = spinner("Searching..."); - let mut messages = search::search(session, &criteria)?; + let mut messages = search::search(session, &mut criteria)?; sp.finish_and_clear(); if let Some(account) = account_name { for msg in &mut messages { @@ -1411,7 +1434,7 @@ fn cmd_mark( }; format!( "Failed to store flags in '{}' ({total} already updated{partial})", - display::sanitize_terminal_field(folder) + display::sanitize_folder_name(folder) ) }; search::select_verified(session, folder, *uid_validity) @@ -1447,6 +1470,32 @@ struct CountRow { account: Option, folder: Option, count: usize, + /// Messages not already counted in an earlier folder of the account. + /// Differs from `count` only on Gmail, which lists a message once per + /// label. Totals add these. + new: usize, +} + +/// How many of `uids` (matches in the selected Gmail folder) have a Gmail +/// message ID not in `seen`, which collects them. UIDs without an ID count +/// as new. +fn count_new_gmail_messages( + session: &mut connection::ImapSession, + uids: &std::collections::HashSet, + seen: &mut std::collections::HashSet, +) -> Result { + let sorted: Vec = uids.iter().copied().collect(); + let mut identified = 0; + let mut new = 0; + for chunk in search::build_uid_set(&sorted) { + for (uid, id) in session.gmail_message_ids(&chunk)? { + if uids.contains(&uid) { + identified += 1; + new += usize::from(seen.insert(id)); + } + } + } + Ok(uids.len() - identified + new) } fn count_rows_for_account( @@ -1463,9 +1512,10 @@ fn count_rows_for_account( let folder_names = search::searchable_folders(session)?; let mut results = Vec::new(); + let mut seen_gmail_ids = std::collections::HashSet::new(); for folder in &folder_names { - let safe_folder = display::sanitize_terminal_field(folder); + let safe_folder = display::sanitize_folder_name(folder); if let Err(e) = search::validate_folder_name(folder) { eprintln!("Warning: skipping folder '{safe_folder}': {e}"); continue; @@ -1481,10 +1531,23 @@ fn count_rows_for_account( Ok(uids) => { let count = uids.len(); if count > 0 { + let new = if session.has_capability("X-GM-EXT-1") { + count_new_gmail_messages(session, &uids, &mut seen_gmail_ids) + .unwrap_or_else(|e| { + eprintln!( + "Warning: could not identify Gmail label duplicates in '{safe_folder}': {}", + display::sanitize_terminal_field(&format!("{e:#}")) + ); + count + }) + } else { + count + }; results.push(CountRow { account: account.name.clone(), folder: Some(folder.clone()), count, + new, }); } } @@ -1499,19 +1562,17 @@ fn count_rows_for_account( Ok(results) } else { - search::validate_folder_name(&criteria.folder)?; - session.examine(&criteria.folder).with_context(|| { - format!( - "Failed to examine '{}'", - display::sanitize_terminal_field(&criteria.folder) - ) - })?; + let (folder, safe_folder) = source_folder(session, &criteria.folder)?; + session + .examine(&folder) + .with_context(|| format!("Failed to examine '{safe_folder}'"))?; let uids = session.uid_search(&query).context("IMAP SEARCH failed")?; Ok(vec![CountRow { account: account.name.clone(), - folder: Some(criteria.folder), + folder: Some(folder), count: uids.len(), + new: uids.len(), }]) } } @@ -1522,7 +1583,7 @@ fn display_count_json( all_folders: bool, include_account: bool, ) { - let total: usize = rows.iter().map(|row| row.count).sum(); + let total: usize = rows.iter().map(|row| row.new).sum(); if !include_account { if all_folders { @@ -1569,7 +1630,7 @@ fn display_count_json( }) }) .collect(); - let account_total: usize = account_rows.iter().map(|row| row.count).sum(); + let account_total: usize = account_rows.iter().map(|row| row.new).sum(); serde_json::json!({ "account": account.label(), "folders": folders, @@ -1601,7 +1662,7 @@ fn display_count_json( } fn display_count_text(rows: &[CountRow], all_folders: bool, include_account: bool) { - let total: usize = rows.iter().map(|row| row.count).sum(); + let total: usize = rows.iter().map(|row| row.new).sum(); if !include_account { if rows.is_empty() { @@ -1611,7 +1672,7 @@ fn display_count_text(rows: &[CountRow], all_folders: bool, include_account: boo println!( "{} message(s) in {}", row.count, - display::sanitize_terminal_field(row.folder.as_deref().unwrap_or("")) + display::sanitize_folder_name(row.folder.as_deref().unwrap_or("")) ); } if rows.len() > 1 { @@ -1621,7 +1682,7 @@ fn display_count_text(rows: &[CountRow], all_folders: bool, include_account: boo println!( "{} message(s) in {}", row.count, - display::sanitize_terminal_field(row.folder.as_deref().unwrap_or("")) + display::sanitize_folder_name(row.folder.as_deref().unwrap_or("")) ); } return; @@ -1645,7 +1706,7 @@ fn display_count_text(rows: &[CountRow], all_folders: bool, include_account: boo row.account.as_deref().unwrap_or(""), ))]; if all_folders { - cells.push(Cell::new(display::sanitize_terminal_field( + cells.push(Cell::new(display::sanitize_folder_name( row.folder.as_deref().unwrap_or(""), ))); } @@ -1707,16 +1768,17 @@ fn cmd_read_accounts(accounts: &[config::ResolvedAccount], args: &ReadArgs) -> R let (mut account_messages, fetched, folder) = with_account_session(account, |session| { let mut criteria = args.filter.to_criteria(limit, &account.default_folder); criteria.uid = args.uid; + criteria.client_order = accounts.len() > 1; let sp = spinner(&format!("Searching {}...", safe_label(account))); - let result = search::search(session, &criteria); + let result = search::search(session, &mut criteria); sp.finish_and_clear(); let mut account_messages = result?; if let Some(uid) = args.uid { if account_messages.is_empty() { bail!( "No message with UID {uid} matches in '{}'", - display::sanitize_terminal_field(&criteria.folder) + display::sanitize_folder_name(&criteria.folder) ); } } @@ -1737,7 +1799,9 @@ fn cmd_read_accounts(accounts: &[config::ResolvedAccount], args: &ReadArgs) -> R messages.append(&mut account_messages); } - sort_and_limit_messages(&mut messages, limit); + if accounts.len() > 1 { + sort_and_limit_messages(&mut messages, limit); + } if args.json { println!( "{}", @@ -1832,14 +1896,14 @@ fn run() -> Result<()> { Commands::Delete(args) => { let account = &accounts[0]; with_account_session(account, |session| { - let criteria = args.filter.to_criteria(args.limit, &account.default_folder); + let mut criteria = args.filter.to_criteria(args.limit, &account.default_folder); let trash = args .trash_folder .as_deref() .unwrap_or(&account.trash_folder); delete::delete_with_account( session, - &criteria, + &mut criteria, trash, args.yes, args.dry_run, @@ -1850,10 +1914,10 @@ fn run() -> Result<()> { Commands::Move(args) => { let account = &accounts[0]; with_account_session(account, |session| { - let criteria = args.filter.to_criteria(args.limit, &account.default_folder); + let mut criteria = args.filter.to_criteria(args.limit, &account.default_folder); delete::search_and_move_with_account( session, - &criteria, + &mut criteria, &args.dest, args.yes, args.dry_run, diff --git a/src/read.rs b/src/read.rs index 54651f0..bb3498a 100644 --- a/src/read.rs +++ b/src/read.rs @@ -5,7 +5,9 @@ use serde::Serialize; use std::collections::HashMap; use crate::connection::ImapSession; -use crate::display::{sanitize_terminal_body, sanitize_terminal_field, MessageRow}; +use crate::display::{ + sanitize_folder_name, sanitize_terminal_body, sanitize_terminal_field, MessageRow, +}; use crate::search; pub type MessageBodyMap = HashMap<(Option, String, u32), Vec>; @@ -59,7 +61,7 @@ pub fn fetch_message_bodies( let fetches = session.uid_fetch(chunk, "BODY.PEEK[]").with_context(|| { format!( "Failed to fetch messages from '{}'", - sanitize_terminal_field(folder) + sanitize_folder_name(folder) ) })?; @@ -547,6 +549,8 @@ mod tests { answered: false, flagged: false, uid_validity: None, + gmail_msgid: None, + arrival: None, } } diff --git a/src/search.rs b/src/search.rs index 662ee59..a1c90a4 100644 --- a/src/search.rs +++ b/src/search.rs @@ -1,13 +1,21 @@ use anyhow::{bail, Context, Result}; use regex::Regex; -use std::collections::BTreeMap; +use std::collections::{BTreeMap, HashMap, HashSet}; use crate::connection::ImapSession; -use crate::display::{sanitize_terminal_field, MessageRow}; +use crate::display::{ + sanitize_folder_name, sanitize_terminal_field, MessageRow, ARRIVAL_SLACK_SECS, +}; use crate::draft::MailboxListing; use imap::types::Flag; -pub const LITERAL_PLUS_REQUIRED: &str = "Non-ASCII search requires server support for LITERAL+"; +pub const NON_ASCII_SEARCH_UNSUPPORTED: &str = + "Non-ASCII search requires server support for LITERAL+ or LITERAL-"; +pub const LONG_LITERAL_UNSUPPORTED: &str = + "Non-ASCII search terms over 4096 bytes require server support for LITERAL+"; + +/// Largest non-synchronizing literal a LITERAL- server accepts (RFC 7888 §5). +const LITERAL_MINUS_MAX: usize = 4096; pub struct SearchCriteria { pub folder: String, @@ -30,6 +38,10 @@ pub struct SearchCriteria { pub answered: bool, pub draft: bool, pub limit: Option, + /// Order rows by [`MessageRow::sort_time`] on the client instead of with + /// server SORT, because they will be merged with other accounts' rows by + /// that key (`--all-accounts`); SORT orders by the uncapped Date header. + pub client_order: bool, } /// Strip CRLF and control chars to prevent IMAP command injection. @@ -57,15 +69,33 @@ fn quote_search_term(value: &str) -> String { } /// Reject a built query the connected server cannot receive. Non-ASCII -/// terms are sent as LITERAL+ literals; without that extension the search -/// fails here rather than silently matching nothing. +/// terms are sent as non-synchronizing literals: any size with LITERAL+, at +/// most 4096 bytes with LITERAL-. Without either the search fails here +/// rather than silently matching nothing. pub fn ensure_query_supported(session: &ImapSession, query: &str) -> Result<()> { - if !query.is_ascii() && !session.has_capability("LITERAL+") { - bail!(LITERAL_PLUS_REQUIRED); + if query.is_ascii() || session.has_capability("LITERAL+") { + return Ok(()); + } + if !session.has_capability("LITERAL-") { + bail!(NON_ASCII_SEARCH_UNSUPPORTED); + } + if literal_lengths(query).any(|length| length > LITERAL_MINUS_MAX) { + bail!(LONG_LITERAL_UNSUPPORTED); } Ok(()) } +/// Byte lengths of the `{N+}` literals in a built query. Terms are stripped +/// of control characters, so every CRLF ends a literal prefix; text after +/// the last CRLF is literal data or a quoted term, never a prefix. +fn literal_lengths(query: &str) -> impl Iterator + '_ { + let prefixes = query.rsplit_once("\r\n").map_or("", |(before, _)| before); + prefixes.split("\r\n").filter_map(|before| { + let (_, digits) = before.strip_suffix("+}")?.rsplit_once('{')?; + digits.parse().ok() + }) +} + /// Mailbox names are sent to the server unchanged; reject names that cannot /// be sent safely instead of silently altering them. pub fn validate_folder_name(folder: &str) -> Result<()> { @@ -124,10 +154,7 @@ fn resolve_relative_date(s: &str) -> Option> { }; let unit = &caps[2]; - let now_secs = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_secs() as i64; + let now_secs = now_epoch_secs(); Some(match unit { "d" | "w" => { @@ -153,6 +180,13 @@ fn resolve_relative_date(s: &str) -> Option> { }) } +fn now_epoch_secs() -> i64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_secs() as i64 +} + fn days_in_month(year: i64, month: u32) -> u32 { match month { 1 | 3 | 5 | 7 | 8 | 10 | 12 => 31, @@ -409,157 +443,258 @@ pub fn build_uid_set(uids: &[u32]) -> Vec { chunks } +/// Rows of one folder matching `query`, newest first, at most `limit`. +/// With `merged` (one folder of `--all-folders`), rows carry their folder. +/// With `server_sort`, the server's SORT order is used when available; +/// otherwise rows are ordered by [`MessageRow::sort_time`], the key results +/// are merged by, so cutting each folder or account to `limit` keeps the +/// overall newest. fn fetch_messages( session: &mut ImapSession, folder: &str, query: &str, - include_folder: bool, + merged: bool, + server_sort: bool, limit: Option, ) -> Result> { validate_folder_name(folder)?; - let safe_folder = sanitize_terminal_field(folder); - let opened = session - .examine(folder) - .with_context(|| format!("Failed to examine folder '{safe_folder}'"))?; - let uid_validity = opened.uid_validity; - - // Try server-side SORT first, fall back to SEARCH + client sort - let (ordered_uids, pre_sorted) = match try_uid_sort(session, query)? { + let opened = session.examine(folder).with_context(|| { + format!( + "Failed to examine folder '{}'", + sanitize_folder_name(folder) + ) + })?; + let rows = RowFetch { + folder, + include_folder: merged, + uid_validity: opened.uid_validity, + }; + + let sorted = if server_sort { + try_uid_sort(session, query)? + } else { + None + }; + let mut messages = match sorted { + // Server SORT order is kept, and the limit applies before FETCH. Some(mut uids) => { - // With server SORT, we can truncate before FETCH if let Some(n) = limit { uids.truncate(n); } - (uids, true) - } - None => { - let uid_set = session.uid_search(query).context("IMAP SEARCH failed")?; - let mut uids: Vec = uid_set.into_iter().collect(); - uids.sort(); - (uids, false) + let mut by_uid = rows.fetch(session, &uids)?; + uids.iter().filter_map(|uid| by_uid.remove(uid)).collect() } + None => newest_by_date(session, query, limit, &rows)?, }; - if ordered_uids.is_empty() { + // Gmail lists a message in every label folder; multi-folder results + // carry its message ID so `search_excluding` can keep one row. + if merged && !messages.is_empty() && session.has_capability("X-GM-EXT-1") { + let uids: Vec = messages.iter().map(|message| message.uid).collect(); + let mut ids = HashMap::new(); + for chunk in &build_uid_set(&uids) { + ids.extend( + session + .gmail_message_ids(chunk) + .context("IMAP FETCH X-GM-MSGID failed")?, + ); + } + for message in &mut messages { + message.gmail_msgid = ids.get(&message.uid).copied(); + } + } + Ok(messages) +} + +/// Below this many matches, fetching every header is cheaper than extra +/// SEARCH round trips. +const NARROW_ABOVE: usize = 500; +/// Recent windows tried, in days, before fetching every match. +const RECENT_WINDOW_DAYS: [i64; 3] = [7, 31, 366]; +const DAY_SECS: i64 = 86_400; + +/// Without SORT: matching rows, newest first by [`MessageRow::sort_time`] +/// (ties: newest UID first), truncated to `limit`. Headers cost time per +/// message (about 1 ms each on Gmail), so with a limit and many matches the +/// search is first narrowed with SINCE to recent windows (SINCE uses the +/// arrival date, which servers index; SENTSINCE parses every Date header +/// and took 15 s on a 147,000-message Proton Bridge folder). A message +/// outside a window starting on day `d` arrived before `d` + 1 day (its +/// zone shifts the day by less than one) and sorts at most +/// [`ARRIVAL_SLACK_SECS`] after arriving. Once `limit` rows sort at or after +/// `d` + 1 day + that slack, they are the newest. +fn newest_by_date( + session: &mut ImapSession, + query: &str, + limit: Option, + rows: &RowFetch, +) -> Result> { + if limit == Some(0) { return Ok(Vec::new()); } + let uids: Vec = session + .uid_search(query) + .context("IMAP SEARCH failed")? + .into_iter() + .collect(); + let mut fetched = HashMap::new(); + + if let Some(n) = limit.filter(|&n| uids.len() > n.max(NARROW_ABOVE)) { + let now = now_epoch_secs(); + for days in RECENT_WINDOW_DAYS { + let day_start = (now - days * DAY_SECS).div_euclid(DAY_SECS) * DAY_SECS; + let (year, month, day) = epoch_to_date(day_start); + let window = session + .uid_search(&format!( + "{query} SINCE {}", + format_imap_date(day, month, year)? + )) + .context("IMAP SEARCH failed")?; + if window.len() < n { + continue; + } + let missing: Vec = window + .iter() + .copied() + .filter(|uid| !fetched.contains_key(uid)) + .collect(); + fetched.extend(rows.fetch(session, &missing)?); + let mut newest: Vec<(i64, u32)> = window + .iter() + .filter_map(|uid| fetched.get(uid).map(|row| (row.sort_time(), *uid))) + .collect(); + newest.sort_unstable_by(|a, b| b.cmp(a)); + if newest + .get(n - 1) + .is_some_and(|(time, _)| *time >= day_start + DAY_SECS + ARRIVAL_SLACK_SECS) + { + return Ok(newest[..n] + .iter() + .filter_map(|(_, uid)| fetched.remove(uid)) + .collect()); + } + } + } + + let missing: Vec = uids + .into_iter() + .filter(|uid| !fetched.contains_key(uid)) + .collect(); + fetched.extend(rows.fetch(session, &missing)?); + let mut messages: Vec = fetched.into_values().collect(); + messages.sort_unstable_by_key(|message| std::cmp::Reverse((message.sort_time(), message.uid))); + if let Some(n) = limit { + messages.truncate(n); + } + Ok(messages) +} + +/// Builds header rows for UIDs of the selected mailbox. +struct RowFetch<'a> { + folder: &'a str, + include_folder: bool, + uid_validity: Option, +} - let uid_chunks = build_uid_set(&ordered_uids); - let requested: std::collections::HashSet = ordered_uids.iter().copied().collect(); - - // FETCH results may come back in arbitrary order; index by UID - let mut by_uid = std::collections::HashMap::new(); - for chunk in &uid_chunks { - let mut warned_invalid_uid = false; - let fetches = session - .uid_fetch( - chunk, - "(UID FLAGS RFC822.SIZE BODY.PEEK[HEADER.FIELDS (Subject From Date Message-ID In-Reply-To References)])", - ) - .context("IMAP FETCH failed")?; - - for fetch in fetches.iter() { - let uid = match fetch.uid { - Some(u) if u > 0 => u, - _ => { - if !warned_invalid_uid { - eprintln!( - "Warning: skipping fetched message(s) with missing/invalid UID in '{safe_folder}'" - ); - warned_invalid_uid = true; +impl RowFetch<'_> { + /// Rows for `uids`, keyed by UID. Servers may interleave unsolicited + /// FETCH responses (flag changes made by other clients): only requested + /// UIDs become rows, and a flags-only response never replaces a row that + /// has headers. + fn fetch(&self, session: &mut ImapSession, uids: &[u32]) -> Result> { + let requested: HashSet = uids.iter().copied().collect(); + let mut by_uid = HashMap::new(); + for chunk in &build_uid_set(uids) { + let mut warned_invalid_uid = false; + let fetches = session + .uid_fetch( + chunk, + "(UID FLAGS RFC822.SIZE INTERNALDATE BODY.PEEK[HEADER.FIELDS (Subject From Date Message-ID In-Reply-To References)])", + ) + .context("IMAP FETCH failed")?; + + for fetch in fetches.iter() { + let uid = match fetch.uid { + Some(u) if u > 0 => u, + _ => { + if !warned_invalid_uid { + eprintln!( + "Warning: skipping fetched message(s) with missing/invalid UID in '{}'", + sanitize_folder_name(self.folder) + ); + warned_invalid_uid = true; + } + continue; } + }; + if !requested.contains(&uid) + || (fetch.header().is_none() && by_uid.contains_key(&uid)) + { continue; } - }; - // Servers may interleave unsolicited FETCH responses (flag changes - // made by other clients). Only searched UIDs become rows, and a - // flags-only response never replaces a row that has headers. - if !requested.contains(&uid) || (fetch.header().is_none() && by_uid.contains_key(&uid)) - { - continue; + by_uid.insert(uid, self.row(uid, fetch)); } - let size = fetch.size.unwrap_or(0); - let header_bytes = fetch.header().unwrap_or(b""); - let header_str = String::from_utf8_lossy(header_bytes); - - let (mut subject, mut from, mut date) = (String::new(), String::new(), String::new()); - let (mut message_id, mut in_reply_to, mut references) = (None, Vec::new(), Vec::new()); - - let parsed = mailparse::parse_headers(header_bytes); - if let Ok((headers, _)) = parsed { - for h in &headers { - match h.get_key().to_lowercase().as_str() { - "subject" => subject = h.get_value(), - "from" => from = h.get_value(), - "date" => date = h.get_value(), - "message-id" => message_id = message_ids(&h.get_value()).into_iter().next(), - "in-reply-to" => in_reply_to = message_ids(&h.get_value()), - "references" => references = message_ids(&h.get_value()), - _ => {} - } + } + Ok(by_uid) + } + + fn row(&self, uid: u32, fetch: &imap::types::Fetch<'_>) -> MessageRow { + let header_bytes = fetch.header().unwrap_or(b""); + let (mut subject, mut from, mut date) = (String::new(), String::new(), String::new()); + let (mut message_id, mut in_reply_to, mut references) = (None, Vec::new(), Vec::new()); + + if let Ok((headers, _)) = mailparse::parse_headers(header_bytes) { + for h in &headers { + match h.get_key().to_lowercase().as_str() { + "subject" => subject = h.get_value(), + "from" => from = h.get_value(), + "date" => date = h.get_value(), + "message-id" => message_id = message_ids(&h.get_value()).into_iter().next(), + "in-reply-to" => in_reply_to = message_ids(&h.get_value()), + "references" => references = message_ids(&h.get_value()), + _ => {} } - } else { - for line in header_str.lines() { - if let Some(v) = line.strip_prefix("Subject: ") { - subject = v.to_string(); - } else if let Some(v) = line.strip_prefix("From: ") { - from = v.to_string(); - } else if let Some(v) = line.strip_prefix("Date: ") { - date = v.to_string(); - } else if let Some(v) = line.strip_prefix("Message-ID: ") { - message_id = message_ids(v).into_iter().next(); - } else if let Some(v) = line.strip_prefix("In-Reply-To: ") { - in_reply_to = message_ids(v); - } else if let Some(v) = line.strip_prefix("References: ") { - references = message_ids(v); - } + } + } else { + for line in String::from_utf8_lossy(header_bytes).lines() { + if let Some(v) = line.strip_prefix("Subject: ") { + subject = v.to_string(); + } else if let Some(v) = line.strip_prefix("From: ") { + from = v.to_string(); + } else if let Some(v) = line.strip_prefix("Date: ") { + date = v.to_string(); + } else if let Some(v) = line.strip_prefix("Message-ID: ") { + message_id = message_ids(v).into_iter().next(); + } else if let Some(v) = line.strip_prefix("In-Reply-To: ") { + in_reply_to = message_ids(v); + } else if let Some(v) = line.strip_prefix("References: ") { + references = message_ids(v); } } - - let flags = fetch.flags(); - let has_flag = |wanted, name| has_system_flag(flags, wanted, name); - - let timestamp = mailparse::dateparse(&date).unwrap_or(0); - by_uid.insert( - uid, - MessageRow { - account: None, - uid, - folder: if include_folder { - Some(folder.to_string()) - } else { - None - }, - from, - subject, - date, - timestamp, - size, - message_id, - in_reply_to, - references, - seen: has_flag(Flag::Seen, "\\Seen"), - answered: has_flag(Flag::Answered, "\\Answered"), - flagged: has_flag(Flag::Flagged, "\\Flagged"), - uid_validity, - }, - ); } - } - if pre_sorted { - // Preserve server SORT order - Ok(ordered_uids - .into_iter() - .filter_map(|uid| by_uid.remove(&uid)) - .collect()) - } else { - let mut messages: Vec = by_uid.into_values().collect(); - messages.sort_by_key(|message| std::cmp::Reverse(message.timestamp)); - if let Some(n) = limit { - messages.truncate(n); + let flags = fetch.flags(); + let has_flag = |wanted, name| has_system_flag(flags, wanted, name); + let timestamp = mailparse::dateparse(&date).unwrap_or(0); + MessageRow { + account: None, + uid, + folder: self.include_folder.then(|| self.folder.to_string()), + from, + subject, + date, + timestamp, + size: fetch.size.unwrap_or(0), + message_id, + in_reply_to, + references, + seen: has_flag(Flag::Seen, "\\Seen"), + answered: has_flag(Flag::Answered, "\\Answered"), + flagged: has_flag(Flag::Flagged, "\\Flagged"), + uid_validity: self.uid_validity, + gmail_msgid: None, + arrival: fetch.internal_date().map(|date| date.timestamp()), } - Ok(messages) } } @@ -629,15 +764,20 @@ pub fn message_ids(value: &str) -> Vec { ids } -/// Mailboxes excluded from `--all-folders`: special-use `\All`, `\Trash`, and -/// `\Junk` mailboxes, and well-known aggregate, trash, and spam names (exact, -/// case-insensitive) for servers without special-use attributes. -pub fn folders_to_skip(mailbox: &MailboxListing) -> bool { - if mailbox.attributes.iter().any(|attribute| { - ["\\All", "\\Trash", "\\Junk"] - .iter() - .any(|special| attribute.eq_ignore_ascii_case(special)) - }) { +/// Trash, spam, and aggregate mailboxes `--all-folders` never searches: +/// special-use `\All`, `\Trash`, and `\Junk` mailboxes, and well-known +/// names (exact, case-insensitive) for servers without special-use +/// attributes. On Proton Mail Bridge, also the label views: the `Labels` +/// container and `\Flagged` (`Starred`), since every message there also +/// sits in exactly one regular folder. +fn is_excluded_special(mailbox: &MailboxListing, proton_bridge: bool) -> bool { + if ["\\All", "\\Trash", "\\Junk"] + .iter() + .any(|special| mailbox.has_attribute(special)) + { + return true; + } + if proton_bridge && (mailbox.name == "Labels" || mailbox.has_attribute("\\Flagged")) { return true; } let lower = mailbox.name.to_lowercase(); @@ -653,6 +793,26 @@ pub fn folders_to_skip(mailbox: &MailboxListing) -> bool { ) } +/// The listed names `--all-folders` searches: selectable mailboxes that are +/// neither excluded special mailboxes nor nested in one (`Trash/Old`). +/// Children of other containers that cannot be opened, such as Gmail's +/// `[Gmail]`, are searched. +fn searchable(folders: &[MailboxListing], proton_bridge: bool) -> Vec { + let excluded: Vec<&MailboxListing> = folders + .iter() + .filter(|mailbox| is_excluded_special(mailbox, proton_bridge)) + .collect(); + folders + .iter() + .filter(|mailbox| { + mailbox.is_selectable() + && !is_excluded_special(mailbox, proton_bridge) + && !excluded.iter().any(|parent| parent.contains(mailbox)) + }) + .map(|mailbox| mailbox.name.clone()) + .collect() +} + /// Whether two mailbox names refer to the same mailbox: exact comparison, /// except INBOX, which IMAP defines as case-insensitive. pub fn same_mailbox(a: &str, b: &str) -> bool { @@ -662,79 +822,111 @@ pub fn same_mailbox(a: &str, b: &str) -> bool { /// List every mailbox name included in `--all-folders` operations. pub fn searchable_folders(session: &mut ImapSession) -> Result> { let folders = session.list_all().context("Failed to list folders")?; - Ok(folders - .into_iter() - .filter(|mailbox| !folders_to_skip(mailbox)) - .map(|mailbox| mailbox.name) - .collect()) + Ok(searchable(&folders, session.is_proton_bridge())) } -pub fn search(session: &mut ImapSession, criteria: &SearchCriteria) -> Result> { +/// Search per `criteria`. A single source folder is first resolved to the +/// server's listed name (see [`resolve_folder`]) and `criteria.folder` is +/// updated, so later UID actions use the same mailbox. +pub fn search(session: &mut ImapSession, criteria: &mut SearchCriteria) -> Result> { search_excluding(session, criteria, None) } -/// Search for messages to move to `exclude`: with `--all-folders` that -/// mailbox is not searched, and naming it as the single source is an error. +/// Search for messages to move to `exclude` (a listed name): with +/// `--all-folders` that mailbox is not searched, and naming it as the single +/// source is an error. pub fn search_excluding( session: &mut ImapSession, - criteria: &SearchCriteria, + criteria: &mut SearchCriteria, exclude: Option<&str>, ) -> Result> { let query = build_query(criteria)?; ensure_query_supported(session, &query)?; if criteria.all_folders { - let folder_names = searchable_folders(session)?; - + // INBOX first: of a Gmail message's label copies the first one found + // is kept, and the INBOX copy is the natural one to show and act on. + let mut folder_names = searchable_folders(session)?; + folder_names.sort_by_key(|name| !same_mailbox(name, "INBOX")); + + // Every folder is cut to `limit` in sort-time order, the same key as + // the merge below (stable, so folder order breaks ties), so the cut + // rows include every row of the overall newest `limit`, even after + // Gmail label duplicates are dropped. let mut all_messages = Vec::new(); for folder in &folder_names { if exclude.is_some_and(|excluded| same_mailbox(folder, excluded)) { continue; } - match fetch_messages(session, folder, &query, true, None) { + match fetch_messages(session, folder, &query, true, false, criteria.limit) { Ok(msgs) => all_messages.extend(msgs), Err(e) => { eprintln!( "Warning: skipping folder '{}': {}", - sanitize_terminal_field(folder), + sanitize_folder_name(folder), sanitize_terminal_field(&format!("{e:#}")) ); } } } - all_messages.sort_by_key(|message| std::cmp::Reverse(message.timestamp)); + let mut all_messages = drop_gmail_label_duplicates(all_messages); + all_messages.sort_by_key(|message| std::cmp::Reverse(message.sort_time())); if let Some(n) = criteria.limit { all_messages.truncate(n); } Ok(all_messages) } else { + criteria.folder = resolve_folder(session, &criteria.folder)?; if exclude.is_some_and(|excluded| same_mailbox(&criteria.folder, excluded)) { bail!( "Source and destination folder are the same ('{}')", - sanitize_terminal_field(&criteria.folder) + sanitize_folder_name(&criteria.folder) ); } - ensure_folder_exists(session, &criteria.folder)?; - fetch_messages(session, &criteria.folder, &query, false, criteria.limit) + fetch_messages( + session, + &criteria.folder, + &query, + false, + !criteria.client_order, + criteria.limit, + ) } } -/// Confirm that `folder` names an existing mailbox. The LIST pattern is the -/// constant `*` so user input never acts as a wildcard pattern; names are -/// compared exactly (INBOX case-insensitively, as IMAP defines). -pub fn ensure_folder_exists(session: &mut ImapSession, folder: &str) -> Result<()> { - validate_folder_name(folder)?; +/// Gmail lists a message once per label folder. Keep the first row per +/// Gmail message ID (INBOX is searched first). Rows without an ID (other +/// servers) are all kept. +fn drop_gmail_label_duplicates(messages: Vec) -> Vec { + let mut seen = HashSet::new(); + messages + .into_iter() + .filter(|message| message.gmail_msgid.is_none_or(|id| seen.insert(id))) + .collect() +} + +/// The server's listed name for a user-supplied folder, or `None` if it is +/// not listed. Accepts the listed name itself or its decoded form +/// ([`MailboxListing::find`]). The LIST pattern is the constant `*` so user +/// input never acts as a wildcard pattern. +pub fn lookup_folder(session: &mut ImapSession, requested: &str) -> Result> { + validate_folder_name(requested)?; let folders = session.list_all().context("Failed to list folders")?; - let exists = folders - .iter() - .any(|mailbox| same_mailbox(&mailbox.name, folder)); - if !exists { - bail!( - "Folder '{}' does not exist. Use `slashmail status` to list available folders.", - sanitize_terminal_field(folder) - ); - } - Ok(()) + Ok(MailboxListing::find(&folders, requested).map(|mailbox| mailbox.name.clone())) +} + +/// Like [`lookup_folder`], but a folder that is not listed is an error. +pub fn resolve_folder(session: &mut ImapSession, requested: &str) -> Result { + lookup_folder(session, requested)?.ok_or_else(|| missing_folder(requested)) +} + +/// The error for a user-supplied folder that is not listed. The name is +/// shown as typed, not decoded. +pub fn missing_folder(requested: &str) -> anyhow::Error { + anyhow::anyhow!( + "Folder '{}' does not exist. Use `slashmail status` to list available folders.", + sanitize_terminal_field(requested) + ) } /// Group searched rows by source mailbox with the UIDVALIDITY captured at @@ -750,7 +942,7 @@ pub fn group_message_uids<'a>( Some(value) if value != 0 => value, _ => bail!( "Missing UIDVALIDITY for searched mailbox '{}'", - sanitize_terminal_field(folder) + sanitize_folder_name(folder) ), }; let (expected, uids) = groups @@ -759,7 +951,7 @@ pub fn group_message_uids<'a>( if *expected != uid_validity { bail!( "Searched messages from '{}' have inconsistent UIDVALIDITY", - sanitize_terminal_field(folder) + sanitize_folder_name(folder) ); } uids.push(message.uid); @@ -794,7 +986,7 @@ fn open_verified( writable: bool, ) -> Result<()> { validate_folder_name(folder)?; - let safe_folder = sanitize_terminal_field(folder); + let safe_folder = sanitize_folder_name(folder); let opened = if writable { session.select(folder) } else { @@ -827,6 +1019,17 @@ pub fn existing_uids(session: &mut ImapSession, uid_set: &str) -> Result>(), [12, 10]); + } + #[test] fn sanitize_removes_control_chars() { assert_eq!(sanitize("hello"), "hello"); @@ -1211,6 +1414,8 @@ mod tests { answered: false, flagged: false, uid_validity, + gmail_msgid: None, + arrival: None, } } @@ -1298,6 +1503,7 @@ mod tests { answered: false, draft: false, limit: None, + client_order: false, } } @@ -1363,21 +1569,38 @@ mod tests { } #[test] - fn all_folders_skips_special_use_trash_and_junk_by_attribute() { + fn all_folders_skips_trash_junk_and_aggregates_with_their_children() { let listing = |name: &str, attributes: &[&str]| MailboxListing { name: name.into(), attributes: attributes.iter().map(|a| a.to_string()).collect(), + delimiter: Some("/".into()), }; - assert!(folders_to_skip(&listing("Deleted Items", &["\\Trash"]))); - assert!(folders_to_skip(&listing( - "Junk Email", - &["\\HasNoChildren", "\\junk"] - ))); - assert!(folders_to_skip(&listing("Everything", &["\\All"]))); - assert!(folders_to_skip(&listing("Trash", &[]))); - assert!(!folders_to_skip(&listing("Deleted Items", &[]))); - assert!(!folders_to_skip(&listing("Archive", &["\\Archive"]))); - assert!(!folders_to_skip(&listing("Small mail", &[]))); + let folders = [ + listing("INBOX", &[]), + listing("Deleted Items", &["\\Trash"]), + listing("Deleted Items/Old", &[]), + listing("Deleted Items Archive", &[]), + listing("Junk Email", &["\\HasNoChildren", "\\junk"]), + listing("Everything", &["\\All"]), + listing("[Gmail]", &["\\HasChildren", "\\Noselect"]), + listing("[Gmail]/Starred", &["\\Flagged"]), + listing("[Gmail]/Trash", &["\\HasChildren", "\\Trash"]), + listing("[Gmail]/Trash/MWM", &[]), + listing("Trash", &[]), + listing("Trash/2024", &[]), + listing("Archive", &["\\Archive"]), + listing("Small mail", &[]), + ]; + assert_eq!( + searchable(&folders, false), + [ + "INBOX", + "Deleted Items Archive", + "[Gmail]/Starred", + "Archive", + "Small mail" + ] + ); } #[test] diff --git a/src/utf7.rs b/src/utf7.rs new file mode 100644 index 0000000..fdc3ae0 --- /dev/null +++ b/src/utf7.rs @@ -0,0 +1,211 @@ +//! IMAP modified UTF-7 mailbox names (RFC 3501 §5.1.3). +//! +//! Servers list non-ASCII mailbox names in this encoding (`Envoy&AOk-s`). +//! Slashmail keeps listed names unchanged internally and in JSON, decodes +//! them only for terminal display, and matches user-supplied names against +//! the listing with [`encode`] (see `MailboxListing::find`). + +use std::borrow::Cow; + +const ALPHABET: &[u8; 64] = b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+,"; + +fn is_printable_ascii(character: char) -> bool { + (' '..='~').contains(&character) +} + +/// Decode a wire name, or `None` when it is not canonical modified UTF-7 or +/// would display misleadingly. Rejected: shifted runs that encode printable +/// ASCII or control characters, runs with nothing visible (only spaces, +/// joiners, or invisible or bidi formatting), and adjacent runs (null +/// shifts). So a name with an encoded run never displays as a plain-ASCII +/// name. A rejected name is shown as listed, so on a server that also lists +/// names unencoded (raw UTF-8 or a bare `&`), two names can look the same; +/// likewise two non-ASCII names can look alike (homoglyphs, joiners next to +/// visible characters). +pub fn decode(name: &str) -> Option { + let mut decoded = String::with_capacity(name.len()); + let mut rest = name; + while let Some(character) = rest.chars().next() { + if !is_printable_ascii(character) { + return None; + } + rest = &rest[1..]; + if character != '&' { + decoded.push(character); + continue; + } + let (run, after) = rest.split_once('-')?; + if run.is_empty() { + decoded.push('&'); + } else { + decode_run(run.as_bytes(), &mut decoded)?; + if after.starts_with('&') && !after.starts_with("&-") { + return None; + } + } + rest = after; + } + Some(decoded) +} + +fn decode_run(run: &[u8], decoded: &mut String) -> Option<()> { + let mut units = Vec::with_capacity(run.len() * 6 / 16); + let mut bits: u32 = 0; + let mut bit_count = 0; + for &byte in run { + let value = ALPHABET.iter().position(|&symbol| symbol == byte)? as u32; + bits = (bits << 6) | value; + bit_count += 6; + if bit_count >= 16 { + bit_count -= 16; + units.push((bits >> bit_count) as u16); + bits &= (1 << bit_count) - 1; + } + } + // Only zero padding bits, fewer than one base64 digit, may remain. + if bit_count >= 6 || bits != 0 { + return None; + } + let text = char::decode_utf16(units) + .collect::>() + .ok()?; + if text + .chars() + .any(|character| is_printable_ascii(character) || character.is_control()) + || text.chars().all(crate::display::renders_blank) + { + return None; + } + decoded.push_str(&text); + Some(()) +} + +/// Encode a name: printable ASCII is kept (`&` becomes `&-`) and every other +/// run becomes base64 UTF-16. +pub fn encode(name: &str) -> String { + let mut encoded = String::with_capacity(name.len()); + let mut pending = Vec::new(); + for character in name.chars() { + if is_printable_ascii(character) { + flush_run(&mut pending, &mut encoded); + match character { + '&' => encoded.push_str("&-"), + _ => encoded.push(character), + } + } else { + let mut buffer = [0; 2]; + pending.extend_from_slice(character.encode_utf16(&mut buffer)); + } + } + flush_run(&mut pending, &mut encoded); + encoded +} + +fn flush_run(units: &mut Vec, encoded: &mut String) { + if units.is_empty() { + return; + } + encoded.push('&'); + let mut bits: u32 = 0; + let mut bit_count = 0; + for unit in units.drain(..) { + bits = (bits << 16) | u32::from(unit); + bit_count += 16; + while bit_count >= 6 { + bit_count -= 6; + encoded.push(ALPHABET[((bits >> bit_count) & 0x3f) as usize] as char); + } + bits &= (1 << bit_count) - 1; + } + if bit_count > 0 { + encoded.push(ALPHABET[((bits << (6 - bit_count)) & 0x3f) as usize] as char); + } + encoded.push('-'); +} + +/// A listed name for display: decoded when valid, otherwise unchanged. +pub fn display(name: &str) -> Cow<'_, str> { + match decode(name) { + Some(decoded) => Cow::Owned(decoded), + None => Cow::Borrowed(name), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn rfc_3501_example_round_trips() { + let wire = "~peter/mail/&U,BTFw-/&ZeVnLIqe-"; + let text = "~peter/mail/台北/日本語"; + assert_eq!(decode(wire).as_deref(), Some(text)); + assert_eq!(encode(text), wire); + } + + #[test] + fn padding_lengths_and_surrogate_pairs_round_trip() { + for text in [ + "é", + "Envoyés", + "éé", + "ééé", + "R&D", + "&", + "é&", + "📧 Mail", + "a😀b€", + "❤\u{fe0f} Family", + ] { + let wire = encode(text); + assert!(wire.is_ascii(), "{wire}"); + assert_eq!(decode(&wire).as_deref(), Some(text), "{wire}"); + } + assert_eq!(encode("Messages envoyés"), "Messages envoy&AOk-s"); + assert_eq!(encode("R&D"), "R&-D"); + assert_eq!(encode("é&"), "&AOk-&-"); + } + + #[test] + fn invalid_or_non_canonical_names_do_not_decode() { + for wire in [ + "&AOk", // unterminated shift + "&AOk!-", // character outside the alphabet + "&AOl-", // nonzero padding bits + "&AOkA-", // a whole spare base64 digit + "&2D0-", // lone high surrogate + "&AFQ-rash", // encodes printable ASCII ("T") + "&ABs-AINBOX", // encodes ESC, which would hide "A" + "&AJs-@Trash", // encodes C1 CSI, which would hide "@" + "&AOk-&AOk-", // null shift between runs + "Envoyés", // raw non-ASCII + "Bad\rName", // control character + ] { + assert_eq!(decode(wire), None, "{wire:?}"); + } + } + + #[test] + fn runs_with_nothing_visible_do_not_decode() { + // Each would otherwise display like "Trash" or "Tr ash". + for text in [ + "Trash\u{200c}", + "Tr\u{200b}ash", + "\u{202e}Trash", + "Trash\u{a0}", + "Trash\u{3164}", + "Tr\u{2800}ash", + "Trash\u{180b}", + ] { + assert_eq!(decode(&encode(text)), None, "{text:?}"); + assert_eq!(display(&encode(text)), encode(text)); + } + } + + #[test] + fn display_falls_back_to_the_listed_name() { + assert_eq!(display("Messages envoy&AOk-s"), "Messages envoyés"); + assert_eq!(display("&AFQ-rash"), "&AFQ-rash"); + assert_eq!(display("Ärger"), "Ärger"); + } +} diff --git a/static/index.html b/static/index.html index f2a5f33..a6f8250 100644 --- a/static/index.html +++ b/static/index.html @@ -26,7 +26,7 @@ - + @@ -174,9 +174,9 @@

Install

cargo install slashmail

Or download a binary:

  • @@ -522,7 +522,7 @@

    Your mail, your machine.