fix: decode RFC2047 subjects, OR TO/FROM in to-mode query, tz-aware Date, drop LIST completion line (v0.1.7)

Subject headers were matched/scanned raw (never RFC2047-decoded), so any
provider that base64/quoted-printable encodes non-ASCII subjects silently
failed subject filtering and code extraction. match_field='to' only
server-queried TO, so an older TO-matching message permanently hid a
forwarded-From match the client-side check was documented to accept.
_age_seconds treated a naive datetime (from a "-0000" Date header) as local
time instead of UTC, corrupting max_age freshness on non-UTC hosts. get_folders
appended aioimaplib's tagged LIST completion text as a phantom folder name on
every call against any RFC-compliant server.

Also corrects __init__.py's __version__, which was left at 0.1.5 through the
0.1.6 release.

Signed-off-by: disqualifier <dev@disqualifier.me>
This commit is contained in:
2026-07-02 17:02:56 -04:00
parent a340067048
commit e5ea5cd417
6 changed files with 69 additions and 20 deletions
+8 -5
View File
@@ -11,22 +11,22 @@ This reads codes from email; it does not generate them (that is `pyotp`'s job).
`requirements.txt`: `requirements.txt`:
``` ```
aiomail @ git+ssh://git@git.rethinkstudios.io/rethink-public/aiomail.git@v0.1.6 aiomail @ git+ssh://git@git.rethinkstudios.io/rethink-public/aiomail.git@v0.1.7
# OAuth token providers (Microsoft / Google) need the extra: # OAuth token providers (Microsoft / Google) need the extra:
aiomail[oauth] @ git+ssh://git@git.rethinkstudios.io/rethink-public/aiomail.git@v0.1.6 aiomail[oauth] @ git+ssh://git@git.rethinkstudios.io/rethink-public/aiomail.git@v0.1.7
``` ```
Direct: Direct:
```bash ```bash
pip install "aiomail @ git+ssh://git@git.rethinkstudios.io/rethink-public/aiomail.git@v0.1.6" pip install "aiomail @ git+ssh://git@git.rethinkstudios.io/rethink-public/aiomail.git@v0.1.7"
pip install "aiomail[oauth] @ git+ssh://git@git.rethinkstudios.io/rethink-public/aiomail.git@v0.1.6" pip install "aiomail[oauth] @ git+ssh://git@git.rethinkstudios.io/rethink-public/aiomail.git@v0.1.7"
``` ```
Requires `aioimaplib` and `beautifulsoup4` (pulled transitively). The `oauth` Requires `aioimaplib` and `beautifulsoup4` (pulled transitively). The `oauth`
extra adds `aiohttp` for the refresh-token providers. extra adds `aiohttp` for the refresh-token providers.
Drop the `@v0.1.6` suffix from the line above to install the latest unpinned. Drop the `@v0.1.7` suffix from the line above to install the latest unpinned.
## Password auth ## Password auth
@@ -66,6 +66,9 @@ await retrieve_otp(client, sender=re.compile(r"no-?reply@.*\.io")) # regex
await retrieve_otp(client, sender=lambda f: f.endswith("@x.com")) # callable await retrieve_otp(client, sender=lambda f: f.endswith("@x.com")) # callable
``` ```
Subject headers are RFC2047-decoded before matching/extraction, so providers
that encode non-ASCII subjects (`=?utf-8?B?...?=`) still match on plain text.
Code extraction is tunable too — `patterns` (regexes, first group wins) and Code extraction is tunable too — `patterns` (regexes, first group wins) and
`lengths` (standalone digit-run fallback): `lengths` (standalone digit-run fallback):
+1 -1
View File
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
[project] [project]
name = "aiomail" name = "aiomail"
version = "0.1.6" version = "0.1.7"
description = "async IMAP one-time-code retrieval with password/OAuth2 auth and dynamic matching" description = "async IMAP one-time-code retrieval with password/OAuth2 auth and dynamic matching"
requires-python = ">=3.10" requires-python = ">=3.10"
dependencies = [ dependencies = [
+3 -1
View File
@@ -11,6 +11,7 @@ from .extract import (
DEFAULT_PATTERNS, DEFAULT_PATTERNS,
MatchSpec, MatchSpec,
as_predicate, as_predicate,
decode_header_value,
extract_code, extract_code,
) )
from .retrieve import DEFAULT_FOLDERS, retrieve_otp from .retrieve import DEFAULT_FOLDERS, retrieve_otp
@@ -22,6 +23,7 @@ __all__ = [
"IMAPClient", "IMAPClient",
"extract_code", "extract_code",
"as_predicate", "as_predicate",
"decode_header_value",
"retrieve_otp", "retrieve_otp",
"MatchSpec", "MatchSpec",
"DEFAULT_PATTERNS", "DEFAULT_PATTERNS",
@@ -29,4 +31,4 @@ __all__ = [
"DEFAULT_FOLDERS", "DEFAULT_FOLDERS",
] ]
__version__ = "0.1.5" __version__ = "0.1.7"
+13 -6
View File
@@ -33,16 +33,21 @@ log = logging.getLogger(__name__)
_LIST_RE = re.compile(rb'^\([^)]*\)\s+(?:"[^"]*"|NIL)\s+(.+)$') _LIST_RE = re.compile(rb'^\([^)]*\)\s+(?:"[^"]*"|NIL)\s+(.+)$')
def _folder_name(raw: bytes) -> str: def _folder_name(raw: bytes) -> Optional[str]:
"""extract the folder name from a LIST reply line, delimiter-agnostic """extract the folder name from a LIST reply line, delimiter-agnostic
parses the real reply form `(flags) "<delim>" <name>` so any server hierarchy parses the real reply form `(flags) "<delim>" <name>` so any server hierarchy
delimiter works (not just "/"); falls back to the last quoted/space token if the delimiter works (not just "/"); returns None if the line doesn't match the
line doesn't match the canonical shape. canonical shape. aioimaplib appends the tagged completion text (e.g. `b"LIST
completed."`) as the final entry of the same response list that carries the
untagged LIST lines — it never matches LIST syntax, so returning None here
(instead of a last-token rsplit fallback) lets the caller drop it instead of
treating it as a phantom folder name.
""" """
match = _LIST_RE.match(raw.strip()) match = _LIST_RE.match(raw.strip())
name = match.group(1).decode() if match else raw.decode().rsplit(" ", 1)[-1] if not match:
return name.strip().strip('"') return None
return match.group(1).decode().strip().strip('"')
class IMAPClient: class IMAPClient:
@@ -206,9 +211,11 @@ class IMAPClient:
folders: List[str] = [] folders: List[str] = []
for folder in folder_list or []: for folder in folder_list or []:
try: try:
folders.append(_folder_name(folder)) name = _folder_name(folder)
except Exception: except Exception:
continue continue
if name is not None:
folders.append(name)
return folders return folders
async def select(self, folder: str) -> bool: async def select(self, folder: str) -> bool:
+19 -1
View File
@@ -8,6 +8,7 @@ filter senders and subjects.
import email.message import email.message
import logging import logging
import re import re
from email.header import decode_header, make_header
from typing import Callable, Iterable, Iterator, Optional, Pattern, Sequence, Union from typing import Callable, Iterable, Iterator, Optional, Pattern, Sequence, Union
from bs4 import BeautifulSoup from bs4 import BeautifulSoup
@@ -39,6 +40,23 @@ def _compile(patterns: Sequence[Union[str, Pattern]]) -> list[Pattern]:
return out return out
def decode_header_value(raw: str) -> str:
"""decode an RFC2047 encoded-word header (=?charset?B/Q?...?=) to text
messages are parsed with email.message_from_bytes (compat32 policy), which
leaves encoded-word headers raw; decode before scanning so the subject-first
pass sees real text instead of base64/quoted-printable. falls back to the
raw value if decoding fails.
"""
if not raw:
return raw
try:
return str(make_header(decode_header(raw)))
except (UnicodeDecodeError, LookupError, ValueError) as exc:
log.debug("header decode failed (%s): %s", raw, exc)
return raw
def _decode_part(part: email.message.Message) -> Optional[str]: def _decode_part(part: email.message.Message) -> Optional[str]:
"""decode a single message part to text, tolerating bad charsets""" """decode a single message part to text, tolerating bad charsets"""
payload = part.get_payload(decode=True) payload = part.get_payload(decode=True)
@@ -97,7 +115,7 @@ def extract_code(
compiled = _compile(patterns) compiled = _compile(patterns)
length_set = set(lengths) length_set = set(lengths)
subject = message.get("Subject", "") or "" subject = decode_header_value(message.get("Subject", "") or "")
hit = _scan(subject, compiled, length_set) hit = _scan(subject, compiled, length_set)
if hit: if hit:
return hit return hit
+25 -6
View File
@@ -8,11 +8,19 @@ branches inside the function.
import asyncio import asyncio
import logging import logging
import time import time
from datetime import timezone
from email.utils import parsedate_to_datetime from email.utils import parsedate_to_datetime
from typing import Iterable, List, Optional, Pattern, Sequence, Union from typing import Iterable, List, Optional, Pattern, Sequence, Union
from .client import IMAPClient from .client import IMAPClient
from .extract import DEFAULT_LENGTHS, DEFAULT_PATTERNS, MatchSpec, as_predicate, extract_code from .extract import (
DEFAULT_LENGTHS,
DEFAULT_PATTERNS,
MatchSpec,
as_predicate,
decode_header_value,
extract_code,
)
log = logging.getLogger(__name__) log = logging.getLogger(__name__)
@@ -26,12 +34,17 @@ def _server_query(sender: MatchSpec, subject: MatchSpec, match_field: str = "fro
only plain strings translate to server-side filters; regex and callable specs only plain strings translate to server-side filters; regex and callable specs
fall back to ALL and are filtered client-side, so dynamic matching always works fall back to ALL and are filtered client-side, so dynamic matching always works
even when the server cannot express it. `match_field` selects which header the even when the server cannot express it. `match_field` selects which header the
`sender` spec searches: "from" filters by the sender address (default), "to" `sender` spec searches: "from" filters by the sender address (default); "to"
filters by the recipient address (the per-user alias the code was sent to). filters by the recipient address (the per-user alias the code was sent to) OR
the From header, matching the client-side check's forwarded-From fallback so
the server query never narrows out a result the client would have accepted.
""" """
parts: List[str] = [] parts: List[str] = []
if isinstance(sender, str): if isinstance(sender, str):
parts.append(f'TO "{sender}"' if match_field == "to" else f'FROM "{sender}"') if match_field == "to":
parts.append(f'OR TO "{sender}" FROM "{sender}"')
else:
parts.append(f'FROM "{sender}"')
if isinstance(subject, str): if isinstance(subject, str):
parts.append(f'SUBJECT "{subject}"') parts.append(f'SUBJECT "{subject}"')
return f"({' '.join(parts)})" if parts else "ALL" return f"({' '.join(parts)})" if parts else "ALL"
@@ -43,7 +56,13 @@ def _age_seconds(message) -> Optional[float]:
if not raw: if not raw:
return None return None
try: try:
return time.time() - parsedate_to_datetime(raw).timestamp() dt = parsedate_to_datetime(raw)
if dt.tzinfo is None:
# parsedate_to_datetime returns a naive datetime for a "-0000" zone
# (RFC 2822: unknown/unspecified offset); treat it as UTC rather
# than letting .timestamp() interpret the wall time as local
dt = dt.replace(tzinfo=timezone.utc)
return time.time() - dt.timestamp()
except (TypeError, ValueError) as exc: except (TypeError, ValueError) as exc:
log.debug("date parse failed (%s): %s", raw, exc) log.debug("date parse failed (%s): %s", raw, exc)
return None return None
@@ -97,7 +116,7 @@ async def retrieve_otp(
continue continue
from_hdr = message.get("From", "") from_hdr = message.get("From", "")
subj_hdr = message.get("Subject", "") subj_hdr = decode_header_value(message.get("Subject", ""))
if match_field == "to": if match_field == "to":
to_hdr = message.get("To", "") to_hdr = message.get("To", "")
matched = sender_ok(to_hdr) or sender_ok(from_hdr) matched = sender_ok(to_hdr) or sender_ok(from_hdr)