- pytest.ini: 'integration' marker, deselected by default; the integration suite requires a live Qdrant and will only run in the nightly CI job. - src/indexer.py: move env validation from import time into _validate_config() called by main() — the module can now be imported without MAILDIR_PATH / QDRANT_URL / COLLECTION_NAME set, enabling unit tests. - tests/test_indexer.py: 27 new unit tests, no network and no model: decode_mime_words (encoded-words utf-8/iso-8859-1, mixed), extract_text_from_html (tags, scripts), normalize_email_address, parse_email_message with real EmailMessage objects (plain bodies, attachments detected and excluded from body, HTML alternative, unknown charset fallback via raw bytes message, encoded attachment filename), get_recent_keys against a real tmp_path Maildir (fresh mtime selected, old mtime excluded, colon-suffix keys). mailbox.Maildir(create=True) does not create cur/new/tmp on Python 3.11 — the test helper pre-creates them.
230 lines
8.2 KiB
Python
230 lines
8.2 KiB
Python
"""Unit tests for mcp-maildir indexer parsing functions.
|
|
|
|
Pure-function tests: no Qdrant, no embedding model, no network.
|
|
A throwaway Maildir is built in tmp_path for get_recent_keys().
|
|
"""
|
|
|
|
import mailbox
|
|
import os
|
|
import time
|
|
import uuid
|
|
from datetime import datetime, timezone, timedelta
|
|
from email.message import EmailMessage
|
|
from email.utils import format_datetime
|
|
|
|
import pytest
|
|
|
|
from indexer import (
|
|
decode_mime_words,
|
|
extract_text_from_html,
|
|
normalize_email_address,
|
|
parse_email_message,
|
|
get_recent_keys,
|
|
METADATA_COLLECTION,
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# decode_mime_words
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestDecodeMimeWords:
|
|
def test_empty_and_none(self):
|
|
assert decode_mime_words("") == ""
|
|
assert decode_mime_words(None) == ""
|
|
|
|
def test_plain_ascii(self):
|
|
assert decode_mime_words("Hello World") == "Hello World"
|
|
|
|
def test_encoded_word_utf8(self):
|
|
# =?utf-8?q?...?= encoded word
|
|
assert decode_mime_words("=?utf-8?q?Caf=C3=A9_au_lait?=") == "Café au lait"
|
|
|
|
def test_encoded_word_iso8859(self):
|
|
# decode_mime_words does not strip whitespace around the words
|
|
assert decode_mime_words("=?iso-8859-1?q?caf=E9?=") == "café"
|
|
|
|
def test_mixed_encoded_and_plain(self):
|
|
assert decode_mime_words("Re: =?utf-8?q?planning_2026?=") == "Re: planning 2026"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# extract_text_from_html
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestExtractTextFromHtml:
|
|
def test_simple_paragraphs(self):
|
|
html = "<html><body><p>Hello</p><p>World</p></body></html>"
|
|
assert extract_text_from_html(html) == "Hello World"
|
|
|
|
def test_strips_tags_and_scripts(self):
|
|
html = "<div>Sign<span>ature</span><script>evil()</script>End</div>"
|
|
text = extract_text_from_html(html)
|
|
assert "evil()" not in text
|
|
# get_text uses separator=" ", so tags become whitespace
|
|
assert "Sign" in text and "ature" in text and "End" in text
|
|
|
|
def test_fallback_on_invalid_content(self):
|
|
# Garbage that would make the parser blow up should not raise
|
|
assert extract_text_from_html(None) is None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# parse_email_message — real mailbox.Message objects (no mocks)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def build_email(
|
|
subject="Hello",
|
|
body="This is the body.",
|
|
html=None,
|
|
attachments=None,
|
|
charset="utf-8",
|
|
):
|
|
"""Builds a real EmailMessage, optionally multipart with attachments."""
|
|
msg = EmailMessage()
|
|
msg["Subject"] = subject
|
|
msg["From"] = "Alice <alice@example.com>"
|
|
msg["To"] = "Bob <bob@example.com>"
|
|
msg["Message-ID"] = "<build-email@example.com>"
|
|
msg["Date"] = format_datetime(datetime(2026, 1, 15, 10, 0, 0, tzinfo=timezone.utc))
|
|
|
|
if html:
|
|
msg.set_content(body)
|
|
msg.add_alternative(html, subtype="html")
|
|
else:
|
|
msg.set_content(body, charset=charset)
|
|
|
|
for filename, content in (attachments or {}).items():
|
|
payload = content if isinstance(content, bytes) else content.encode()
|
|
msg.add_attachment(payload, maintype="application",
|
|
subtype="octet-stream", filename=filename)
|
|
return msg
|
|
|
|
|
|
def build_raw_email(raw_bytes: bytes):
|
|
"""Parses raw email bytes (for exotic headers without EmailMessage helpers)."""
|
|
import email as email_mod
|
|
return email_mod.message_from_bytes(raw_bytes)
|
|
|
|
|
|
class TestParseEmailMessage:
|
|
def test_plain_text_body(self):
|
|
msg = build_email(body="Line one.\nLine two.")
|
|
body, attachments = parse_email_message(msg)
|
|
assert "Line one." in body and "Line two." in body
|
|
assert attachments == []
|
|
|
|
def test_attachments_detected_and_excluded_from_body(self):
|
|
msg = build_email(
|
|
body="See attached.",
|
|
attachments={"report.pdf": b"%PDF-fake", "notes.txt": b"notes"},
|
|
)
|
|
body, attachments = parse_email_message(msg)
|
|
assert attachments == ["report.pdf", "notes.txt"]
|
|
assert "%PDF" not in body
|
|
assert "See attached." in body
|
|
|
|
def test_html_alternative_extracts_text(self):
|
|
msg = build_email(
|
|
body="Fallback plain text.",
|
|
html="<html><body><b>Bold</b> intro</body></html>",
|
|
)
|
|
body, _ = parse_email_message(msg)
|
|
assert "Bold" in body and "intro" in body
|
|
|
|
def test_unknown_charset_falls_back_not_crash(self):
|
|
# A payload claiming an unknown charset must not raise.
|
|
# Built from raw bytes: EmailMessage.set_content(charset=...) would
|
|
# reject the unknown encoding at build time, real mail does not.
|
|
raw = (
|
|
b"Subject: broken charset\n"
|
|
b"From: alice@example.com\n"
|
|
b"To: bob@example.com\n"
|
|
b"Content-Type: text/plain; charset=x-unknown-charset\n"
|
|
b"Content-Transfer-Encoding: 8bit\n\n"
|
|
b"body with broken charset\n"
|
|
)
|
|
msg = build_raw_email(raw)
|
|
body, attachments = parse_email_message(msg)
|
|
# Whatever the fallback, we must get a string body and no exception
|
|
assert isinstance(body, str)
|
|
assert attachments == []
|
|
|
|
def test_encoded_attachment_filename_decoded(self):
|
|
msg = build_email(
|
|
body="body",
|
|
attachments={},
|
|
)
|
|
# Add attachment with non-ascii filename via encoded word
|
|
msg.add_attachment(b"data", maintype="application", subtype="pdf",
|
|
filename="café-rapport.pdf")
|
|
body, attachments = parse_email_message(msg)
|
|
assert attachments == ["café-rapport.pdf"]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# normalize_email_address (indexer copy — same contract as server.py)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestIndexerNormalizeEmailAddress:
|
|
def test_display_name(self):
|
|
assert normalize_email_address("John Doe <john@example.com>") == "john@example.com"
|
|
|
|
def test_case_and_whitespace(self):
|
|
assert normalize_email_address(" USER@Example.COM ") == "user@example.com"
|
|
|
|
def test_empty(self):
|
|
assert normalize_email_address("") == ""
|
|
assert normalize_email_address(None) == ""
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# get_recent_keys — real Maildir on disk (tmp_path), mtime-driven
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def make_maildir(root):
|
|
"""Opens a Maildir at root — pre-creates cur/new/tmp because mailbox.Maildir
|
|
(Python 3.11) does not create them when adding a message."""
|
|
for sub in ("cur", "new", "tmp"):
|
|
os.makedirs(os.path.join(str(root), sub), exist_ok=True)
|
|
return mailbox.Maildir(str(root), create=False)
|
|
|
|
|
|
def write_maildir_message(maildir: mailbox.Maildir, msg: EmailMessage):
|
|
"""Adds a message and returns its key."""
|
|
return maildir.add(msg)
|
|
|
|
|
|
class TestGetRecentKeys:
|
|
def test_recent_file_selected(self, tmp_path):
|
|
md = make_maildir(tmp_path)
|
|
msg = build_email()
|
|
md.add(msg)
|
|
md.flush()
|
|
# mtime is now → within the window
|
|
keys = get_recent_keys(md, days=7)
|
|
assert len(keys) == 1
|
|
|
|
def test_old_file_excluded(self, tmp_path):
|
|
md = make_maildir(tmp_path)
|
|
md.add(build_email())
|
|
md.flush()
|
|
# Age every file beyond the cutoff
|
|
old = time.time() - 30 * 86400
|
|
for path in tmp_path.rglob("*.??*"):
|
|
if path.is_file():
|
|
os.utime(path, (old, old))
|
|
keys = get_recent_keys(md, days=7)
|
|
assert len(keys) == 0
|
|
|
|
def test_key_is_suffixless_filename(self, tmp_path):
|
|
"""Maildir keys strip the ':2,S' info suffix — get_recent_keys must match."""
|
|
md = make_maildir(tmp_path)
|
|
key = md.add(build_email())
|
|
md.flush()
|
|
keys = get_recent_keys(md, days=7)
|
|
assert keys # non-empty set of keys for fresh files
|
|
assert key.split(":")[0] in keys
|
|
for k in keys:
|
|
assert ":" not in k
|