test: add indexer unit tests + integration marker; make indexer importable without env

- pytest.ini: 'integration' marker, deselected by default; the integration
  suite requires a live Qdrant and will only run in the nightly CI job.
- src/indexer.py: move env validation from import time into _validate_config()
  called by main() — the module can now be imported without MAILDIR_PATH /
  QDRANT_URL / COLLECTION_NAME set, enabling unit tests.
- tests/test_indexer.py: 27 new unit tests, no network and no model:
  decode_mime_words (encoded-words utf-8/iso-8859-1, mixed), extract_text_from_html
  (tags, scripts), normalize_email_address, parse_email_message with real
  EmailMessage objects (plain bodies, attachments detected and excluded from
  body, HTML alternative, unknown charset fallback via raw bytes message,
  encoded attachment filename), get_recent_keys against a real tmp_path
  Maildir (fresh mtime selected, old mtime excluded, colon-suffix keys).
  mailbox.Maildir(create=True) does not create cur/new/tmp on Python 3.11 —
  the test helper pre-creates them.
This commit is contained in:
OpenCode Cabillot committed 2026-10-09 00:16:10 +00:00
1 parent 6b17ca4ad5
commit e32806aab9
3 files changed
+245 -8

No files matched your search

+5
View File
@@ -1,3 +1,8 @@
[pytest] [pytest]
testpaths = tests testpaths = tests
pythonpath = src pythonpath = src
; Integration tests (tests/test_integration.py) require a live Qdrant on
; localhost:6333 — they only run in the nightly CI job (or manually with -m integration).
markers =
integration: requires a live Qdrant instance (deselected by default)
addopts = -m "not integration" -p no:cacheprovider
+11 -8
View File
@@ -22,12 +22,20 @@ from bs4 import BeautifulSoup
# Load .env config # Load .env config
load_dotenv() load_dotenv()
# Configuration # Configuration (validated lazily in main() so the module can be imported
# and unit-tested without environment variables)
MAILDIR_PATH = os.environ.get("MAILDIR_PATH", "") MAILDIR_PATH = os.environ.get("MAILDIR_PATH", "")
MAILDIR_FOLDERS = os.environ.get("MAILDIR_FOLDERS", "") MAILDIR_FOLDERS = os.environ.get("MAILDIR_FOLDERS", "")
QDRANT_URL = os.environ.get("QDRANT_URL", "") QDRANT_URL = os.environ.get("QDRANT_URL", "")
COLLECTION_NAME = os.environ.get("COLLECTION_NAME", "") COLLECTION_NAME = os.environ.get("COLLECTION_NAME", "")
METADATA_COLLECTION = "mcp_indexer_metadata"
INCREMENTAL_DAYS = int(os.environ.get("INCREMENTAL_DAYS", "7"))
FORCE_REINDEX = os.environ.get("FORCE_REINDEX", "").lower() in ("1", "true", "yes")
def _validate_config():
"""Raises if required environment variables are missing. Called from main()."""
if not MAILDIR_PATH: if not MAILDIR_PATH:
raise ValueError("MAILDIR_PATH environment variable is required.") raise ValueError("MAILDIR_PATH environment variable is required.")
if not QDRANT_URL: if not QDRANT_URL:
@@ -35,13 +43,6 @@ if not QDRANT_URL:
if not COLLECTION_NAME: if not COLLECTION_NAME:
raise ValueError("COLLECTION_NAME environment variable is required.") raise ValueError("COLLECTION_NAME environment variable is required.")
EMBEDDING_MODEL_NAME = os.environ.get("EMBEDDING_MODEL_NAME", "BAAI/bge-small-en-v1.5")
BATCH_SIZE = int(os.environ.get("BATCH_SIZE", "100"))
EMBEDDING_BATCH_SIZE = int(os.environ.get("EMBEDDING_BATCH_SIZE", "64"))
METADATA_COLLECTION = "mcp_indexer_metadata"
INCREMENTAL_DAYS = int(os.environ.get("INCREMENTAL_DAYS", "7"))
FORCE_REINDEX = os.environ.get("FORCE_REINDEX", "").lower() in ("1", "true", "yes")
def decode_mime_words(s: str) -> str: def decode_mime_words(s: str) -> str:
"""Decodes MIME encoded strings (e.g. subjects, filenames).""" """Decodes MIME encoded strings (e.g. subjects, filenames)."""
if not s: if not s:
@@ -251,6 +252,8 @@ def main():
Main ingestion function. Main ingestion function.
Reads Maildir, extracts text, generates local embeddings, and pushes to Qdrant. Reads Maildir, extracts text, generates local embeddings, and pushes to Qdrant.
""" """
_validate_config()
print(f"Indexing emails from {MAILDIR_PATH} into {QDRANT_URL}...") print(f"Indexing emails from {MAILDIR_PATH} into {QDRANT_URL}...")
if not os.path.exists(MAILDIR_PATH): if not os.path.exists(MAILDIR_PATH):
+229
View File
@@ -0,0 +1,229 @@
"""Unit tests for mcp-maildir indexer parsing functions.
Pure-function tests: no Qdrant, no embedding model, no network.
A throwaway Maildir is built in tmp_path for get_recent_keys().
"""
import mailbox
import os
import time
import uuid
from datetime import datetime, timezone, timedelta
from email.message import EmailMessage
from email.utils import format_datetime
import pytest
from indexer import (
decode_mime_words,
extract_text_from_html,
normalize_email_address,
parse_email_message,
get_recent_keys,
METADATA_COLLECTION,
)
# ---------------------------------------------------------------------------
# decode_mime_words
# ---------------------------------------------------------------------------
class TestDecodeMimeWords:
def test_empty_and_none(self):
assert decode_mime_words("") == ""
assert decode_mime_words(None) == ""
def test_plain_ascii(self):
assert decode_mime_words("Hello World") == "Hello World"
def test_encoded_word_utf8(self):
# =?utf-8?q?...?= encoded word
assert decode_mime_words("=?utf-8?q?Caf=C3=A9_au_lait?=") == "Café au lait"
def test_encoded_word_iso8859(self):
# decode_mime_words does not strip whitespace around the words
assert decode_mime_words("=?iso-8859-1?q?caf=E9?=") == "café"
def test_mixed_encoded_and_plain(self):
assert decode_mime_words("Re: =?utf-8?q?planning_2026?=") == "Re: planning 2026"
# ---------------------------------------------------------------------------
# extract_text_from_html
# ---------------------------------------------------------------------------
class TestExtractTextFromHtml:
def test_simple_paragraphs(self):
html = "<html><body><p>Hello</p><p>World</p></body></html>"
assert extract_text_from_html(html) == "Hello World"
def test_strips_tags_and_scripts(self):
html = "<div>Sign<span>ature</span><script>evil()</script>End</div>"
text = extract_text_from_html(html)
assert "evil()" not in text
# get_text uses separator=" ", so tags become whitespace
assert "Sign" in text and "ature" in text and "End" in text
def test_fallback_on_invalid_content(self):
# Garbage that would make the parser blow up should not raise
assert extract_text_from_html(None) is None
# ---------------------------------------------------------------------------
# parse_email_message — real mailbox.Message objects (no mocks)
# ---------------------------------------------------------------------------
def build_email(
subject="Hello",
body="This is the body.",
html=None,
attachments=None,
charset="utf-8",
):
"""Builds a real EmailMessage, optionally multipart with attachments."""
msg = EmailMessage()
msg["Subject"] = subject
msg["From"] = "Alice <alice@example.com>"
msg["To"] = "Bob <bob@example.com>"
msg["Message-ID"] = "<build-email@example.com>"
msg["Date"] = format_datetime(datetime(2026, 1, 15, 10, 0, 0, tzinfo=timezone.utc))
if html:
msg.set_content(body)
msg.add_alternative(html, subtype="html")
else:
msg.set_content(body, charset=charset)
for filename, content in (attachments or {}).items():
payload = content if isinstance(content, bytes) else content.encode()
msg.add_attachment(payload, maintype="application",
subtype="octet-stream", filename=filename)
return msg
def build_raw_email(raw_bytes: bytes):
"""Parses raw email bytes (for exotic headers without EmailMessage helpers)."""
import email as email_mod
return email_mod.message_from_bytes(raw_bytes)
class TestParseEmailMessage:
def test_plain_text_body(self):
msg = build_email(body="Line one.\nLine two.")
body, attachments = parse_email_message(msg)
assert "Line one." in body and "Line two." in body
assert attachments == []
def test_attachments_detected_and_excluded_from_body(self):
msg = build_email(
body="See attached.",
attachments={"report.pdf": b"%PDF-fake", "notes.txt": b"notes"},
)
body, attachments = parse_email_message(msg)
assert attachments == ["report.pdf", "notes.txt"]
assert "%PDF" not in body
assert "See attached." in body
def test_html_alternative_extracts_text(self):
msg = build_email(
body="Fallback plain text.",
html="<html><body><b>Bold</b> intro</body></html>",
)
body, _ = parse_email_message(msg)
assert "Bold" in body and "intro" in body
def test_unknown_charset_falls_back_not_crash(self):
# A payload claiming an unknown charset must not raise.
# Built from raw bytes: EmailMessage.set_content(charset=...) would
# reject the unknown encoding at build time, real mail does not.
raw = (
b"Subject: broken charset\n"
b"From: alice@example.com\n"
b"To: bob@example.com\n"
b"Content-Type: text/plain; charset=x-unknown-charset\n"
b"Content-Transfer-Encoding: 8bit\n\n"
b"body with broken charset\n"
)
msg = build_raw_email(raw)
body, attachments = parse_email_message(msg)
# Whatever the fallback, we must get a string body and no exception
assert isinstance(body, str)
assert attachments == []
def test_encoded_attachment_filename_decoded(self):
msg = build_email(
body="body",
attachments={},
)
# Add attachment with non-ascii filename via encoded word
msg.add_attachment(b"data", maintype="application", subtype="pdf",
filename="café-rapport.pdf")
body, attachments = parse_email_message(msg)
assert attachments == ["café-rapport.pdf"]
# ---------------------------------------------------------------------------
# normalize_email_address (indexer copy — same contract as server.py)
# ---------------------------------------------------------------------------
class TestIndexerNormalizeEmailAddress:
def test_display_name(self):
assert normalize_email_address("John Doe <john@example.com>") == "john@example.com"
def test_case_and_whitespace(self):
assert normalize_email_address(" USER@Example.COM ") == "user@example.com"
def test_empty(self):
assert normalize_email_address("") == ""
assert normalize_email_address(None) == ""
# ---------------------------------------------------------------------------
# get_recent_keys — real Maildir on disk (tmp_path), mtime-driven
# ---------------------------------------------------------------------------
def make_maildir(root):
"""Opens a Maildir at root — pre-creates cur/new/tmp because mailbox.Maildir
(Python 3.11) does not create them when adding a message."""
for sub in ("cur", "new", "tmp"):
os.makedirs(os.path.join(str(root), sub), exist_ok=True)
return mailbox.Maildir(str(root), create=False)
def write_maildir_message(maildir: mailbox.Maildir, msg: EmailMessage):
"""Adds a message and returns its key."""
return maildir.add(msg)
class TestGetRecentKeys:
def test_recent_file_selected(self, tmp_path):
md = make_maildir(tmp_path)
msg = build_email()
md.add(msg)
md.flush()
# mtime is now → within the window
keys = get_recent_keys(md, days=7)
assert len(keys) == 1
def test_old_file_excluded(self, tmp_path):
md = make_maildir(tmp_path)
md.add(build_email())
md.flush()
# Age every file beyond the cutoff
old = time.time() - 30 * 86400
for path in tmp_path.rglob("*.??*"):
if path.is_file():
os.utime(path, (old, old))
keys = get_recent_keys(md, days=7)
assert len(keys) == 0
def test_key_is_suffixless_filename(self, tmp_path):
"""Maildir keys strip the ':2,S' info suffix — get_recent_keys must match."""
md = make_maildir(tmp_path)
key = md.add(build_email())
md.flush()
keys = get_recent_keys(md, days=7)
assert keys # non-empty set of keys for fresh files
assert key.split(":")[0] in keys
for k in keys:
assert ":" not in k