diff --git a/pytest.ini b/pytest.ini
index 80432c2..fe9256e 100644
--- a/pytest.ini
+++ b/pytest.ini
@@ -1,3 +1,8 @@
[pytest]
testpaths = tests
pythonpath = src
+; Integration tests (tests/test_integration.py) require a live Qdrant on
+; localhost:6333 — they only run in the nightly CI job (or manually with -m integration).
+markers =
+ integration: requires a live Qdrant instance (deselected by default)
+addopts = -m "not integration" -p no:cacheprovider
diff --git a/src/indexer.py b/src/indexer.py
index 8682934..944fcc1 100644
--- a/src/indexer.py
+++ b/src/indexer.py
@@ -22,26 +22,27 @@ from bs4 import BeautifulSoup
# Load .env config
load_dotenv()
-# Configuration
+# Configuration (validated lazily in main() so the module can be imported
+# and unit-tested without environment variables)
MAILDIR_PATH = os.environ.get("MAILDIR_PATH", "")
MAILDIR_FOLDERS = os.environ.get("MAILDIR_FOLDERS", "")
QDRANT_URL = os.environ.get("QDRANT_URL", "")
COLLECTION_NAME = os.environ.get("COLLECTION_NAME", "")
-if not MAILDIR_PATH:
- raise ValueError("MAILDIR_PATH environment variable is required.")
-if not QDRANT_URL:
- raise ValueError("QDRANT_URL environment variable is required.")
-if not COLLECTION_NAME:
- raise ValueError("COLLECTION_NAME environment variable is required.")
-
-EMBEDDING_MODEL_NAME = os.environ.get("EMBEDDING_MODEL_NAME", "BAAI/bge-small-en-v1.5")
-BATCH_SIZE = int(os.environ.get("BATCH_SIZE", "100"))
-EMBEDDING_BATCH_SIZE = int(os.environ.get("EMBEDDING_BATCH_SIZE", "64"))
METADATA_COLLECTION = "mcp_indexer_metadata"
INCREMENTAL_DAYS = int(os.environ.get("INCREMENTAL_DAYS", "7"))
FORCE_REINDEX = os.environ.get("FORCE_REINDEX", "").lower() in ("1", "true", "yes")
+
+def _validate_config():
+ """Raises if required environment variables are missing. Called from main()."""
+ if not MAILDIR_PATH:
+ raise ValueError("MAILDIR_PATH environment variable is required.")
+ if not QDRANT_URL:
+ raise ValueError("QDRANT_URL environment variable is required.")
+ if not COLLECTION_NAME:
+ raise ValueError("COLLECTION_NAME environment variable is required.")
+
def decode_mime_words(s: str) -> str:
"""Decodes MIME encoded strings (e.g. subjects, filenames)."""
if not s:
@@ -251,6 +252,8 @@ def main():
Main ingestion function.
Reads Maildir, extracts text, generates local embeddings, and pushes to Qdrant.
"""
+ _validate_config()
+
print(f"Indexing emails from {MAILDIR_PATH} into {QDRANT_URL}...")
if not os.path.exists(MAILDIR_PATH):
diff --git a/tests/test_indexer.py b/tests/test_indexer.py
new file mode 100644
index 0000000..4b33ce7
--- /dev/null
+++ b/tests/test_indexer.py
@@ -0,0 +1,229 @@
+"""Unit tests for mcp-maildir indexer parsing functions.
+
+Pure-function tests: no Qdrant, no embedding model, no network.
+A throwaway Maildir is built in tmp_path for get_recent_keys().
+"""
+
+import mailbox
+import os
+import time
+import uuid
+from datetime import datetime, timezone, timedelta
+from email.message import EmailMessage
+from email.utils import format_datetime
+
+import pytest
+
+from indexer import (
+ decode_mime_words,
+ extract_text_from_html,
+ normalize_email_address,
+ parse_email_message,
+ get_recent_keys,
+ METADATA_COLLECTION,
+)
+
+
+# ---------------------------------------------------------------------------
+# decode_mime_words
+# ---------------------------------------------------------------------------
+
+class TestDecodeMimeWords:
+ def test_empty_and_none(self):
+ assert decode_mime_words("") == ""
+ assert decode_mime_words(None) == ""
+
+ def test_plain_ascii(self):
+ assert decode_mime_words("Hello World") == "Hello World"
+
+ def test_encoded_word_utf8(self):
+ # =?utf-8?q?...?= encoded word
+ assert decode_mime_words("=?utf-8?q?Caf=C3=A9_au_lait?=") == "Café au lait"
+
+ def test_encoded_word_iso8859(self):
+ # decode_mime_words does not strip whitespace around the words
+ assert decode_mime_words("=?iso-8859-1?q?caf=E9?=") == "café"
+
+ def test_mixed_encoded_and_plain(self):
+ assert decode_mime_words("Re: =?utf-8?q?planning_2026?=") == "Re: planning 2026"
+
+
+# ---------------------------------------------------------------------------
+# extract_text_from_html
+# ---------------------------------------------------------------------------
+
+class TestExtractTextFromHtml:
+ def test_simple_paragraphs(self):
+ html = "
Hello
World
"
+ assert extract_text_from_html(html) == "Hello World"
+
+ def test_strips_tags_and_scripts(self):
+ html = "SignatureEnd
"
+ text = extract_text_from_html(html)
+ assert "evil()" not in text
+ # get_text uses separator=" ", so tags become whitespace
+ assert "Sign" in text and "ature" in text and "End" in text
+
+ def test_fallback_on_invalid_content(self):
+ # Garbage that would make the parser blow up should not raise
+ assert extract_text_from_html(None) is None
+
+
+# ---------------------------------------------------------------------------
+# parse_email_message — real mailbox.Message objects (no mocks)
+# ---------------------------------------------------------------------------
+
+def build_email(
+ subject="Hello",
+ body="This is the body.",
+ html=None,
+ attachments=None,
+ charset="utf-8",
+):
+ """Builds a real EmailMessage, optionally multipart with attachments."""
+ msg = EmailMessage()
+ msg["Subject"] = subject
+ msg["From"] = "Alice "
+ msg["To"] = "Bob "
+ msg["Message-ID"] = ""
+ msg["Date"] = format_datetime(datetime(2026, 1, 15, 10, 0, 0, tzinfo=timezone.utc))
+
+ if html:
+ msg.set_content(body)
+ msg.add_alternative(html, subtype="html")
+ else:
+ msg.set_content(body, charset=charset)
+
+ for filename, content in (attachments or {}).items():
+ payload = content if isinstance(content, bytes) else content.encode()
+ msg.add_attachment(payload, maintype="application",
+ subtype="octet-stream", filename=filename)
+ return msg
+
+
+def build_raw_email(raw_bytes: bytes):
+ """Parses raw email bytes (for exotic headers without EmailMessage helpers)."""
+ import email as email_mod
+ return email_mod.message_from_bytes(raw_bytes)
+
+
+class TestParseEmailMessage:
+ def test_plain_text_body(self):
+ msg = build_email(body="Line one.\nLine two.")
+ body, attachments = parse_email_message(msg)
+ assert "Line one." in body and "Line two." in body
+ assert attachments == []
+
+ def test_attachments_detected_and_excluded_from_body(self):
+ msg = build_email(
+ body="See attached.",
+ attachments={"report.pdf": b"%PDF-fake", "notes.txt": b"notes"},
+ )
+ body, attachments = parse_email_message(msg)
+ assert attachments == ["report.pdf", "notes.txt"]
+ assert "%PDF" not in body
+ assert "See attached." in body
+
+ def test_html_alternative_extracts_text(self):
+ msg = build_email(
+ body="Fallback plain text.",
+ html="Bold intro",
+ )
+ body, _ = parse_email_message(msg)
+ assert "Bold" in body and "intro" in body
+
+ def test_unknown_charset_falls_back_not_crash(self):
+ # A payload claiming an unknown charset must not raise.
+ # Built from raw bytes: EmailMessage.set_content(charset=...) would
+ # reject the unknown encoding at build time, real mail does not.
+ raw = (
+ b"Subject: broken charset\n"
+ b"From: alice@example.com\n"
+ b"To: bob@example.com\n"
+ b"Content-Type: text/plain; charset=x-unknown-charset\n"
+ b"Content-Transfer-Encoding: 8bit\n\n"
+ b"body with broken charset\n"
+ )
+ msg = build_raw_email(raw)
+ body, attachments = parse_email_message(msg)
+ # Whatever the fallback, we must get a string body and no exception
+ assert isinstance(body, str)
+ assert attachments == []
+
+ def test_encoded_attachment_filename_decoded(self):
+ msg = build_email(
+ body="body",
+ attachments={},
+ )
+ # Add attachment with non-ascii filename via encoded word
+ msg.add_attachment(b"data", maintype="application", subtype="pdf",
+ filename="café-rapport.pdf")
+ body, attachments = parse_email_message(msg)
+ assert attachments == ["café-rapport.pdf"]
+
+
+# ---------------------------------------------------------------------------
+# normalize_email_address (indexer copy — same contract as server.py)
+# ---------------------------------------------------------------------------
+
+class TestIndexerNormalizeEmailAddress:
+ def test_display_name(self):
+ assert normalize_email_address("John Doe ") == "john@example.com"
+
+ def test_case_and_whitespace(self):
+ assert normalize_email_address(" USER@Example.COM ") == "user@example.com"
+
+ def test_empty(self):
+ assert normalize_email_address("") == ""
+ assert normalize_email_address(None) == ""
+
+
+# ---------------------------------------------------------------------------
+# get_recent_keys — real Maildir on disk (tmp_path), mtime-driven
+# ---------------------------------------------------------------------------
+
+def make_maildir(root):
+ """Opens a Maildir at root — pre-creates cur/new/tmp because mailbox.Maildir
+ (Python 3.11) does not create them when adding a message."""
+ for sub in ("cur", "new", "tmp"):
+ os.makedirs(os.path.join(str(root), sub), exist_ok=True)
+ return mailbox.Maildir(str(root), create=False)
+
+
+def write_maildir_message(maildir: mailbox.Maildir, msg: EmailMessage):
+ """Adds a message and returns its key."""
+ return maildir.add(msg)
+
+
+class TestGetRecentKeys:
+ def test_recent_file_selected(self, tmp_path):
+ md = make_maildir(tmp_path)
+ msg = build_email()
+ md.add(msg)
+ md.flush()
+ # mtime is now → within the window
+ keys = get_recent_keys(md, days=7)
+ assert len(keys) == 1
+
+ def test_old_file_excluded(self, tmp_path):
+ md = make_maildir(tmp_path)
+ md.add(build_email())
+ md.flush()
+ # Age every file beyond the cutoff
+ old = time.time() - 30 * 86400
+ for path in tmp_path.rglob("*.??*"):
+ if path.is_file():
+ os.utime(path, (old, old))
+ keys = get_recent_keys(md, days=7)
+ assert len(keys) == 0
+
+ def test_key_is_suffixless_filename(self, tmp_path):
+ """Maildir keys strip the ':2,S' info suffix — get_recent_keys must match."""
+ md = make_maildir(tmp_path)
+ key = md.add(build_email())
+ md.flush()
+ keys = get_recent_keys(md, days=7)
+ assert keys # non-empty set of keys for fresh files
+ assert key.split(":")[0] in keys
+ for k in keys:
+ assert ":" not in k