aboutsummaryrefslogtreecommitdiffstats
path: root/tests
diff options
context:
space:
mode:
Diffstat (limited to 'tests')
-rw-r--r--tests/fixtures/reportable.eml22
-rw-r--r--tests/test_case.py10
-rw-r--r--tests/test_cli.py25
-rw-r--r--tests/test_offline.py8
-rw-r--r--tests/test_parse.py136
-rw-r--r--tests/test_report.py1600
6 files changed, 1800 insertions, 1 deletions
diff --git a/tests/fixtures/reportable.eml b/tests/fixtures/reportable.eml
new file mode 100644
index 0000000..e939059
--- /dev/null
+++ b/tests/fixtures/reportable.eml
@@ -0,0 +1,22 @@
+Received: from relay.example.org (relay.example.org [192.0.2.10])
+ by mx.example.org with ESMTP id abc123
+ for <you@example.org>; Mon, 07 Sep 2026 09:12:44 +0000
+Received: from sender.invalid (sender.invalid [203.0.113.42])
+ by relay.example.org with ESMTP id def456
+ for <you@example.org>; Mon, 07 Sep 2026 09:12:40 +0000
+Return-Path: <bounce@sender.invalid>
+Authentication-Results: mx.example.org; spf=fail; dkim=none; dmarc=fail
+Received-SPF: fail (mx.example.org: domain of sender.invalid does not designate 203.0.113.42)
+From: "Example Bank" <phish@sender.invalid>
+To: victim@example.org
+Cc: colleague@example.org
+Delivered-To: victim@example.org
+X-Original-To: victim@example.org
+Reply-To: "Support" <reply@sender.invalid>
+Subject: Your account requires verification
+Date: Mon, 07 Sep 2026 09:12:40 +0000
+Message-ID: <case-one@sender.invalid>
+MIME-Version: 1.0
+Content-Type: text/plain; charset=utf-8
+
+Please verify at http://login.sender.invalid/verify?id=abc123
diff --git a/tests/test_case.py b/tests/test_case.py
index c6964a9..ab7eeeb 100644
--- a/tests/test_case.py
+++ b/tests/test_case.py
@@ -45,6 +45,16 @@ class TestCaseCreation(unittest.TestCase):
manifest = json.loads((created.path / "manifest.json").read_text())
self.assertEqual(manifest["format"], case.FORMAT_VERSION)
+ def test_every_block_is_seeded_present_and_empty(self):
+ # A created-but-unparsed case must have the same SHAPE as a parsed
+ # one, so a later reader indexes a block rather than guarding every
+ # access. report will read headers and would hit a KeyError.
+ created = case.create(self.root, b"x")
+ manifest = json.loads((created.path / "manifest.json").read_text())
+ for block in ("iocs", "auth", "contacts", "destinations", "headers"):
+ self.assertIn(block, manifest)
+ self.assertEqual(manifest[block], [])
+
def test_two_cases_do_not_collide(self):
a = case.create(self.root, b"one")
b = case.create(self.root, b"two")
diff --git a/tests/test_cli.py b/tests/test_cli.py
index 1c023c3..b8fffea 100644
--- a/tests/test_cli.py
+++ b/tests/test_cli.py
@@ -145,7 +145,30 @@ class TestParse(unittest.TestCase):
str(FIXTURES / "simple.eml"),
)
text = (pathlib.Path(out.strip()) / "manifest.json").read_text()
- self.assertNotIn("example.org", text)
+ self.assertNotIn("you@example.org", text)
+ # The bare domain is still barred everywhere the IOCs live. The
+ # headers block is the one exception and it is a NARROW one: the
+ # whitelist publishes the boundary Received line and
+ # Authentication-Results, and both name our own receiving relay in a
+ # "by"/authserv-id clause. That is the user's mail host, not the
+ # user's identity, and a desk learns it from the report's own From
+ # regardless. The address itself must still be absent, which the
+ # assertion above and report_headers' own tests cover.
+ import json
+
+ manifest = json.loads(text)
+ headers = manifest.pop("headers")
+ self.assertNotIn("example.org", json.dumps(manifest))
+ # And nothing shaped like an address survives in the exception.
+ # Both spellings: you%40example.org is not a hypothetical, it is why
+ # leaky.eml exists, and docs/plans/2026-09-09-contacts.md records a
+ # From of phish@victim%40example.org.invalid.
+ blob = json.dumps(headers)
+ self.assertNotIn("@example.org", blob)
+ self.assertNotIn("you%40example.org", blob)
+ names = [name for name, _ in headers]
+ for name in ("To", "Cc", "Delivered-To", "X-Original-To"):
+ self.assertNotIn(name, names)
def test_a_missing_config_points_at_init(self):
code, _, err = self._run(
diff --git a/tests/test_offline.py b/tests/test_offline.py
index 19d9cee..547bd96 100644
--- a/tests/test_offline.py
+++ b/tests/test_offline.py
@@ -53,6 +53,14 @@ class NothingOpensASocket(unittest.TestCase):
b"Subject: test\r\n\r\nbody\r\n")
parse.iocs(raw, trusted=["192.0.2.0/24"])
+ def test_selecting_report_headers_opens_no_socket(self):
+ # A new entry point into the parse path, so it is held to the same
+ # guarantee: choosing what to publish resolves nothing.
+ raw = (b"Received: from relay.example.invalid ([192.0.2.10])\r\n"
+ b"From: sender@example.invalid\r\n"
+ b"Subject: test\r\n\r\nbody\r\n")
+ parse.report_headers(raw, trusted=["192.0.2.0/24"])
+
def test_resolving_with_an_injected_fetch_opens_no_socket(self):
iocs = [{"id": "ioc-1", "type": "ipv4", "value": "198.51.100.7"}]
bootstraps = {
diff --git a/tests/test_parse.py b/tests/test_parse.py
index 11345c7..6e3b7a4 100644
--- a/tests/test_parse.py
+++ b/tests/test_parse.py
@@ -290,5 +290,141 @@ class TestIocAssembly(unittest.TestCase):
self.assertNotIn("you%40example.org", blob)
+class ReportHeaders(unittest.TestCase):
+ def test_the_whitelist_keeps_what_a_desk_needs(self):
+ headers = parse.report_headers(load("reportable.eml"),
+ trusted=["192.0.2.0/24"])
+ names = [name for name, _ in headers]
+ for wanted in ("From", "Subject", "Date", "Message-ID", "Reply-To",
+ "Return-Path", "Authentication-Results", "Received-SPF"):
+ self.assertIn(wanted, names)
+
+ def test_recipient_headers_never_survive_the_whitelist(self):
+ # The first property, at the one place a report reproduces header
+ # text verbatim. A blacklist would have to remember each of these;
+ # the whitelist never names them at all.
+ headers = parse.report_headers(load("reportable.eml"),
+ trusted=["192.0.2.0/24"])
+ names = [name for name, _ in headers]
+ blob = repr(headers)
+ for name in ("To", "Cc", "Delivered-To", "X-Original-To"):
+ self.assertNotIn(name, names)
+ self.assertNotIn("victim@example.org", blob)
+ self.assertNotIn("colleague@example.org", blob)
+
+ def test_received_stops_at_the_boundary_hop(self):
+ # 192.0.2.10 is ours, so its Received line is our own infrastructure
+ # and must not be published; the hop below it is the one being
+ # reported and is kept.
+ headers = parse.report_headers(load("reportable.eml"),
+ trusted=["192.0.2.0/24"])
+ received = [value for name, value in headers if name == "Received"]
+ self.assertEqual(len(received), 1)
+ self.assertIn("203.0.113.42", received[0])
+ self.assertNotIn("mx.example.org with ESMTP id abc123", received[0])
+
+ def test_the_published_hop_carries_no_envelope_recipient(self):
+ # The boundary Received line is written by OUR OWN relay, and its
+ # optional "for <addr>" clause is the envelope recipient: the
+ # victim's address, verbatim, in the one header a report reproduces
+ # in full. Truncating the chain is not enough on its own.
+ headers = parse.report_headers(load("reportable.eml"),
+ trusted=["192.0.2.0/24"])
+ received = [value for name, value in headers if name == "Received"]
+ self.assertNotIn("you@example.org", received[0])
+ self.assertNotIn("for <", received[0])
+ # The rest of the hop survives; this is a cut, not a blanking.
+ self.assertIn("203.0.113.42", received[0])
+
+ def test_every_for_clause_shape_loses_the_address(self):
+ # RFC 5321 4.4 puts For inside Opt-info, so With, ID, Via or a CFWS
+ # comment may legitimately follow it, and its ABNF is
+ # 1*( Path / Mailbox ) where Mailbox carries no angle brackets.
+ # Anchoring on "for" being immediately followed by the clause
+ # terminator matched only the neatest shape and let four routine
+ # ones through, each publishing the victim's address.
+ hop = "from a.invalid (a.invalid [203.0.113.5]) by mx.example.org "
+ shapes = (
+ "for <you@example.org> (envelope-from <b@c.invalid>); Mon, 07 Sep 2026 09:12:40 +0000",
+ "for you@example.org; Mon, 07 Sep 2026 09:12:40 +0000",
+ "for <you@example.org> with ESMTP; Mon, 07 Sep 2026 09:12:40 +0000",
+ "id qq; Mon, 07 Sep 2026 09:12:40 +0000 (for <you@example.org>)",
+ "for <you@example.org>; Mon, 07 Sep 2026 09:12:40 +0000",
+ )
+ for tail in shapes:
+ with self.subTest(tail=tail):
+ stripped = parse._strip_envelope_recipient(hop + tail)
+ self.assertNotIn("you@example.org", stripped)
+ # The hop's own evidence survives: this is a cut, not a
+ # blanking, and a rule that ate the line would pass the
+ # assertion above while destroying the report.
+ self.assertIn("203.0.113.5", stripped)
+ self.assertIn("mx.example.org", stripped)
+
+ def test_stripping_leaves_no_doubled_space_or_stray_separator(self):
+ # Cosmetic in isolation, but the result is published verbatim to a
+ # third party, so a mangled line reads as a broken tool.
+ hop = ("from a.invalid (a.invalid [203.0.113.5]) by mx.example.org"
+ " for <you@example.org>; Mon, 07 Sep 2026 09:12:40 +0000")
+ stripped = parse._strip_envelope_recipient(hop)
+ self.assertNotIn(" ", stripped)
+ self.assertNotIn(" ;", stripped)
+ self.assertIn("mx.example.org; Mon", stripped)
+
+ def test_a_comment_holding_only_the_clause_leaves_no_debris(self):
+ hop = ("from a.invalid (a.invalid [203.0.113.5]) by mx.example.org"
+ " id qq; Mon, 07 Sep 2026 09:12:40 +0000 (for <you@example.org>)")
+ stripped = parse._strip_envelope_recipient(hop)
+ self.assertNotIn("you@example.org", stripped)
+ self.assertFalse(stripped.endswith("("))
+ self.assertTrue(stripped.endswith("+0000"))
+
+ def test_the_envelope_sender_comment_survives_the_cut(self):
+ # envelope-from is the SENDER, which is what the report is about, so
+ # cutting the recipient must not take it along.
+ hop = ("from a.invalid (a.invalid [203.0.113.5]) by mx.example.org"
+ " for <you@example.org> (envelope-from <bounce@sender.invalid>);"
+ " Mon, 07 Sep 2026 09:12:40 +0000")
+ stripped = parse._strip_envelope_recipient(hop)
+ self.assertNotIn("you@example.org", stripped)
+ self.assertIn("bounce@sender.invalid", stripped)
+
+ def test_the_whitelist_does_not_filter_attacker_free_text(self):
+ # A DOCUMENTED LIMIT, not a guarantee. The spec keeps Subject and the
+ # From display name knowing both are attacker-controlled free text,
+ # because they are what lets a desk recognise a campaign. An attacker
+ # who writes the recipient's own address into one, obfuscated or not,
+ # gets it published: the whitelist governs WHICH headers travel, never
+ # what is inside one.
+ #
+ # This is asserted so the limit is visible and deliberate. Do not
+ # "fix" it by filtering free text, which is the judgement-shaped
+ # problem AGENTS.md names as the source of every leak here. The
+ # sweep over real mail is what covers this class, per AGENTS.md.
+ raw = (b"Received: from a.invalid (a.invalid [203.0.113.5])"
+ b" by mx.example.org with ESMTP id X;"
+ b" Mon, 07 Sep 2026 09:12:40 +0000\r\n"
+ b"From: <phish@sender.invalid>\r\n"
+ b"Subject: Verify you%40example.org\r\n\r\nbody\r\n")
+ headers = parse.report_headers(raw, trusted=["192.0.2.0/24"])
+ subject = dict(headers)["Subject"]
+ self.assertIn("you%40example.org", subject)
+
+ def test_a_forged_chain_publishes_no_hop_below_the_boundary(self):
+ # The same job test_a_forged_chain_stops_at_the_first_untrusted_hop
+ # does for sending_ip(), asserted over what actually gets published:
+ # 198.51.100.7 is an innocent party the attacker named.
+ headers = parse.report_headers(load("forged-chain.eml"),
+ trusted=["192.0.2.0/24"])
+ received = [value for name, value in headers if name == "Received"]
+ # Asserted in BOTH directions: dropping Received altogether would
+ # satisfy the "not published" half on its own, and a test that
+ # passes when the feature is missing protects nothing.
+ self.assertEqual(len(received), 1)
+ self.assertIn("203.0.113.99", received[0])
+ self.assertTrue(all("198.51.100.7" not in value for value in received))
+ self.assertTrue(all("198.51.100.8" not in value for value in received))
+
+
if __name__ == "__main__":
unittest.main()
diff --git a/tests/test_report.py b/tests/test_report.py
new file mode 100644
index 0000000..6a340c9
--- /dev/null
+++ b/tests/test_report.py
@@ -0,0 +1,1600 @@
+import copy
+import email
+import email.policy
+import json
+import unittest
+
+from abusectl import report
+
+
+class Grouping(unittest.TestCase):
+ def test_two_contacts_at_one_address_become_one_destination(self):
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ {"iocs": ["ioc-2"], "query": "example.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual(len(destinations), 1)
+ self.assertEqual(destinations[0]["target"], "abuse@host.invalid")
+ self.assertEqual(destinations[0]["iocs"], ["ioc-1", "ioc-2"])
+
+ def test_a_contact_with_two_addresses_reaches_both_desks(self):
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["a@host.invalid", "b@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual([d["target"] for d in destinations],
+ ["a@host.invalid", "b@host.invalid"])
+ # Both desks carry the indicator, and each gets its own id: a
+ # destination that reached only one desk, or two rows sharing an
+ # id, would pass an assertion on the sorted targets alone.
+ self.assertEqual([d["iocs"] for d in destinations],
+ [["ioc-1"], ["ioc-1"]])
+ self.assertEqual(len({d["id"] for d in destinations}), 2)
+ for destination in destinations:
+ self.assertEqual(destination["id"],
+ report.email_destination_id(
+ destination["target"]))
+
+ def test_a_contact_with_no_address_creates_no_destination(self):
+ """The contact that resolved must still produce its destination.
+
+ Asserted alongside one that DOES resolve, because "no destination
+ for this contact" is also what returning nothing at all looks
+ like, and that is not the behaviour being described.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "example.invalid", "abuse": [],
+ "source": "rdap", "error": "no abuse role published"},
+ {"iocs": ["ioc-2"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual([d["target"] for d in destinations],
+ ["abuse@host.invalid"])
+ self.assertEqual(destinations[0]["iocs"], ["ioc-2"])
+
+ def test_destinations_carry_stable_ids_and_pending_status(self):
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(contacts)[0]
+ self.assertEqual(destination["id"],
+ report.email_destination_id("abuse@host.invalid"))
+ self.assertEqual(destination["kind"], "email")
+ self.assertEqual(destination["status"], "pending")
+
+ def test_an_id_is_the_literal_shape_a_reviewer_will_read(self):
+ """Pin the shape, since it becomes a filename in bodies/.
+
+ Computed by hand rather than by calling the code under test, so
+ this fails if the derivation changes rather than following it.
+ """
+ self.assertEqual(report.email_destination_id("abuse@host.invalid"),
+ "email-bc50e369")
+
+ def test_ids_are_derived_per_destination_not_per_contact(self):
+ """A contact that resolved to no desk must not shift another's id.
+
+ The obvious implementation numbers destinations by position, and a
+ skipped contact then either burns an id or renumbers the rest.
+ Both are wrong for the same reason: an id names a desk.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "example.invalid", "abuse": [],
+ "source": "rdap", "error": "no abuse role published"},
+ {"iocs": ["ioc-2"], "query": "198.51.100.7",
+ "abuse": ["a@host.invalid"], "source": "rdap"},
+ {"iocs": ["ioc-3"], "query": "198.51.100.8",
+ "abuse": ["b@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual([d["id"] for d in destinations],
+ [report.email_destination_id("a@host.invalid"),
+ report.email_destination_id("b@host.invalid")])
+ self.assertEqual([d["target"] for d in destinations],
+ ["a@host.invalid", "b@host.invalid"])
+
+ def test_one_desk_listed_twice_by_one_contact_is_one_destination(self):
+ """A duplicate in a contact's own abuse list must not duplicate a desk.
+
+ RDAP jCards are attacker-adjacent data: an entity can publish the
+ same address in two vcard rows, and one destination per ADDRESS is
+ the rule regardless of how many rows produced it.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid", "abuse@host.invalid"],
+ "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual(len(destinations), 1)
+ self.assertEqual(destinations[0]["iocs"], ["ioc-1"])
+
+ def test_one_desk_spelled_with_two_domain_cases_is_one_destination(self):
+ """A domain is case-insensitive, so two spellings are one desk."""
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@Host.Invalid"], "source": "rdap"},
+ {"iocs": ["ioc-2"], "query": "example.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual(len(destinations), 1)
+ self.assertEqual(destinations[0]["iocs"], ["ioc-1", "ioc-2"])
+ self.assertEqual(destinations[0]["target"], "abuse@Host.Invalid")
+
+ def test_the_domain_folds_under_a_local_part_that_does_not(self):
+ """Isolate the domain fold from the role fold.
+
+ The version of this test that first shipped used a lowercase
+ local part throughout, so it exercised only the domain and passed
+ while a capitalised role name produced two destinations.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["J.Smith@Host.Invalid"], "source": "rdap"},
+ {"iocs": ["ioc-2"], "query": "example.invalid",
+ "abuse": ["J.Smith@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual(len(destinations), 1)
+ self.assertEqual(destinations[0]["iocs"], ["ioc-1", "ioc-2"])
+
+ def test_two_spellings_of_one_desk_share_an_id(self):
+ """The id derives from the same normalised form the grouping uses.
+
+ Otherwise the spelling RDAP happened to publish first would decide
+ a body's filename, and a re-run that saw the other spelling first
+ would look like a different desk.
+ """
+ self.assertEqual(report.email_destination_id("abuse@Host.Invalid"),
+ report.email_destination_id("abuse@host.invalid"))
+
+ def test_a_role_mailbox_folds_in_both_halves(self):
+ """"Abuse@Host.Invalid" and "abuse@host.invalid" are one desk.
+
+ The case that first shipped folded the domain only, so a jCard
+ publishing the role name capitalised produced two destinations and
+ two mails to one desk. RFC 2142 mandates the role mailboxes and
+ requires them case-insensitive, so no host runs "Abuse@" and
+ "abuse@" as different desks.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["Abuse@Host.Invalid"], "source": "rdap"},
+ {"iocs": ["ioc-2"], "query": "example.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual(len(destinations), 1)
+ self.assertEqual(destinations[0]["iocs"], ["ioc-1", "ioc-2"])
+ self.assertEqual(destinations[0]["target"], "Abuse@Host.Invalid")
+
+ def test_one_contact_publishing_a_role_mailbox_twice_folds_it(self):
+ """The same fold applies within one contact's own abuse list.
+
+ rdap.abuse_addresses dedupes case-sensitively, so a jCard with two
+ vcard rows spelling the role differently delivers both here.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["Abuse@host.invalid", "abuse@host.invalid"],
+ "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual(len(destinations), 1)
+ self.assertEqual(destinations[0]["iocs"], ["ioc-1"])
+
+ def test_every_rfc2142_role_this_tool_can_meet_folds(self):
+ for role in ("abuse", "postmaster", "security", "noc", "hostmaster"):
+ with self.subTest(role=role):
+ self.assertEqual(
+ report.email_destination_id(f"{role.title()}@host.invalid"),
+ report.email_destination_id(f"{role}@host.invalid"))
+
+ def test_a_personal_local_part_is_left_alone(self):
+ """Only the receiving host knows whether ITS local parts fold.
+
+ A named mailbox is not a standardised role, so folding it could
+ silently merge two desks a host genuinely distinguishes and drop
+ one of them. Two mails to one desk is the lesser failure, and the
+ role names above are where the duplicate actually happens.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["J.Smith@host.invalid", "j.smith@host.invalid"],
+ "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual(sorted(d["target"] for d in destinations),
+ ["J.Smith@host.invalid", "j.smith@host.invalid"])
+ self.assertNotEqual(destinations[0]["id"], destinations[1]["id"])
+
+ def test_a_destination_starts_with_no_body(self):
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ self.assertIsNone(report.email_destinations(contacts)[0]["body"])
+
+ def test_the_contacts_passed_in_are_not_modified(self):
+ """The caller's contacts are the manifest's own array.
+
+ case.py is the only writer of a manifest, so a grouping pass that
+ edited what it was handed would write through it from outside.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ before = copy.deepcopy(contacts)
+ report.email_destinations(contacts)
+ self.assertEqual(contacts, before)
+
+ def test_a_destinations_ioc_list_is_its_own(self):
+ """Not aliased to the contact's list it was built from.
+
+ Holds today because the grouping starts a fresh list, but nothing
+ else pins it: an implementation that reused contact["iocs"] for a
+ single-contact destination would pass every other test here and
+ leave a destination and a contact sharing one list in a manifest
+ about to be written.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(contacts)[0]
+ destination["iocs"].append("ioc-2")
+ self.assertEqual(contacts[0]["iocs"], ["ioc-1"])
+
+ def test_the_same_contacts_produce_the_same_ids_twice(self):
+ """Ids must not depend on dict iteration luck or set ordering."""
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["b@host.invalid", "a@host.invalid"], "source": "rdap"},
+ {"iocs": ["ioc-2"], "query": "example.invalid",
+ "abuse": ["c@host.invalid"], "source": "rdap"},
+ ]
+ first = report.email_destinations(contacts)
+ second = report.email_destinations(contacts)
+ self.assertEqual([(d["id"], d["target"]) for d in first],
+ [(d["id"], d["target"]) for d in second])
+ self.assertEqual([d["target"] for d in first],
+ ["b@host.invalid", "a@host.invalid",
+ "c@host.invalid"])
+
+ def test_a_desks_id_survives_another_desk_appearing(self):
+ """An id names a DESK, not a position in this run's list.
+
+ Task 8 writes each body to bodies/<id>.xarf and records its hash
+ against that id. With a positional id, re-running contacts on a
+ case that gained an indicator renumbers every desk after the new
+ one, so bodies/<id>.xarf on disk belongs to a different desk than
+ the manifest's entry of that id, and the edit check compares one
+ desk's body against another's.
+ """
+ established = {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["b@host.invalid"], "source": "rdap"}
+ first = report.email_destinations([established])
+
+ # A later contacts run finds an indicator whose desk sorts ahead.
+ newcomer = {"iocs": ["ioc-2"], "query": "example.invalid",
+ "abuse": ["a@new.invalid"], "source": "rdap"}
+ second = report.email_destinations([newcomer, established])
+
+ by_target = {d["target"]: d["id"] for d in second}
+ self.assertEqual(by_target["b@host.invalid"], first[0]["id"])
+ self.assertNotEqual(by_target["a@new.invalid"], first[0]["id"])
+
+ def test_an_ids_position_does_not_leak_into_it(self):
+ """The same desk alone and third in a list gets one id."""
+ alone = report.email_destinations([
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["desk@host.invalid"], "source": "rdap"},
+ ])
+ crowded = report.email_destinations([
+ {"iocs": ["ioc-2"], "query": "198.51.100.8",
+ "abuse": ["one@host.invalid", "two@host.invalid"],
+ "source": "rdap"},
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["desk@host.invalid"], "source": "rdap"},
+ ])
+ self.assertEqual(crowded[2]["target"], "desk@host.invalid")
+ self.assertEqual(crowded[2]["id"], alone[0]["id"])
+
+
+class Unreportable(unittest.TestCase):
+ def test_an_ioc_with_no_desk_is_listed_with_its_reason(self):
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ {"iocs": ["ioc-2", "ioc-3"], "query": "example.invalid",
+ "abuse": [], "source": "rdap",
+ "error": "no abuse role published"},
+ ]
+ self.assertEqual(
+ report.unreportable(contacts),
+ [
+ {"ioc": "ioc-2", "reason": "no abuse role published"},
+ {"ioc": "ioc-3", "reason": "no abuse role published"},
+ ],
+ )
+
+ def test_a_missing_reason_still_produces_an_entry(self):
+ contacts = [{"iocs": ["ioc-9"], "query": "x.invalid", "abuse": [],
+ "source": "rdap"}]
+ self.assertEqual(
+ report.unreportable(contacts),
+ [{"ioc": "ioc-9", "reason": "no abuse address resolved"}],
+ )
+
+ def test_an_empty_reason_does_not_read_as_no_reason(self):
+ """`error: ""` must not be reported as the literal empty string.
+
+ A contact entry is written by contacts.resolve, but a manifest is
+ a file on disk that a user edits during review. An empty reason
+ renders as a blank cell in the report the user reads, which says
+ nothing at all; the default at least says what happened.
+ """
+ contacts = [{"iocs": ["ioc-9"], "query": "x.invalid", "abuse": [],
+ "source": "rdap", "error": ""}]
+ self.assertEqual(
+ report.unreportable(contacts),
+ [{"ioc": "ioc-9", "reason": "no abuse address resolved"}],
+ )
+
+ def test_nothing_unreportable_is_an_empty_list_not_an_error(self):
+ contacts = [{"iocs": ["ioc-1"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"}]
+ self.assertEqual(report.unreportable(contacts), [])
+
+ def test_an_ioc_that_reached_a_desk_elsewhere_is_not_unreportable(self):
+ """Hosts fold, so one IOC can sit in a resolved and an unresolved
+ contact at once. It IS reportable, and listing it says otherwise.
+
+ The plan's implementation listed it regardless, which puts an
+ indicator in both the destination list and the "no desk found"
+ list of one manifest. A reviewer reading the second acts on an
+ indicator that is already on its way to a desk, and the whole
+ point of the array is that it can be trusted without diffing.
+ """
+ contacts = [
+ {"iocs": ["ioc-1", "ioc-2"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ {"iocs": ["ioc-2", "ioc-3"], "query": "example.invalid",
+ "abuse": [], "source": "rdap",
+ "error": "no abuse role published"},
+ ]
+ self.assertEqual(
+ report.unreportable(contacts),
+ [{"ioc": "ioc-3", "reason": "no abuse role published"}],
+ )
+
+ def test_one_ioc_unresolved_twice_is_listed_once(self):
+ """Two contacts, both unresolved, one shared indicator.
+
+ A duplicate row is a second line in the report about one
+ indicator, and the reasons may differ, so which one wins has to
+ be decided rather than left to whichever contact came last.
+ First reason seen wins, matching the first-seen ordering the
+ destinations use.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "example.invalid", "abuse": [],
+ "source": "rdap", "error": "no abuse role published"},
+ {"iocs": ["ioc-1"], "query": "198.51.100.7", "abuse": [],
+ "source": "rdap", "error": "no rdap server for this tld, "
+ "or no answer"},
+ ]
+ self.assertEqual(
+ report.unreportable(contacts),
+ [{"ioc": "ioc-1", "reason": "no abuse role published"}],
+ )
+
+ def test_the_contacts_passed_in_are_not_modified(self):
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "example.invalid", "abuse": [],
+ "source": "rdap", "error": "no abuse role published"},
+ {"iocs": ["ioc-2"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ before = copy.deepcopy(contacts)
+ report.unreportable(contacts)
+ self.assertEqual(contacts, before)
+
+
+class MalformedAddresses(unittest.TestCase):
+ """An abuse "address" with no @ cannot be mailed.
+
+ RDAP jCard data is third-party and occasionally malformed, and a
+ destination built from such a value carries an unsendable target with
+ status "pending". That is the failure mode the unreportable array
+ exists to prevent: the indicator appears reportable, no desk ever
+ receives it, and nothing in the manifest says so.
+ """
+
+ def test_a_target_with_no_at_creates_no_destination(self):
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "example.invalid",
+ "abuse": ["not-an-address"], "source": "rdap"},
+ {"iocs": ["ioc-2"], "query": "198.51.100.7",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual([d["target"] for d in destinations],
+ ["abuse@host.invalid"])
+ self.assertEqual(destinations[0]["iocs"], ["ioc-2"])
+
+ def test_an_ioc_whose_only_address_is_malformed_is_unreportable(self):
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "example.invalid",
+ "abuse": ["not-an-address"], "source": "rdap"},
+ ]
+ self.assertEqual(
+ report.unreportable(contacts),
+ [{"ioc": "ioc-1",
+ "reason": "no usable abuse address published"}],
+ )
+
+ def test_a_usable_address_beside_a_malformed_one_still_reports(self):
+ """The good half of a jCard must survive the bad half.
+
+ Discarding the contact wholesale would lose a real desk over a
+ neighbouring malformed row.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "example.invalid",
+ "abuse": ["not-an-address", "abuse@host.invalid"],
+ "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ self.assertEqual([d["target"] for d in destinations],
+ ["abuse@host.invalid"])
+ self.assertEqual(report.unreportable(contacts), [])
+
+ def test_an_addresss_own_error_is_not_overwritten_by_the_default(self):
+ """A contact that has both a reason and a malformed address.
+
+ The contact's own error says more than "no usable address", so it
+ wins; the default is only for a contact that offered no reason.
+ """
+ contacts = [
+ {"iocs": ["ioc-1"], "query": "example.invalid",
+ "abuse": ["not-an-address"], "source": "rdap",
+ "error": "no abuse role published"},
+ ]
+ self.assertEqual(
+ report.unreportable(contacts),
+ [{"ioc": "ioc-1", "reason": "no abuse role published"}],
+ )
+
+ def test_the_two_lists_partition_every_indicator(self):
+ """The invariant the pair is for: each IOC is in exactly one.
+
+ Every other test here pins one side. This pins the relationship,
+ which is what a reviewer actually relies on: an indicator missing
+ from both is silently unreported, and one in both is reported and
+ also flagged as unreported. Both failures come from the two
+ functions disagreeing about what counts as a desk, so they are
+ asserted against one input that exercises every branch.
+ """
+ contacts = [
+ {"iocs": ["ioc-1", "ioc-2"], "query": "198.51.100.7",
+ "abuse": ["Abuse@Host.Invalid"], "source": "rdap"},
+ {"iocs": ["ioc-2", "ioc-3"], "query": "example.invalid",
+ "abuse": [], "source": "rdap",
+ "error": "no abuse role published"},
+ {"iocs": ["ioc-4"], "query": "other.invalid",
+ "abuse": ["not-an-address"], "source": "rdap"},
+ {"iocs": ["ioc-5"], "query": "mixed.invalid",
+ "abuse": ["broken", "abuse@host.invalid"], "source": "rdap"},
+ ]
+ destinations = report.email_destinations(contacts)
+ reported = {i for d in destinations for i in d["iocs"]}
+ flagged = {e["ioc"] for e in report.unreportable(contacts)}
+ every = {i for c in contacts for i in c["iocs"]}
+
+ self.assertEqual(reported & flagged, set())
+ self.assertEqual(reported | flagged, every)
+ self.assertEqual(reported, {"ioc-1", "ioc-2", "ioc-5"})
+ self.assertEqual(flagged, {"ioc-3", "ioc-4"})
+
+ def test_an_empty_or_whitespace_target_is_not_a_desk(self):
+ for value in ("", " ", "@host.invalid", "abuse@"):
+ with self.subTest(value=value):
+ contacts = [{"iocs": ["ioc-1"], "query": "example.invalid",
+ "abuse": [value], "source": "rdap"}]
+ self.assertEqual(report.email_destinations(contacts), [])
+ self.assertEqual(
+ report.unreportable(contacts),
+ [{"ioc": "ioc-1",
+ "reason": "no usable abuse address published"}])
+
+
+IDENTITY = {"name": "A Reporter", "org": "Example Consulting",
+ "email": "reporter@example.org"}
+
+MANIFEST = {
+ "format": 1,
+ "case_id": "2026-09-07-aaaa",
+ "iocs": [
+ {"id": "ioc-1", "type": "ipv4", "value": "203.0.113.42",
+ "origin": "received-chain", "confidence": "boundary-hop"},
+ {"id": "ioc-2", "type": "url",
+ "value": "http://login.sender.invalid/verify?id=REDACTED",
+ "origin": "body"},
+ ],
+ "auth": {"spf": "fail", "dkim": "none", "dmarc": "fail"},
+ "headers": [
+ ("From", '"Example Bank" <phish@sender.invalid>'),
+ ("Subject", "Your account requires verification"),
+ ("Date", "Mon, 07 Sep 2026 09:12:40 +0000"),
+ ],
+ "contacts": [
+ {"iocs": ["ioc-1", "ioc-2"], "query": "203.0.113.42",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ],
+}
+
+# 120 characters, well past the 72-column wrap, built from .invalid only.
+LONG_URL = ("http://very-long-host-name.example.invalid/a/rather/deep/path/"
+ "segment/tree/verify?campaign=REDACTED&id=REDACTED")
+
+
+class TextPart(unittest.TestCase):
+ def setUp(self):
+ destination = report.email_destinations(MANIFEST["contacts"])[0]
+ self.text = report.text_part(MANIFEST, destination, IDENTITY)
+
+ def test_the_redaction_note_is_always_present(self):
+ self.assertIn("Recipient identifiers", self.text)
+
+ def test_the_reporter_identity_appears(self):
+ self.assertIn("A Reporter", self.text)
+ self.assertIn("Example Consulting", self.text)
+ self.assertIn("reporter@example.org", self.text)
+
+ def test_the_destinations_own_indicators_appear(self):
+ self.assertIn("203.0.113.42", self.text)
+ self.assertIn("http://login.sender.invalid/verify?id=REDACTED",
+ self.text)
+
+ def test_an_indicator_belonging_to_another_desk_does_not_appear(self):
+ manifest = dict(MANIFEST)
+ manifest["iocs"] = MANIFEST["iocs"] + [
+ {"id": "ioc-9", "type": "ipv4", "value": "192.0.2.99",
+ "origin": "received-chain"},
+ ]
+ destination = report.email_destinations(MANIFEST["contacts"])[0]
+ text = report.text_part(manifest, destination, IDENTITY)
+ self.assertNotIn("192.0.2.99", text)
+
+ def test_no_line_exceeds_seventy_two_columns(self):
+ for line in self.text.splitlines():
+ self.assertLessEqual(len(line), 72, line)
+
+ # --- the parts the plan got wrong -------------------------------------
+
+ def _with(self, **fields) -> str:
+ manifest = copy.deepcopy(MANIFEST)
+ manifest.update(fields)
+ destination = report.email_destinations(manifest["contacts"])[0]
+ return report.text_part(manifest, destination, IDENTITY)
+
+ def test_a_long_url_is_whole_and_still_within_seventy_two_columns(self):
+ """A truncated URL is a WRONG indicator, not a shortened one.
+
+ The plan wrapped three lines with a `[:72]` slice and left the
+ indicator list unwrapped. Both halves are the same defect: a desk
+ acting on a prefix acts on a resource that is not the one reported,
+ and a prefix reads as complete because nothing says otherwise.
+
+ So the value must survive intact, reassemblable by a reader, and
+ every line must still fit. Asserting only "the URL is in the text"
+ would pass on a long unwrapped line, and asserting only the column
+ limit would pass on a truncation; the two together admit neither.
+ """
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [
+ {"id": "ioc-1", "type": "url", "value": LONG_URL,
+ "origin": "body"},
+ ]
+ manifest["contacts"] = [
+ {"iocs": ["ioc-1"], "query": "example.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ text = report.text_part(manifest, destination, IDENTITY)
+
+ for line in text.splitlines():
+ self.assertLessEqual(len(line), 72, line)
+ # Rejoining the continuation lines must give back the exact value.
+ self.assertIn(LONG_URL, report.unwrap(text))
+
+ def test_a_long_header_value_is_not_truncated(self):
+ """A header is what the message DECLARED, and a cut one misstates it.
+
+ The subject here is attacker-controlled free text of a length no
+ column limit accommodates. Truncating it publishes something the
+ message did not say.
+ """
+ long_subject = ("Your account requires verification before "
+ "the end of the working day or it will be "
+ "suspended permanently")
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["headers"] = [("Subject", long_subject)]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ text = report.text_part(manifest, destination, IDENTITY)
+
+ for line in text.splitlines():
+ self.assertLessEqual(len(line), 72, line)
+ self.assertIn(long_subject, report.unwrap(text))
+
+ def test_a_long_reporter_identity_is_not_truncated(self):
+ """The identity is the one thing disclosed deliberately.
+
+ A cut address is an address nobody can reply to, which defeats the
+ line's only purpose. The plan sliced it at 72.
+ """
+ identity = {"name": "A Reporter With Rather A Long Name",
+ "org": "Example Consulting And Partners Limited",
+ "email": "a.reporter@consulting.example.org"}
+ destination = report.email_destinations(MANIFEST["contacts"])[0]
+ text = report.text_part(MANIFEST, destination, identity)
+
+ for line in text.splitlines():
+ self.assertLessEqual(len(line), 72, line)
+ joined = report.unwrap(text)
+ self.assertIn("A Reporter With Rather A Long Name", joined)
+ self.assertIn("a.reporter@consulting.example.org", joined)
+
+ def test_headers_survive_a_json_round_trip(self):
+ """case.load() returns lists, not tuples: JSON has no tuple.
+
+ The header block is the one place a pair is destructured, so it is
+ the one place the round-trip shape can break the report.
+ """
+ manifest = json.loads(json.dumps(MANIFEST))
+ self.assertIsInstance(manifest["headers"][0], list)
+ destination = report.email_destinations(manifest["contacts"])[0]
+ text = report.text_part(manifest, destination, IDENTITY)
+ self.assertIn("Your account requires verification", text)
+
+ def test_an_origin_reads_as_english_not_as_an_internal_token(self):
+ """`header-list_unsubscribe` is a parser's word, not a desk's.
+
+ A desk deciding whether to act needs to know where an indicator was
+ seen. An internal token with an underscore in it reads as debug
+ output and makes the whole report look machine-dumped.
+ """
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [
+ {"id": "ioc-1", "type": "url", "value": "http://a.invalid/x",
+ "origin": "header-list_unsubscribe"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ text = report.text_part(manifest, destination, IDENTITY)
+ self.assertNotIn("header-list_unsubscribe", text)
+ self.assertIn("List-Unsubscribe", text)
+
+ def test_an_unknown_origin_is_shown_rather_than_dropped(self):
+ """A newer parse.py may invent an origin this table does not know.
+
+ Dropping it would silently lose the one line saying where an
+ indicator came from, so an unknown token is shown as-is: ugly beats
+ absent, and it is visible enough to get the table updated.
+ """
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [
+ {"id": "ioc-1", "type": "url", "value": "http://a.invalid/x",
+ "origin": "some-future-origin"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ text = report.text_part(manifest, destination, IDENTITY)
+ self.assertIn("some-future-origin", text)
+
+ def test_a_boundary_hop_is_described_as_the_sending_ip(self):
+ self.assertIn("sending IP", self.text)
+
+ def test_an_ioc_id_a_destination_names_but_the_manifest_lacks(self):
+ """A manifest is a file the user edits, so the two can disagree.
+
+ The report must still be produced for the indicators that do exist
+ rather than raising, and must not invent a row for the missing one.
+ """
+ destination = {"id": "email-x", "kind": "email",
+ "target": "abuse@host.invalid",
+ "iocs": ["ioc-1", "ioc-404"], "body": None,
+ "status": "pending"}
+ text = report.text_part(MANIFEST, destination, IDENTITY)
+ self.assertIn("203.0.113.42", text)
+ self.assertNotIn("ioc-404", text)
+
+ def test_a_manifest_with_no_auth_or_headers_still_reports(self):
+ """Both blocks are optional and an empty one must not print a
+ heading with nothing under it."""
+ text = self._with(auth={}, headers=[])
+ self.assertIn("203.0.113.42", text)
+ self.assertNotIn("Message as declared", text)
+ self.assertNotIn("Authentication results", text)
+
+
+class BackslashRoundTrip(unittest.TestCase):
+ """A value's own backslash must never be read as a wrap marker.
+
+ The continuation marker is a trailing "\\", and a URL path may legally
+ end in one. Until this was fixed the two were indistinguishable, so an
+ attacker who read this source could append a backslash and make their
+ own indicator garble itself in the report an abuse desk reads. That is
+ an adversarial trigger on attacker-supplied text, not an edge case.
+
+ The property asserted throughout is the only one that closes it:
+ unwrap(text_part(...)) contains the value EXACTLY, for every value,
+ wrapped or not. Asserting "the value appears" without unwrap, or
+ asserting only on long values, both leave the short case open, and the
+ short case is the one that needs no wrapping to corrupt.
+ """
+
+ def _render_ioc(self, value: str) -> str:
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [{"id": "ioc-1", "type": "url", "value": value,
+ "origin": "body"}]
+ manifest["contacts"] = [
+ {"iocs": ["ioc-1"], "query": "example.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ return report.text_part(manifest, destination, IDENTITY)
+
+ def _assert_round_trips(self, value: str) -> None:
+ text = self._render_ioc(value)
+ for line in text.splitlines():
+ self.assertLessEqual(len(line), 72, line)
+ self.assertIn(value, report.unwrap(text))
+
+ def test_a_short_value_ending_in_a_backslash_survives(self):
+ """The case that needs no wrapping at all to corrupt.
+
+ Nothing is wrapped here, yet unwrap() used to eat the following
+ line, merging the indicator with its own origin annotation and
+ rendering "http://a.invalid/xseen in a link in the message body".
+ One corrupt line where there were two, and a wrong indicator.
+ """
+ self._assert_round_trips("http://a.invalid/x\\")
+
+ def test_a_long_value_ending_in_a_backslash_survives(self):
+ self._assert_round_trips("http://a.invalid/" + "b" * 90 + "\\")
+
+ def test_a_value_with_an_interior_backslash_survives(self):
+ self._assert_round_trips("http://a.invalid/x\\y/z")
+
+ def test_a_value_ending_in_two_backslashes_survives(self):
+ """Whatever escaping is chosen must not have its own off-by-one.
+
+ Doubling every backslash makes a trailing pair into four, and a
+ decoder that consumes them greedily or in the wrong order gives
+ back one backslash or three. This is the test that catches that.
+ """
+ self._assert_round_trips("http://a.invalid/x\\\\")
+
+ def test_adversarial_values_round_trip_exactly(self):
+ """A handful of shapes chosen to sit on the seams.
+
+ The two boundary values matter most: a value that exactly fills a
+ line and one a single character over it are where an off-by-one in
+ the wrap arithmetic lives, and a backslash landing exactly on the
+ break column is where escaping and wrapping interact.
+ """
+ indent = 2
+ room = 72 - indent
+ values = [
+ "http://a.invalid/x\\",
+ "http://a.invalid/x\\y/z",
+ "http://a.invalid/x\\\\",
+ "\\" + "a" * 40,
+ "a" * 40 + "\\",
+ "http://a.invalid/" + "b" * 90 + "\\",
+ "a" * room, # exactly fills the line
+ "a" * (room + 1), # one character over
+ "a" * (room - 1) + "\\", # backslash at the break
+ "a" * room + "\\",
+ "\\\\" + "c" * 80 + "\\\\",
+ ]
+ for value in values:
+ with self.subTest(value=value):
+ self._assert_round_trips(value)
+
+ def test_a_backslash_landing_on_the_break_column_survives(self):
+ """The case that a passing suite still missed.
+
+ Escaping doubles each backslash, and a break falling BETWEEN the
+ two halves of a pair splits the run unwrap() counts the parity of.
+ Both halves are then misread, a real marker reads as content, the
+ continuation line is orphaned and the tail of the value is silently
+ dropped. It needs a backslash at exactly the break column, so no
+ hand-written case found it; a randomised sweep failed 454 of 3538.
+
+ Walking the backslash across every position around the boundary is
+ what makes this deterministic rather than luck.
+ """
+ room = 72 - 2 # indent is two spaces for an indicator line
+ for offset in range(-4, 5):
+ position = room + offset
+ if position < 1:
+ continue
+ value = "a" * position + "\\" + "b" * 30
+ with self.subTest(offset=offset):
+ self._assert_round_trips(value)
+
+ def test_a_run_of_backslashes_across_the_break_survives(self):
+ """A run is where an off-by-one in the back-off hides.
+
+ Backing off one character is correct only if the character it lands
+ on is the first half of a pair; a run of three or four exercises
+ whether the parity test looks at the run rather than at one
+ character.
+ """
+ room = 72 - 2
+ for length in range(1, 6):
+ for offset in range(-3, 4):
+ position = room + offset
+ if position < 1:
+ continue
+ value = "a" * position + "\\" * length + "b" * 20
+ with self.subTest(length=length, offset=offset):
+ self._assert_round_trips(value)
+
+ def test_a_value_that_is_entirely_backslashes_survives(self):
+ """Escaping doubles the length, so this is the worst case for both
+ the wrap arithmetic and the parity test at once."""
+ for length in (1, 2, 3, 34, 35, 36, 70, 71):
+ with self.subTest(length=length):
+ self._assert_round_trips("\\" * length)
+
+ def test_an_attacker_subject_cannot_corrupt_the_header_block(self):
+ """Subject is attacker-controlled and sits beside headers it can eat.
+
+ This is worse than the URL case: a trailing backslash on Subject
+ used to swallow the following line, rendering
+ "Subject: Verify nowDate: Mon, 07 Sep 2026 09:12:40 +0000". The
+ attacker's own text destroys a DIFFERENT field's value, so the
+ block misstates what the message declared, which is the one thing
+ that block exists to report faithfully.
+ """
+ subject = "Verify now\\"
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["headers"] = [
+ ("Subject", subject),
+ ("Date", "Mon, 07 Sep 2026 09:12:40 +0000"),
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ text = report.text_part(manifest, destination, IDENTITY)
+
+ for line in text.splitlines():
+ self.assertLessEqual(len(line), 72, line)
+ joined = report.unwrap(text)
+ self.assertIn(f"Subject: {subject}", joined)
+ # The Date must survive intact rather than being absorbed.
+ self.assertIn("Date: Mon, 07 Sep 2026 09:12:40 +0000", joined)
+
+ def test_a_display_name_ending_in_a_backslash_survives(self):
+ """The From display name is attacker-controlled too, and a sweep
+ has already found a spoofed one."""
+ value = '"Example Bank\\" <phish@sender.invalid>'
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["headers"] = [("From", value),
+ ("Subject", "Your account requires check")]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ text = report.text_part(manifest, destination, IDENTITY)
+
+ joined = report.unwrap(text)
+ self.assertIn(f"From: {value}", joined)
+ self.assertIn("Subject: Your account requires check", joined)
+
+ def test_unwrap_reads_a_marker_by_parity_not_by_a_trailing_backslash(self):
+ """unwrap() is PUBLIC, so its input is not only our own output.
+
+ A case manifest is a file the user edits and a desk may script
+ against the text part, so unwrap() must decide correctly on a line
+ it did not generate. Inside generated text the wrap back-off means
+ a marker always follows an even run, so parity and a plain
+ endswith() agree and neither is distinguishable by a round-trip
+ test. They disagree here, on a line ending in an escaped pair and
+ nothing else: that is content, and the following line must NOT be
+ absorbed into it.
+ """
+ # "a\\" escaped is a value ending in one literal backslash, whole
+ # on its line. endswith() reads the second half as a marker.
+ self.assertEqual(report.unwrap(" a\\\\\n next"), " a\\\n next")
+ # An odd run IS a marker: two escaped halves plus the marker.
+ self.assertEqual(report.unwrap(" a\\\\\\\n next"), " a\\next")
+
+ def test_an_identity_containing_a_backslash_survives(self):
+ identity = {"name": "A Reporter\\", "org": "Example Consulting",
+ "email": "reporter@example.org"}
+ destination = report.email_destinations(MANIFEST["contacts"])[0]
+ text = report.text_part(MANIFEST, destination, identity)
+ self.assertIn("A Reporter\\", report.unwrap(text))
+ self.assertIn("Generated by abusectl.", report.unwrap(text))
+
+
+class FeedbackPart(unittest.TestCase):
+ def setUp(self):
+ destination = report.email_destinations(MANIFEST["contacts"])[0]
+ self.fields = report.feedback_fields(MANIFEST, destination)
+ self.lookup = dict(self.fields)
+
+ def test_the_three_rfc5965_required_fields_are_present(self):
+ self.assertEqual(self.lookup["Feedback-Type"], "abuse")
+ self.assertEqual(self.lookup["Version"], "1")
+ self.assertTrue(self.lookup["User-Agent"].startswith("abusectl/"))
+
+ def test_the_xarf_report_type_is_phishing(self):
+ self.assertEqual(self.lookup["Report-Type"], "phishing")
+
+ def test_source_is_the_primary_indicator(self):
+ self.assertEqual(self.lookup["Source"], "203.0.113.42")
+ self.assertEqual(self.lookup["Source-IP"], "203.0.113.42")
+
+ def test_every_url_appears_as_a_reported_uri(self):
+ uris = [value for name, value in self.fields if name == "Reported-Uri"]
+ self.assertEqual(
+ uris, ["http://login.sender.invalid/verify?id=REDACTED"]
+ )
+
+ def test_a_destination_with_no_ip_omits_source_ip(self):
+ contacts = [{"iocs": ["ioc-2"], "query": "sender.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"}]
+ destination = report.email_destinations(contacts)[0]
+ lookup = dict(report.feedback_fields(MANIFEST, destination))
+ self.assertNotIn("Source-IP", lookup)
+ self.assertEqual(
+ lookup["Source"], "http://login.sender.invalid/verify?id=REDACTED"
+ )
+
+ # --- the parts the plan got wrong -------------------------------------
+
+ def _fields_for(self, iocs: list[dict]) -> list[tuple[str, str]]:
+ """Render the machine part for a hand-built IOC list.
+
+ Every IOC reaches one destination, so the field list is exactly what
+ those indicators produce and nothing is filtered out behind the test.
+ """
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = iocs
+ manifest["contacts"] = [
+ {"iocs": [entry["id"] for entry in iocs], "query": "x.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ return report.feedback_fields(manifest, destination)
+
+ def test_source_ip_is_emitted_at_most_once(self):
+ """RFC 5965 says Source-IP appears "once maximum".
+
+ The plan emitted one per IP. A strict parser meeting a repeated
+ single-occurrence field either rejects the part or keeps whichever
+ occurrence it saw last, so the field a repeat was meant to add is
+ the field that displaces the primary one. Every IP still travels,
+ in the text part and in Reported-Uri's sibling below.
+ """
+ fields = self._fields_for([
+ {"id": "ioc-1", "type": "ipv4", "value": "203.0.113.42",
+ "origin": "received-chain", "confidence": "boundary-hop"},
+ {"id": "ioc-2", "type": "ipv4", "value": "203.0.113.43",
+ "origin": "received-chain"},
+ ])
+ ips = [v for n, v in fields if n == "Source-IP"]
+ self.assertEqual(ips, ["203.0.113.42"])
+ self.assertEqual(dict(fields)["Source"], "203.0.113.42")
+
+ def test_no_field_appears_twice_unless_the_rfc_allows_it(self):
+ """The invariant behind the test above, stated once for every field.
+
+ Reported-Uri and Reported-Domain are "any number of times"; every
+ other field this module emits is once-maximum. Asserting only on
+ Source-IP would let the next repeated field ship unnoticed.
+ """
+ fields = self._fields_for([
+ {"id": "ioc-1", "type": "ipv4", "value": "203.0.113.42",
+ "origin": "received-chain"},
+ {"id": "ioc-2", "type": "ipv6", "value": "2001:db8::1",
+ "origin": "received-chain"},
+ {"id": "ioc-3", "type": "url", "value": "http://a.invalid/x",
+ "origin": "body"},
+ {"id": "ioc-4", "type": "url", "value": "http://b.invalid/y",
+ "origin": "body"},
+ {"id": "ioc-5", "type": "domain", "value": "a.invalid",
+ "origin": "header-from"},
+ {"id": "ioc-6", "type": "domain", "value": "b.invalid",
+ "origin": "header-reply_to"},
+ ])
+ seen: dict[str, int] = {}
+ for name, _ in fields:
+ seen[name] = seen.get(name, 0) + 1
+ repeatable = {"Reported-Uri", "Reported-Domain"}
+ for name, count in seen.items():
+ if name not in repeatable:
+ self.assertEqual(count, 1, f"{name} appeared {count} times")
+ self.assertEqual(seen["Reported-Uri"], 2)
+ self.assertEqual(seen["Reported-Domain"], 2)
+
+ def test_an_ipv6_indicator_fills_source_ip_too(self):
+ """"ipv6" is a distinct type string from parse.iocs().
+
+ A branch testing only for "ipv4" drops every IPv6 sender, and the
+ given tests use IPv4 throughout so none of them would notice.
+ """
+ fields = dict(self._fields_for([
+ {"id": "ioc-1", "type": "ipv6", "value": "2001:db8::1",
+ "origin": "received-chain"},
+ ]))
+ self.assertEqual(fields["Source"], "2001:db8::1")
+ self.assertEqual(fields["Source-IP"], "2001:db8::1")
+
+ def test_a_destination_with_no_typed_indicator_omits_source(self):
+ """An empty Source is worse than an absent one.
+
+ "Source:" with nothing after it asserts that the thing being
+ reported is the empty string. A 5965 parser reading a present-but-
+ empty field has been told a value; reading no field it has been
+ told nothing, which is the truth. Only sha256 and observation
+ indicators reach a desk here, and both belong in the text part.
+ """
+ fields = self._fields_for([
+ {"id": "ioc-1", "type": "observation",
+ "value": "display-name-carries-address",
+ "origin": "display-name-from"},
+ ])
+ lookup = dict(fields)
+ self.assertNotIn("Source", lookup)
+ self.assertNotIn("Source-IP", lookup)
+ # The envelope is still well formed: a desk gets a valid part.
+ self.assertEqual(lookup["Feedback-Type"], "abuse")
+ self.assertEqual(lookup["Version"], "1")
+
+ def test_a_type_this_module_does_not_place_is_not_invented_into_one(self):
+ """sha256 and observation have no 5965 or x-arf field.
+
+ Neither is a Source, a Reported-Uri or a Reported-Domain, and
+ forcing one into the nearest-looking field would tell a desk that a
+ file hash is a URI. They travel in the human part, which is where a
+ desk reads what an attachment was.
+ """
+ fields = self._fields_for([
+ {"id": "ioc-1", "type": "ipv4", "value": "203.0.113.42",
+ "origin": "received-chain"},
+ {"id": "ioc-2", "type": "sha256", "value": "a" * 64,
+ "origin": "attachment", "filename": "invoice.zip"},
+ {"id": "ioc-3", "type": "observation",
+ "value": "display-name-carries-address",
+ "origin": "display-name-from"},
+ ])
+ blob = repr(fields)
+ self.assertNotIn("a" * 64, blob)
+ self.assertNotIn("display-name-carries-address", blob)
+ self.assertEqual(dict(fields)["Source"], "203.0.113.42")
+
+ def test_an_ioc_id_the_manifest_lacks_is_skipped_not_raised(self):
+ """A manifest is a file the user edits, so the two can disagree.
+
+ text_part() already tolerates this; the machine part indexed with
+ destination["iocs"] straight into a dict would raise instead, and
+ the two parts of one document must not disagree about whether the
+ case can be reported at all.
+ """
+ destination = {"id": "email-x", "kind": "email",
+ "target": "abuse@host.invalid",
+ "iocs": ["ioc-1", "ioc-404"], "body": None,
+ "status": "pending"}
+ lookup = dict(report.feedback_fields(MANIFEST, destination))
+ self.assertEqual(lookup["Source"], "203.0.113.42")
+ self.assertNotIn("ioc-404", repr(lookup))
+
+ def test_a_destination_with_no_iocs_key_still_renders(self):
+ """destination.get("iocs"), not destination["iocs"]."""
+ destination = {"id": "email-x", "kind": "email",
+ "target": "abuse@host.invalid", "body": None,
+ "status": "pending"}
+ lookup = dict(report.feedback_fields(MANIFEST, destination))
+ self.assertEqual(lookup["Feedback-Type"], "abuse")
+
+ def test_arrival_date_is_not_taken_from_the_senders_date_header(self):
+ """RFC 5965: Arrival-Date is when the generating ADMD's MTA received
+ the message. The Date header is when the SENDER CLAIMS it was sent.
+
+ The plan copied Date into Arrival-Date. On a phishing message that
+ header is attacker-controlled free text, so the report would assert
+ as our own observation a timestamp the attacker chose, and a desk
+ correlating it against their own logs would look in the wrong place
+ or find nothing and discount the report.
+
+ The honest source is the boundary Received hop's own timestamp,
+ which parse.report_headers() already publishes. Parsing one is a
+ date parser this task does not need, so the field is OMITTED: 5965
+ makes it optional, and an absent optional field misstates nothing.
+ """
+ lookup = dict(self.fields)
+ self.assertNotIn("Arrival-Date", lookup)
+ self.assertNotIn("Mon, 07 Sep 2026 09:12:40 +0000", repr(lookup))
+
+
+class FeedbackInjection(unittest.TestCase):
+ """A field value carrying a line break forges a field in the report.
+
+ This is not hypothetical and it is not stopped upstream. redact.py
+ URL-DECODES a redirector's destination parameter to recover it as an
+ indicator, so a message body carrying
+
+ http://r.invalid/go?next=http%3A%2F%2Fa.invalid%2Fx%0AFeedback-Type...
+
+ produces, through parse.iocs() on a real .eml, an IOC whose value is
+ "http://a.invalid/x\\nFeedback-Type: not-abuse". Emitted verbatim, the
+ abuse desk's parser reads a Feedback-Type this tool never asserted, on a
+ report that carries the reporter's identity. That is an attacker writing
+ fields into mail sent under our name.
+
+ The answer here is to PERCENT-ENCODE the control characters rather than
+ to drop the indicator or strip them. Dropping loses a real redirect
+ target; stripping silently rewrites an indicator into a different one a
+ desk would then act on. Percent-encoding is the URL's own native
+ encoding, is exactly reversible, and leaves the value visibly altered
+ rather than quietly wrong.
+ """
+
+ def _value_out(self, value: str) -> str | None:
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [{"id": "ioc-1", "type": "url", "value": value,
+ "origin": "redirect-target"}]
+ manifest["contacts"] = [
+ {"iocs": ["ioc-1"], "query": "x.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ fields = report.feedback_fields(manifest, destination)
+ for name, out in fields:
+ if name == "Reported-Uri":
+ return out
+ return None
+
+ def _assert_no_break(self, fields: list[tuple[str, str]]) -> None:
+ for name, value in fields:
+ for bad in ("\r", "\n", "
", "
", "\v", "\f",
+ "\x1c", "\x1d", "\x1e", "\x85"):
+ self.assertNotIn(bad, name)
+ self.assertNotIn(bad, value)
+
+ def test_a_newline_in_a_value_cannot_forge_a_field(self):
+ out = self._value_out("http://a.invalid/x\nFeedback-Type: not-abuse")
+ self.assertNotIn("\n", out)
+ self.assertIn("%0A", out)
+ # The forged field name must not survive as a line of its own, but
+ # the text of the indicator is still legible and reversible.
+ self.assertEqual(out,
+ "http://a.invalid/x%0AFeedback-Type: not-abuse")
+
+ def test_the_real_parse_output_that_makes_this_reachable(self):
+ """End to end from an .eml, not from a hand-written IOC.
+
+ A test that only feeds feedback_fields() a crafted string proves the
+ encoder works; it does not prove the encoder is needed. This runs
+ the actual redirector through parse.iocs() so the fixture and the
+ defence cannot drift apart.
+ """
+ from abusectl import parse
+ raw = (
+ "Received: from evil.invalid ([203.0.113.9]) by mx.example.org; "
+ "Mon, 07 Sep 2026 09:12:40 +0000\r\n"
+ "From: <phish@sender.invalid>\r\n"
+ "Subject: verify\r\n"
+ "Date: Mon, 07 Sep 2026 09:12:40 +0000\r\n"
+ "Content-Type: text/plain\r\n\r\n"
+ "http://r.invalid/go?next=http%3A%2F%2Fa.invalid%2Fx%0A"
+ "Feedback-Type%3A%20not-abuse\r\n"
+ ).encode()
+ iocs = parse.iocs(raw, trusted=["192.0.2.0/24"])
+ injected = [e for e in iocs if "\n" in e["value"]]
+ self.assertTrue(injected, "the injection vector itself has changed")
+
+ manifest = {"format": 1, "iocs": iocs, "headers": [], "auth": {}}
+ destination = {"id": "email-x", "kind": "email",
+ "target": "abuse@host.invalid",
+ "iocs": [e["id"] for e in iocs], "body": None,
+ "status": "pending"}
+ fields = report.feedback_fields(manifest, destination)
+ self._assert_no_break(fields)
+
+ def test_every_line_breaking_shape_is_neutralised(self):
+ """The adversarial sweep, not a handful of cases.
+
+ U+2028 and U+2029 are in here because Python's own email module
+ raises on them: str.splitlines() treats them as breaks, so a value
+ carrying one would make the whole document fail to assemble in
+ Task 6 rather than merely render oddly.
+ """
+ breaks = ["\n", "\r", "\r\n", "\n\r", "
", "
",
+ "\v", "\f", "\x1c", "\x1d", "\x1e", "\x85"]
+ shapes = []
+ for brk in breaks:
+ shapes += [
+ brk,
+ "http://a.invalid/x" + brk,
+ brk + "http://a.invalid/x",
+ "http://a.invalid/x" + brk + "Feedback-Type: not-abuse",
+ "http://a.invalid/" + brk * 3 + "Source: 192.0.2.1",
+ ]
+ for value in shapes:
+ with self.subTest(value=repr(value)):
+ out = self._value_out(value)
+ self.assertIsNotNone(out)
+ for bad in breaks:
+ if len(bad) == 1:
+ self.assertNotIn(bad, out)
+
+ def test_a_value_that_is_only_a_newline_still_yields_a_field(self):
+ """It must not become an empty value or vanish silently."""
+ out = self._value_out("\n")
+ self.assertEqual(out, "%0A")
+
+ def test_encoding_is_reversible_so_the_indicator_is_not_misstated(self):
+ """The property that makes encoding honest rather than a strip.
+
+ A desk, or a later submit path, must be able to recover exactly what
+ the message declared. Stripping the character would pass every
+ assertion above and hand the desk a DIFFERENT URL.
+ """
+ from urllib.parse import unquote
+ for value in ("http://a.invalid/x\nFeedback-Type: not-abuse",
+ "http://a.invalid/\r\n\r\n",
+ "http://a.invalid/x
y"):
+ with self.subTest(value=repr(value)):
+ self.assertEqual(unquote(self._value_out(value)), value)
+
+ def test_a_literal_percent_is_encoded_so_the_reversal_is_unambiguous(self):
+ """Without this, "%0A" typed by the attacker decodes to a newline.
+
+ A redacted URL legitimately contains percent signs, and an encoder
+ that leaves them alone produces text that unquote() turns into the
+ very control character the encoder existed to remove. The reversal
+ must be a true inverse or it is a second injection one step later.
+ """
+ from urllib.parse import unquote
+ value = "http://a.invalid/x?a=%0AFeedback-Type: not-abuse"
+ out = self._value_out(value)
+ self.assertNotIn("\n", out)
+ self.assertEqual(unquote(out), value)
+
+ def test_an_injected_field_name_in_a_domain_is_neutralised_too(self):
+ """Reported-Domain and Source take the same path as Reported-Uri.
+
+ The defence must not live in one branch. That is the exact shape of
+ the fourth property's three leaks: validation applied per branch
+ gets forgotten on the next branch.
+ """
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [
+ {"id": "ioc-1", "type": "domain",
+ "value": "a.invalid\nSource: 192.0.2.1",
+ "origin": "header-from"},
+ {"id": "ioc-2", "type": "ipv4",
+ "value": "203.0.113.42\nVersion: 9",
+ "origin": "received-chain"},
+ ]
+ manifest["contacts"] = [
+ {"iocs": ["ioc-1", "ioc-2"], "query": "x.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ fields = report.feedback_fields(manifest, destination)
+ self._assert_no_break(fields)
+ self.assertEqual(dict(fields)["Version"], "1")
+ lookup = dict(fields)
+ self.assertEqual(lookup["Source"], "203.0.113.42%0AVersion: 9")
+ self.assertEqual(lookup["Reported-Domain"],
+ "a.invalid%0ASource: 192.0.2.1")
+
+ def test_the_rendered_part_survives_pythons_own_header_setter(self):
+ """The end the whole defence is for: Task 6 assembles with email.
+
+ EmailMessage raises ValueError on a header value containing a break,
+ so an unencoded value does not merely render oddly, it aborts the
+ document. Asserting through the real setter is what makes this a
+ test of the outcome rather than of my own notion of a break.
+ """
+ from email.message import EmailMessage
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [
+ {"id": "ioc-1", "type": "url",
+ "value": "http://a.invalid/x\r\nFeedback-Type: not-abuse
z",
+ "origin": "redirect-target"},
+ ]
+ manifest["contacts"] = [
+ {"iocs": ["ioc-1"], "query": "x.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ part = EmailMessage()
+ for name, value in report.feedback_fields(manifest, destination):
+ part[name] = value
+ rendered = part.as_string()
+ # The property is per LINE, not per substring: the attacker's text
+ # may legitimately appear INSIDE a value, and asserting it absent
+ # would forbid reporting a URL that merely contains the words. What
+ # must not exist is a line a parser reads as a field of its own.
+ names = [line.split(":", 1)[0] for line in rendered.splitlines()
+ if line and not line[0].isspace() and ":" in line]
+ self.assertEqual(names.count("Feedback-Type"), 1)
+ self.assertEqual(
+ [n for n in names if n == "Feedback-Type"], ["Feedback-Type"])
+ for line in rendered.splitlines():
+ self.assertNotEqual(line.strip(), "Feedback-Type: not-abuse")
+
+ def test_a_field_name_is_never_taken_from_data(self):
+ """Names are literals in this module, so no input can invent one.
+
+ Pinned because the obvious "generalise it" refactor is a table
+ mapping an IOC's own type string to a field name, and a manifest is
+ a file the user edits: a type of "x: y\\nFeedback-Type" would then
+ BE a field name. The set is closed on purpose.
+ """
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [
+ {"id": "ioc-1", "type": "url\nFeedback-Type", "value": "x",
+ "origin": "body"},
+ ]
+ manifest["contacts"] = [
+ {"iocs": ["ioc-1"], "query": "x.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ fields = report.feedback_fields(manifest, destination)
+ self.assertEqual(
+ {name for name, _ in fields},
+ {"Feedback-Type", "User-Agent", "Version", "Report-Type"})
+
+ def test_a_non_string_value_does_not_crash_the_report(self):
+ """A manifest is edited by hand and JSON has numbers.
+
+ Not a security property, but a report that raises produces nothing
+ at all, and this is the one module standing between a reviewed case
+ and a sent mail.
+ """
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["iocs"] = [
+ {"id": "ioc-1", "type": "ipv4", "value": 42,
+ "origin": "received-chain"},
+ ]
+ manifest["contacts"] = [
+ {"iocs": ["ioc-1"], "query": "x.invalid",
+ "abuse": ["abuse@host.invalid"], "source": "rdap"},
+ ]
+ destination = report.email_destinations(manifest["contacts"])[0]
+ lookup = dict(report.feedback_fields(manifest, destination))
+ self.assertEqual(lookup["Source"], "42")
+
+
+class Document(unittest.TestCase):
+ def setUp(self):
+ self.destination = report.email_destinations(MANIFEST["contacts"])[0]
+ self.raw = report.build(MANIFEST, self.destination, IDENTITY)
+ self.parsed = email.message_from_string(
+ self.raw, policy=email.policy.default
+ )
+
+ def test_it_is_a_feedback_report_with_three_parts(self):
+ self.assertEqual(self.parsed.get_content_type(), "multipart/report")
+ self.assertEqual(self.parsed.get_param("report-type"),
+ "feedback-report")
+ parts = list(self.parsed.iter_parts())
+ self.assertEqual(
+ [part.get_content_type() for part in parts],
+ ["text/plain", "message/feedback-report", "text/rfc822-headers"],
+ )
+
+ def test_the_envelope_is_addressed_and_identified(self):
+ self.assertEqual(self.parsed["To"], "abuse@host.invalid")
+ self.assertIn("reporter@example.org", self.parsed["From"])
+ self.assertTrue(self.parsed["Subject"])
+
+ def test_the_source_message_is_never_attached(self):
+ self.assertNotIn("message/rfc822", self.raw)
+
+ def test_the_headers_part_carries_what_the_manifest_declared(self):
+ """The third part is the whitelisted headers, unmangled.
+
+ Asserted by PARSING the part as headers rather than by looking for
+ substrings, because what a desk does with this part is parse it.
+ """
+ headers = self._headers_of(self.parsed)
+ self.assertEqual(headers.keys(), ["From", "Subject", "Date"])
+ self.assertEqual(headers["Subject"],
+ "Your account requires verification")
+ self.assertEqual(headers["Date"], "Mon, 07 Sep 2026 09:12:40 +0000")
+ # The From is asserted twice over: on the wire text, which is what
+ # is actually published, and on the address a desk would act on.
+ # A structured header re-renders the display name's quoting on
+ # read-back, so only the raw text pins what was written.
+ self.assertIn('From: "Example Bank" <phish@sender.invalid>', self.raw)
+ self.assertEqual([a.addr_spec for a in headers["From"].addresses],
+ ["phish@sender.invalid"])
+
+ @staticmethod
+ def _headers_of(parsed):
+ """Parse the third part's body the way a desk's parser would."""
+ body = list(parsed.iter_parts())[2].get_content()
+ return email.message_from_string(body, policy=email.policy.default)
+
+
+class DocumentHeaders(unittest.TestCase):
+ """The third part: what a hand-edited manifest can and cannot publish."""
+
+ def _build(self, headers):
+ manifest = copy.deepcopy(MANIFEST)
+ manifest["headers"] = headers
+ destination = report.email_destinations(manifest["contacts"])[0]
+ return report.build(manifest, destination, IDENTITY)
+
+ def _part(self, raw, index=2):
+ parsed = email.message_from_string(raw, policy=email.policy.default)
+ return list(parsed.iter_parts())[index]
+
+ def test_a_recipient_header_in_the_manifest_is_not_published(self):
+ """A manifest is a FILE THE USER EDITS, so it can carry a "To".
+
+ parse.report_headers() would never produce one, which is exactly
+ why asserting on its output proves nothing: the property has to
+ survive a manifest nobody generated. The victim of getting this
+ wrong is the recipient, whose address reaches the abuse desk and
+ through it the attacker.
+ """
+ raw = self._build([
+ ("To", "victim@example.org"),
+ ("Cc", "other@example.org"),
+ ("Delivered-To", "victim@example.org"),
+ ("X-Original-To", "victim@example.org"),
+ ("Subject", "kept"),
+ ])
+ body = self._part(raw).get_content()
+ self.assertNotIn("victim", raw)
+ self.assertNotIn("other@example.org", raw)
+ published = email.message_from_string(body,
+ policy=email.policy.default)
+ self.assertEqual(published.keys(), ["Subject"])
+
+ def test_a_newline_in_a_value_cannot_forge_a_header(self):
+ """Subject is attacker-controlled free text and is kept deliberately.
+
+ Emitted verbatim it forges a header line in a part whose entire
+ content is read as headers. The value must survive intact and the
+ header list must not grow.
+ """
+ raw = self._build([
+ ("Subject", "lure\nFrom: forged@attacker.invalid"),
+ ])
+ published = email.message_from_string(
+ self._part(raw).get_content(), policy=email.policy.default)
+ self.assertEqual(published.keys(), ["Subject"])
+ self.assertEqual(published["Subject"],
+ "lure From: forged@attacker.invalid")
+ self.assertEqual(published["From"], None)
+
+ def test_every_break_character_is_neutralised(self):
+ """Not only LF. Python's own parsers break on more than RFC 5322 does.
+
+ One header in, one header out, for each character in turn.
+ """
+ for ch in ("\r", "\n", "\r\n", "\v", "\f", "\x1c", "\x1d", "\x1e",
+ "\x85", "
", "
"):
+ with self.subTest(ch=repr(ch)):
+ raw = self._build([("Subject", f"a{ch}From: forged@x.invalid")])
+ published = email.message_from_string(
+ self._part(raw).get_content(),
+ policy=email.policy.default)
+ self.assertEqual(published.keys(), ["Subject"])
+ self.assertEqual(published["From"], None)
+
+ def test_an_ordinary_value_is_left_legible(self):
+ """RFC 2047 is applied only when it is needed.
+
+ Encoding every header would render an ordinary Subject as
+ "=?utf-8?q?..." and cost the desk the legibility this part is for.
+ """
+ raw = self._build([("Subject", "Your account requires verification")])
+ # Asserted on the THIRD PART's own text, not on the whole document.
+ # The text part prints the same header under "Message as declared",
+ # so a whole-document substring passes even when this part is
+ # entirely encoded, which a mutation confirmed.
+ body = self._part(raw).get_content()
+ self.assertEqual(body.splitlines(),
+ ["Subject: Your account requires verification"])
+ self.assertNotIn("=?utf-8?", body)
+
+ def test_a_case_with_no_headers_omits_the_part(self):
+ """An empty third part claims the message declared no headers.
+
+ That is never true of a real message. Absent says the report
+ carries none, which is the truth for a case parsed before the
+ headers block existed.
+ """
+ for headers in ([], None):
+ with self.subTest(headers=headers):
+ raw = self._build(headers)
+ parsed = email.message_from_string(
+ raw, policy=email.policy.default)
+ self.assertEqual(
+ [p.get_content_type() for p in parsed.iter_parts()],
+ ["text/plain", "message/feedback-report"])
+ self.assertNotIn("rfc822-headers", raw)
+
+ def test_a_manifest_carrying_only_unpublishable_headers_omits_the_part(
+ self):
+ """The filter must not leave an empty part behind either."""
+ raw = self._build([("To", "victim@example.org")])
+ parsed = email.message_from_string(raw, policy=email.policy.default)
+ self.assertEqual(
+ [p.get_content_type() for p in parsed.iter_parts()],
+ ["text/plain", "message/feedback-report"])
+ self.assertNotIn("victim", raw)
+
+ def test_a_header_name_is_matched_case_insensitively(self):
+ """A header name is case-insensitive, and a hand edit will not match.
+
+ Both directions matter: a lowercased "subject" must still publish,
+ and an uppercased "TO" must still be refused.
+ """
+ raw = self._build([("subject", "lower"), ("SUBJECT", "upper"),
+ ("Subject", "mixed"), ("TO", "victim@example.org")])
+ published = email.message_from_string(
+ self._part(raw).get_content(), policy=email.policy.default)
+ # All three spellings publish. The whitelist is stored lowercase, so
+ # a case-SENSITIVE comparison would still admit "subject" and the
+ # assertion would pass while the other two vanished; a mutation
+ # found exactly that. The capitalised spellings are what pin it.
+ self.assertEqual(published.keys(), ["subject", "SUBJECT", "Subject"])
+ self.assertNotIn("victim", raw)
+
+
+class DocumentIdentity(unittest.TestCase):
+ """The From header: the one identifier disclosed deliberately."""
+
+ def _from(self, identity):
+ destination = report.email_destinations(MANIFEST["contacts"])[0]
+ raw = report.build(MANIFEST, destination, identity)
+ parsed = email.message_from_string(raw, policy=email.policy.default)
+ return parsed["From"].addresses
+
+ def test_a_comma_in_the_org_name_does_not_split_the_address(self):
+ """"Example Consulting, Ltd" is a legitimate name, and a comma is
+ the address-list separator.
+
+ Formatted into an f-string it parses back as TWO addresses, the
+ first a bogus addr-spec with no domain, and a desk replying to the
+ report replies to nobody.
+ """
+ addresses = self._from({"name": "Example Consulting, Ltd",
+ "org": "Example Consulting",
+ "email": "reporter@example.org"})
+ self.assertEqual(len(addresses), 1)
+ self.assertEqual(addresses[0].addr_spec, "reporter@example.org")
+ self.assertEqual(addresses[0].display_name, "Example Consulting, Ltd")
+
+ def test_awkward_names_still_yield_one_reachable_address(self):
+ for name in ('A "Quoted" Reporter', "Angle <brackets>", "Dänilo Ü",
+ "Back\\slash", "semi;colon", "at@sign"):
+ with self.subTest(name=name):
+ addresses = self._from({"name": name, "org": "o",
+ "email": "reporter@example.org"})
+ self.assertEqual(len(addresses), 1)
+ self.assertEqual(addresses[0].addr_spec,
+ "reporter@example.org")
+ self.assertEqual(addresses[0].display_name, name)
+
+
+if __name__ == "__main__":
+ unittest.main()