diff options
Diffstat (limited to 'tests')
| -rw-r--r-- | tests/fixtures/reportable.eml | 22 | ||||
| -rw-r--r-- | tests/test_case.py | 10 | ||||
| -rw-r--r-- | tests/test_cli.py | 25 | ||||
| -rw-r--r-- | tests/test_offline.py | 8 | ||||
| -rw-r--r-- | tests/test_parse.py | 136 | ||||
| -rw-r--r-- | tests/test_report.py | 1600 |
6 files changed, 1800 insertions, 1 deletions
diff --git a/tests/fixtures/reportable.eml b/tests/fixtures/reportable.eml new file mode 100644 index 0000000..e939059 --- /dev/null +++ b/tests/fixtures/reportable.eml @@ -0,0 +1,22 @@ +Received: from relay.example.org (relay.example.org [192.0.2.10]) + by mx.example.org with ESMTP id abc123 + for <you@example.org>; Mon, 07 Sep 2026 09:12:44 +0000 +Received: from sender.invalid (sender.invalid [203.0.113.42]) + by relay.example.org with ESMTP id def456 + for <you@example.org>; Mon, 07 Sep 2026 09:12:40 +0000 +Return-Path: <bounce@sender.invalid> +Authentication-Results: mx.example.org; spf=fail; dkim=none; dmarc=fail +Received-SPF: fail (mx.example.org: domain of sender.invalid does not designate 203.0.113.42) +From: "Example Bank" <phish@sender.invalid> +To: victim@example.org +Cc: colleague@example.org +Delivered-To: victim@example.org +X-Original-To: victim@example.org +Reply-To: "Support" <reply@sender.invalid> +Subject: Your account requires verification +Date: Mon, 07 Sep 2026 09:12:40 +0000 +Message-ID: <case-one@sender.invalid> +MIME-Version: 1.0 +Content-Type: text/plain; charset=utf-8 + +Please verify at http://login.sender.invalid/verify?id=abc123 diff --git a/tests/test_case.py b/tests/test_case.py index c6964a9..ab7eeeb 100644 --- a/tests/test_case.py +++ b/tests/test_case.py @@ -45,6 +45,16 @@ class TestCaseCreation(unittest.TestCase): manifest = json.loads((created.path / "manifest.json").read_text()) self.assertEqual(manifest["format"], case.FORMAT_VERSION) + def test_every_block_is_seeded_present_and_empty(self): + # A created-but-unparsed case must have the same SHAPE as a parsed + # one, so a later reader indexes a block rather than guarding every + # access. report will read headers and would hit a KeyError. + created = case.create(self.root, b"x") + manifest = json.loads((created.path / "manifest.json").read_text()) + for block in ("iocs", "auth", "contacts", "destinations", "headers"): + self.assertIn(block, manifest) + self.assertEqual(manifest[block], []) + def test_two_cases_do_not_collide(self): a = case.create(self.root, b"one") b = case.create(self.root, b"two") diff --git a/tests/test_cli.py b/tests/test_cli.py index 1c023c3..b8fffea 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -145,7 +145,30 @@ class TestParse(unittest.TestCase): str(FIXTURES / "simple.eml"), ) text = (pathlib.Path(out.strip()) / "manifest.json").read_text() - self.assertNotIn("example.org", text) + self.assertNotIn("you@example.org", text) + # The bare domain is still barred everywhere the IOCs live. The + # headers block is the one exception and it is a NARROW one: the + # whitelist publishes the boundary Received line and + # Authentication-Results, and both name our own receiving relay in a + # "by"/authserv-id clause. That is the user's mail host, not the + # user's identity, and a desk learns it from the report's own From + # regardless. The address itself must still be absent, which the + # assertion above and report_headers' own tests cover. + import json + + manifest = json.loads(text) + headers = manifest.pop("headers") + self.assertNotIn("example.org", json.dumps(manifest)) + # And nothing shaped like an address survives in the exception. + # Both spellings: you%40example.org is not a hypothetical, it is why + # leaky.eml exists, and docs/plans/2026-09-09-contacts.md records a + # From of phish@victim%40example.org.invalid. + blob = json.dumps(headers) + self.assertNotIn("@example.org", blob) + self.assertNotIn("you%40example.org", blob) + names = [name for name, _ in headers] + for name in ("To", "Cc", "Delivered-To", "X-Original-To"): + self.assertNotIn(name, names) def test_a_missing_config_points_at_init(self): code, _, err = self._run( diff --git a/tests/test_offline.py b/tests/test_offline.py index 19d9cee..547bd96 100644 --- a/tests/test_offline.py +++ b/tests/test_offline.py @@ -53,6 +53,14 @@ class NothingOpensASocket(unittest.TestCase): b"Subject: test\r\n\r\nbody\r\n") parse.iocs(raw, trusted=["192.0.2.0/24"]) + def test_selecting_report_headers_opens_no_socket(self): + # A new entry point into the parse path, so it is held to the same + # guarantee: choosing what to publish resolves nothing. + raw = (b"Received: from relay.example.invalid ([192.0.2.10])\r\n" + b"From: sender@example.invalid\r\n" + b"Subject: test\r\n\r\nbody\r\n") + parse.report_headers(raw, trusted=["192.0.2.0/24"]) + def test_resolving_with_an_injected_fetch_opens_no_socket(self): iocs = [{"id": "ioc-1", "type": "ipv4", "value": "198.51.100.7"}] bootstraps = { diff --git a/tests/test_parse.py b/tests/test_parse.py index 11345c7..6e3b7a4 100644 --- a/tests/test_parse.py +++ b/tests/test_parse.py @@ -290,5 +290,141 @@ class TestIocAssembly(unittest.TestCase): self.assertNotIn("you%40example.org", blob) +class ReportHeaders(unittest.TestCase): + def test_the_whitelist_keeps_what_a_desk_needs(self): + headers = parse.report_headers(load("reportable.eml"), + trusted=["192.0.2.0/24"]) + names = [name for name, _ in headers] + for wanted in ("From", "Subject", "Date", "Message-ID", "Reply-To", + "Return-Path", "Authentication-Results", "Received-SPF"): + self.assertIn(wanted, names) + + def test_recipient_headers_never_survive_the_whitelist(self): + # The first property, at the one place a report reproduces header + # text verbatim. A blacklist would have to remember each of these; + # the whitelist never names them at all. + headers = parse.report_headers(load("reportable.eml"), + trusted=["192.0.2.0/24"]) + names = [name for name, _ in headers] + blob = repr(headers) + for name in ("To", "Cc", "Delivered-To", "X-Original-To"): + self.assertNotIn(name, names) + self.assertNotIn("victim@example.org", blob) + self.assertNotIn("colleague@example.org", blob) + + def test_received_stops_at_the_boundary_hop(self): + # 192.0.2.10 is ours, so its Received line is our own infrastructure + # and must not be published; the hop below it is the one being + # reported and is kept. + headers = parse.report_headers(load("reportable.eml"), + trusted=["192.0.2.0/24"]) + received = [value for name, value in headers if name == "Received"] + self.assertEqual(len(received), 1) + self.assertIn("203.0.113.42", received[0]) + self.assertNotIn("mx.example.org with ESMTP id abc123", received[0]) + + def test_the_published_hop_carries_no_envelope_recipient(self): + # The boundary Received line is written by OUR OWN relay, and its + # optional "for <addr>" clause is the envelope recipient: the + # victim's address, verbatim, in the one header a report reproduces + # in full. Truncating the chain is not enough on its own. + headers = parse.report_headers(load("reportable.eml"), + trusted=["192.0.2.0/24"]) + received = [value for name, value in headers if name == "Received"] + self.assertNotIn("you@example.org", received[0]) + self.assertNotIn("for <", received[0]) + # The rest of the hop survives; this is a cut, not a blanking. + self.assertIn("203.0.113.42", received[0]) + + def test_every_for_clause_shape_loses_the_address(self): + # RFC 5321 4.4 puts For inside Opt-info, so With, ID, Via or a CFWS + # comment may legitimately follow it, and its ABNF is + # 1*( Path / Mailbox ) where Mailbox carries no angle brackets. + # Anchoring on "for" being immediately followed by the clause + # terminator matched only the neatest shape and let four routine + # ones through, each publishing the victim's address. + hop = "from a.invalid (a.invalid [203.0.113.5]) by mx.example.org " + shapes = ( + "for <you@example.org> (envelope-from <b@c.invalid>); Mon, 07 Sep 2026 09:12:40 +0000", + "for you@example.org; Mon, 07 Sep 2026 09:12:40 +0000", + "for <you@example.org> with ESMTP; Mon, 07 Sep 2026 09:12:40 +0000", + "id qq; Mon, 07 Sep 2026 09:12:40 +0000 (for <you@example.org>)", + "for <you@example.org>; Mon, 07 Sep 2026 09:12:40 +0000", + ) + for tail in shapes: + with self.subTest(tail=tail): + stripped = parse._strip_envelope_recipient(hop + tail) + self.assertNotIn("you@example.org", stripped) + # The hop's own evidence survives: this is a cut, not a + # blanking, and a rule that ate the line would pass the + # assertion above while destroying the report. + self.assertIn("203.0.113.5", stripped) + self.assertIn("mx.example.org", stripped) + + def test_stripping_leaves_no_doubled_space_or_stray_separator(self): + # Cosmetic in isolation, but the result is published verbatim to a + # third party, so a mangled line reads as a broken tool. + hop = ("from a.invalid (a.invalid [203.0.113.5]) by mx.example.org" + " for <you@example.org>; Mon, 07 Sep 2026 09:12:40 +0000") + stripped = parse._strip_envelope_recipient(hop) + self.assertNotIn(" ", stripped) + self.assertNotIn(" ;", stripped) + self.assertIn("mx.example.org; Mon", stripped) + + def test_a_comment_holding_only_the_clause_leaves_no_debris(self): + hop = ("from a.invalid (a.invalid [203.0.113.5]) by mx.example.org" + " id qq; Mon, 07 Sep 2026 09:12:40 +0000 (for <you@example.org>)") + stripped = parse._strip_envelope_recipient(hop) + self.assertNotIn("you@example.org", stripped) + self.assertFalse(stripped.endswith("(")) + self.assertTrue(stripped.endswith("+0000")) + + def test_the_envelope_sender_comment_survives_the_cut(self): + # envelope-from is the SENDER, which is what the report is about, so + # cutting the recipient must not take it along. + hop = ("from a.invalid (a.invalid [203.0.113.5]) by mx.example.org" + " for <you@example.org> (envelope-from <bounce@sender.invalid>);" + " Mon, 07 Sep 2026 09:12:40 +0000") + stripped = parse._strip_envelope_recipient(hop) + self.assertNotIn("you@example.org", stripped) + self.assertIn("bounce@sender.invalid", stripped) + + def test_the_whitelist_does_not_filter_attacker_free_text(self): + # A DOCUMENTED LIMIT, not a guarantee. The spec keeps Subject and the + # From display name knowing both are attacker-controlled free text, + # because they are what lets a desk recognise a campaign. An attacker + # who writes the recipient's own address into one, obfuscated or not, + # gets it published: the whitelist governs WHICH headers travel, never + # what is inside one. + # + # This is asserted so the limit is visible and deliberate. Do not + # "fix" it by filtering free text, which is the judgement-shaped + # problem AGENTS.md names as the source of every leak here. The + # sweep over real mail is what covers this class, per AGENTS.md. + raw = (b"Received: from a.invalid (a.invalid [203.0.113.5])" + b" by mx.example.org with ESMTP id X;" + b" Mon, 07 Sep 2026 09:12:40 +0000\r\n" + b"From: <phish@sender.invalid>\r\n" + b"Subject: Verify you%40example.org\r\n\r\nbody\r\n") + headers = parse.report_headers(raw, trusted=["192.0.2.0/24"]) + subject = dict(headers)["Subject"] + self.assertIn("you%40example.org", subject) + + def test_a_forged_chain_publishes_no_hop_below_the_boundary(self): + # The same job test_a_forged_chain_stops_at_the_first_untrusted_hop + # does for sending_ip(), asserted over what actually gets published: + # 198.51.100.7 is an innocent party the attacker named. + headers = parse.report_headers(load("forged-chain.eml"), + trusted=["192.0.2.0/24"]) + received = [value for name, value in headers if name == "Received"] + # Asserted in BOTH directions: dropping Received altogether would + # satisfy the "not published" half on its own, and a test that + # passes when the feature is missing protects nothing. + self.assertEqual(len(received), 1) + self.assertIn("203.0.113.99", received[0]) + self.assertTrue(all("198.51.100.7" not in value for value in received)) + self.assertTrue(all("198.51.100.8" not in value for value in received)) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_report.py b/tests/test_report.py new file mode 100644 index 0000000..6a340c9 --- /dev/null +++ b/tests/test_report.py @@ -0,0 +1,1600 @@ +import copy +import email +import email.policy +import json +import unittest + +from abusectl import report + + +class Grouping(unittest.TestCase): + def test_two_contacts_at_one_address_become_one_destination(self): + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + {"iocs": ["ioc-2"], "query": "example.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual(len(destinations), 1) + self.assertEqual(destinations[0]["target"], "abuse@host.invalid") + self.assertEqual(destinations[0]["iocs"], ["ioc-1", "ioc-2"]) + + def test_a_contact_with_two_addresses_reaches_both_desks(self): + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["a@host.invalid", "b@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual([d["target"] for d in destinations], + ["a@host.invalid", "b@host.invalid"]) + # Both desks carry the indicator, and each gets its own id: a + # destination that reached only one desk, or two rows sharing an + # id, would pass an assertion on the sorted targets alone. + self.assertEqual([d["iocs"] for d in destinations], + [["ioc-1"], ["ioc-1"]]) + self.assertEqual(len({d["id"] for d in destinations}), 2) + for destination in destinations: + self.assertEqual(destination["id"], + report.email_destination_id( + destination["target"])) + + def test_a_contact_with_no_address_creates_no_destination(self): + """The contact that resolved must still produce its destination. + + Asserted alongside one that DOES resolve, because "no destination + for this contact" is also what returning nothing at all looks + like, and that is not the behaviour being described. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "example.invalid", "abuse": [], + "source": "rdap", "error": "no abuse role published"}, + {"iocs": ["ioc-2"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual([d["target"] for d in destinations], + ["abuse@host.invalid"]) + self.assertEqual(destinations[0]["iocs"], ["ioc-2"]) + + def test_destinations_carry_stable_ids_and_pending_status(self): + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(contacts)[0] + self.assertEqual(destination["id"], + report.email_destination_id("abuse@host.invalid")) + self.assertEqual(destination["kind"], "email") + self.assertEqual(destination["status"], "pending") + + def test_an_id_is_the_literal_shape_a_reviewer_will_read(self): + """Pin the shape, since it becomes a filename in bodies/. + + Computed by hand rather than by calling the code under test, so + this fails if the derivation changes rather than following it. + """ + self.assertEqual(report.email_destination_id("abuse@host.invalid"), + "email-bc50e369") + + def test_ids_are_derived_per_destination_not_per_contact(self): + """A contact that resolved to no desk must not shift another's id. + + The obvious implementation numbers destinations by position, and a + skipped contact then either burns an id or renumbers the rest. + Both are wrong for the same reason: an id names a desk. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "example.invalid", "abuse": [], + "source": "rdap", "error": "no abuse role published"}, + {"iocs": ["ioc-2"], "query": "198.51.100.7", + "abuse": ["a@host.invalid"], "source": "rdap"}, + {"iocs": ["ioc-3"], "query": "198.51.100.8", + "abuse": ["b@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual([d["id"] for d in destinations], + [report.email_destination_id("a@host.invalid"), + report.email_destination_id("b@host.invalid")]) + self.assertEqual([d["target"] for d in destinations], + ["a@host.invalid", "b@host.invalid"]) + + def test_one_desk_listed_twice_by_one_contact_is_one_destination(self): + """A duplicate in a contact's own abuse list must not duplicate a desk. + + RDAP jCards are attacker-adjacent data: an entity can publish the + same address in two vcard rows, and one destination per ADDRESS is + the rule regardless of how many rows produced it. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid", "abuse@host.invalid"], + "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual(len(destinations), 1) + self.assertEqual(destinations[0]["iocs"], ["ioc-1"]) + + def test_one_desk_spelled_with_two_domain_cases_is_one_destination(self): + """A domain is case-insensitive, so two spellings are one desk.""" + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@Host.Invalid"], "source": "rdap"}, + {"iocs": ["ioc-2"], "query": "example.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual(len(destinations), 1) + self.assertEqual(destinations[0]["iocs"], ["ioc-1", "ioc-2"]) + self.assertEqual(destinations[0]["target"], "abuse@Host.Invalid") + + def test_the_domain_folds_under_a_local_part_that_does_not(self): + """Isolate the domain fold from the role fold. + + The version of this test that first shipped used a lowercase + local part throughout, so it exercised only the domain and passed + while a capitalised role name produced two destinations. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["J.Smith@Host.Invalid"], "source": "rdap"}, + {"iocs": ["ioc-2"], "query": "example.invalid", + "abuse": ["J.Smith@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual(len(destinations), 1) + self.assertEqual(destinations[0]["iocs"], ["ioc-1", "ioc-2"]) + + def test_two_spellings_of_one_desk_share_an_id(self): + """The id derives from the same normalised form the grouping uses. + + Otherwise the spelling RDAP happened to publish first would decide + a body's filename, and a re-run that saw the other spelling first + would look like a different desk. + """ + self.assertEqual(report.email_destination_id("abuse@Host.Invalid"), + report.email_destination_id("abuse@host.invalid")) + + def test_a_role_mailbox_folds_in_both_halves(self): + """"Abuse@Host.Invalid" and "abuse@host.invalid" are one desk. + + The case that first shipped folded the domain only, so a jCard + publishing the role name capitalised produced two destinations and + two mails to one desk. RFC 2142 mandates the role mailboxes and + requires them case-insensitive, so no host runs "Abuse@" and + "abuse@" as different desks. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["Abuse@Host.Invalid"], "source": "rdap"}, + {"iocs": ["ioc-2"], "query": "example.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual(len(destinations), 1) + self.assertEqual(destinations[0]["iocs"], ["ioc-1", "ioc-2"]) + self.assertEqual(destinations[0]["target"], "Abuse@Host.Invalid") + + def test_one_contact_publishing_a_role_mailbox_twice_folds_it(self): + """The same fold applies within one contact's own abuse list. + + rdap.abuse_addresses dedupes case-sensitively, so a jCard with two + vcard rows spelling the role differently delivers both here. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["Abuse@host.invalid", "abuse@host.invalid"], + "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual(len(destinations), 1) + self.assertEqual(destinations[0]["iocs"], ["ioc-1"]) + + def test_every_rfc2142_role_this_tool_can_meet_folds(self): + for role in ("abuse", "postmaster", "security", "noc", "hostmaster"): + with self.subTest(role=role): + self.assertEqual( + report.email_destination_id(f"{role.title()}@host.invalid"), + report.email_destination_id(f"{role}@host.invalid")) + + def test_a_personal_local_part_is_left_alone(self): + """Only the receiving host knows whether ITS local parts fold. + + A named mailbox is not a standardised role, so folding it could + silently merge two desks a host genuinely distinguishes and drop + one of them. Two mails to one desk is the lesser failure, and the + role names above are where the duplicate actually happens. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["J.Smith@host.invalid", "j.smith@host.invalid"], + "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual(sorted(d["target"] for d in destinations), + ["J.Smith@host.invalid", "j.smith@host.invalid"]) + self.assertNotEqual(destinations[0]["id"], destinations[1]["id"]) + + def test_a_destination_starts_with_no_body(self): + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + self.assertIsNone(report.email_destinations(contacts)[0]["body"]) + + def test_the_contacts_passed_in_are_not_modified(self): + """The caller's contacts are the manifest's own array. + + case.py is the only writer of a manifest, so a grouping pass that + edited what it was handed would write through it from outside. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + before = copy.deepcopy(contacts) + report.email_destinations(contacts) + self.assertEqual(contacts, before) + + def test_a_destinations_ioc_list_is_its_own(self): + """Not aliased to the contact's list it was built from. + + Holds today because the grouping starts a fresh list, but nothing + else pins it: an implementation that reused contact["iocs"] for a + single-contact destination would pass every other test here and + leave a destination and a contact sharing one list in a manifest + about to be written. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(contacts)[0] + destination["iocs"].append("ioc-2") + self.assertEqual(contacts[0]["iocs"], ["ioc-1"]) + + def test_the_same_contacts_produce_the_same_ids_twice(self): + """Ids must not depend on dict iteration luck or set ordering.""" + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["b@host.invalid", "a@host.invalid"], "source": "rdap"}, + {"iocs": ["ioc-2"], "query": "example.invalid", + "abuse": ["c@host.invalid"], "source": "rdap"}, + ] + first = report.email_destinations(contacts) + second = report.email_destinations(contacts) + self.assertEqual([(d["id"], d["target"]) for d in first], + [(d["id"], d["target"]) for d in second]) + self.assertEqual([d["target"] for d in first], + ["b@host.invalid", "a@host.invalid", + "c@host.invalid"]) + + def test_a_desks_id_survives_another_desk_appearing(self): + """An id names a DESK, not a position in this run's list. + + Task 8 writes each body to bodies/<id>.xarf and records its hash + against that id. With a positional id, re-running contacts on a + case that gained an indicator renumbers every desk after the new + one, so bodies/<id>.xarf on disk belongs to a different desk than + the manifest's entry of that id, and the edit check compares one + desk's body against another's. + """ + established = {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["b@host.invalid"], "source": "rdap"} + first = report.email_destinations([established]) + + # A later contacts run finds an indicator whose desk sorts ahead. + newcomer = {"iocs": ["ioc-2"], "query": "example.invalid", + "abuse": ["a@new.invalid"], "source": "rdap"} + second = report.email_destinations([newcomer, established]) + + by_target = {d["target"]: d["id"] for d in second} + self.assertEqual(by_target["b@host.invalid"], first[0]["id"]) + self.assertNotEqual(by_target["a@new.invalid"], first[0]["id"]) + + def test_an_ids_position_does_not_leak_into_it(self): + """The same desk alone and third in a list gets one id.""" + alone = report.email_destinations([ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["desk@host.invalid"], "source": "rdap"}, + ]) + crowded = report.email_destinations([ + {"iocs": ["ioc-2"], "query": "198.51.100.8", + "abuse": ["one@host.invalid", "two@host.invalid"], + "source": "rdap"}, + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["desk@host.invalid"], "source": "rdap"}, + ]) + self.assertEqual(crowded[2]["target"], "desk@host.invalid") + self.assertEqual(crowded[2]["id"], alone[0]["id"]) + + +class Unreportable(unittest.TestCase): + def test_an_ioc_with_no_desk_is_listed_with_its_reason(self): + contacts = [ + {"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + {"iocs": ["ioc-2", "ioc-3"], "query": "example.invalid", + "abuse": [], "source": "rdap", + "error": "no abuse role published"}, + ] + self.assertEqual( + report.unreportable(contacts), + [ + {"ioc": "ioc-2", "reason": "no abuse role published"}, + {"ioc": "ioc-3", "reason": "no abuse role published"}, + ], + ) + + def test_a_missing_reason_still_produces_an_entry(self): + contacts = [{"iocs": ["ioc-9"], "query": "x.invalid", "abuse": [], + "source": "rdap"}] + self.assertEqual( + report.unreportable(contacts), + [{"ioc": "ioc-9", "reason": "no abuse address resolved"}], + ) + + def test_an_empty_reason_does_not_read_as_no_reason(self): + """`error: ""` must not be reported as the literal empty string. + + A contact entry is written by contacts.resolve, but a manifest is + a file on disk that a user edits during review. An empty reason + renders as a blank cell in the report the user reads, which says + nothing at all; the default at least says what happened. + """ + contacts = [{"iocs": ["ioc-9"], "query": "x.invalid", "abuse": [], + "source": "rdap", "error": ""}] + self.assertEqual( + report.unreportable(contacts), + [{"ioc": "ioc-9", "reason": "no abuse address resolved"}], + ) + + def test_nothing_unreportable_is_an_empty_list_not_an_error(self): + contacts = [{"iocs": ["ioc-1"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}] + self.assertEqual(report.unreportable(contacts), []) + + def test_an_ioc_that_reached_a_desk_elsewhere_is_not_unreportable(self): + """Hosts fold, so one IOC can sit in a resolved and an unresolved + contact at once. It IS reportable, and listing it says otherwise. + + The plan's implementation listed it regardless, which puts an + indicator in both the destination list and the "no desk found" + list of one manifest. A reviewer reading the second acts on an + indicator that is already on its way to a desk, and the whole + point of the array is that it can be trusted without diffing. + """ + contacts = [ + {"iocs": ["ioc-1", "ioc-2"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + {"iocs": ["ioc-2", "ioc-3"], "query": "example.invalid", + "abuse": [], "source": "rdap", + "error": "no abuse role published"}, + ] + self.assertEqual( + report.unreportable(contacts), + [{"ioc": "ioc-3", "reason": "no abuse role published"}], + ) + + def test_one_ioc_unresolved_twice_is_listed_once(self): + """Two contacts, both unresolved, one shared indicator. + + A duplicate row is a second line in the report about one + indicator, and the reasons may differ, so which one wins has to + be decided rather than left to whichever contact came last. + First reason seen wins, matching the first-seen ordering the + destinations use. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "example.invalid", "abuse": [], + "source": "rdap", "error": "no abuse role published"}, + {"iocs": ["ioc-1"], "query": "198.51.100.7", "abuse": [], + "source": "rdap", "error": "no rdap server for this tld, " + "or no answer"}, + ] + self.assertEqual( + report.unreportable(contacts), + [{"ioc": "ioc-1", "reason": "no abuse role published"}], + ) + + def test_the_contacts_passed_in_are_not_modified(self): + contacts = [ + {"iocs": ["ioc-1"], "query": "example.invalid", "abuse": [], + "source": "rdap", "error": "no abuse role published"}, + {"iocs": ["ioc-2"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + before = copy.deepcopy(contacts) + report.unreportable(contacts) + self.assertEqual(contacts, before) + + +class MalformedAddresses(unittest.TestCase): + """An abuse "address" with no @ cannot be mailed. + + RDAP jCard data is third-party and occasionally malformed, and a + destination built from such a value carries an unsendable target with + status "pending". That is the failure mode the unreportable array + exists to prevent: the indicator appears reportable, no desk ever + receives it, and nothing in the manifest says so. + """ + + def test_a_target_with_no_at_creates_no_destination(self): + contacts = [ + {"iocs": ["ioc-1"], "query": "example.invalid", + "abuse": ["not-an-address"], "source": "rdap"}, + {"iocs": ["ioc-2"], "query": "198.51.100.7", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual([d["target"] for d in destinations], + ["abuse@host.invalid"]) + self.assertEqual(destinations[0]["iocs"], ["ioc-2"]) + + def test_an_ioc_whose_only_address_is_malformed_is_unreportable(self): + contacts = [ + {"iocs": ["ioc-1"], "query": "example.invalid", + "abuse": ["not-an-address"], "source": "rdap"}, + ] + self.assertEqual( + report.unreportable(contacts), + [{"ioc": "ioc-1", + "reason": "no usable abuse address published"}], + ) + + def test_a_usable_address_beside_a_malformed_one_still_reports(self): + """The good half of a jCard must survive the bad half. + + Discarding the contact wholesale would lose a real desk over a + neighbouring malformed row. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "example.invalid", + "abuse": ["not-an-address", "abuse@host.invalid"], + "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + self.assertEqual([d["target"] for d in destinations], + ["abuse@host.invalid"]) + self.assertEqual(report.unreportable(contacts), []) + + def test_an_addresss_own_error_is_not_overwritten_by_the_default(self): + """A contact that has both a reason and a malformed address. + + The contact's own error says more than "no usable address", so it + wins; the default is only for a contact that offered no reason. + """ + contacts = [ + {"iocs": ["ioc-1"], "query": "example.invalid", + "abuse": ["not-an-address"], "source": "rdap", + "error": "no abuse role published"}, + ] + self.assertEqual( + report.unreportable(contacts), + [{"ioc": "ioc-1", "reason": "no abuse role published"}], + ) + + def test_the_two_lists_partition_every_indicator(self): + """The invariant the pair is for: each IOC is in exactly one. + + Every other test here pins one side. This pins the relationship, + which is what a reviewer actually relies on: an indicator missing + from both is silently unreported, and one in both is reported and + also flagged as unreported. Both failures come from the two + functions disagreeing about what counts as a desk, so they are + asserted against one input that exercises every branch. + """ + contacts = [ + {"iocs": ["ioc-1", "ioc-2"], "query": "198.51.100.7", + "abuse": ["Abuse@Host.Invalid"], "source": "rdap"}, + {"iocs": ["ioc-2", "ioc-3"], "query": "example.invalid", + "abuse": [], "source": "rdap", + "error": "no abuse role published"}, + {"iocs": ["ioc-4"], "query": "other.invalid", + "abuse": ["not-an-address"], "source": "rdap"}, + {"iocs": ["ioc-5"], "query": "mixed.invalid", + "abuse": ["broken", "abuse@host.invalid"], "source": "rdap"}, + ] + destinations = report.email_destinations(contacts) + reported = {i for d in destinations for i in d["iocs"]} + flagged = {e["ioc"] for e in report.unreportable(contacts)} + every = {i for c in contacts for i in c["iocs"]} + + self.assertEqual(reported & flagged, set()) + self.assertEqual(reported | flagged, every) + self.assertEqual(reported, {"ioc-1", "ioc-2", "ioc-5"}) + self.assertEqual(flagged, {"ioc-3", "ioc-4"}) + + def test_an_empty_or_whitespace_target_is_not_a_desk(self): + for value in ("", " ", "@host.invalid", "abuse@"): + with self.subTest(value=value): + contacts = [{"iocs": ["ioc-1"], "query": "example.invalid", + "abuse": [value], "source": "rdap"}] + self.assertEqual(report.email_destinations(contacts), []) + self.assertEqual( + report.unreportable(contacts), + [{"ioc": "ioc-1", + "reason": "no usable abuse address published"}]) + + +IDENTITY = {"name": "A Reporter", "org": "Example Consulting", + "email": "reporter@example.org"} + +MANIFEST = { + "format": 1, + "case_id": "2026-09-07-aaaa", + "iocs": [ + {"id": "ioc-1", "type": "ipv4", "value": "203.0.113.42", + "origin": "received-chain", "confidence": "boundary-hop"}, + {"id": "ioc-2", "type": "url", + "value": "http://login.sender.invalid/verify?id=REDACTED", + "origin": "body"}, + ], + "auth": {"spf": "fail", "dkim": "none", "dmarc": "fail"}, + "headers": [ + ("From", '"Example Bank" <phish@sender.invalid>'), + ("Subject", "Your account requires verification"), + ("Date", "Mon, 07 Sep 2026 09:12:40 +0000"), + ], + "contacts": [ + {"iocs": ["ioc-1", "ioc-2"], "query": "203.0.113.42", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ], +} + +# 120 characters, well past the 72-column wrap, built from .invalid only. +LONG_URL = ("http://very-long-host-name.example.invalid/a/rather/deep/path/" + "segment/tree/verify?campaign=REDACTED&id=REDACTED") + + +class TextPart(unittest.TestCase): + def setUp(self): + destination = report.email_destinations(MANIFEST["contacts"])[0] + self.text = report.text_part(MANIFEST, destination, IDENTITY) + + def test_the_redaction_note_is_always_present(self): + self.assertIn("Recipient identifiers", self.text) + + def test_the_reporter_identity_appears(self): + self.assertIn("A Reporter", self.text) + self.assertIn("Example Consulting", self.text) + self.assertIn("reporter@example.org", self.text) + + def test_the_destinations_own_indicators_appear(self): + self.assertIn("203.0.113.42", self.text) + self.assertIn("http://login.sender.invalid/verify?id=REDACTED", + self.text) + + def test_an_indicator_belonging_to_another_desk_does_not_appear(self): + manifest = dict(MANIFEST) + manifest["iocs"] = MANIFEST["iocs"] + [ + {"id": "ioc-9", "type": "ipv4", "value": "192.0.2.99", + "origin": "received-chain"}, + ] + destination = report.email_destinations(MANIFEST["contacts"])[0] + text = report.text_part(manifest, destination, IDENTITY) + self.assertNotIn("192.0.2.99", text) + + def test_no_line_exceeds_seventy_two_columns(self): + for line in self.text.splitlines(): + self.assertLessEqual(len(line), 72, line) + + # --- the parts the plan got wrong ------------------------------------- + + def _with(self, **fields) -> str: + manifest = copy.deepcopy(MANIFEST) + manifest.update(fields) + destination = report.email_destinations(manifest["contacts"])[0] + return report.text_part(manifest, destination, IDENTITY) + + def test_a_long_url_is_whole_and_still_within_seventy_two_columns(self): + """A truncated URL is a WRONG indicator, not a shortened one. + + The plan wrapped three lines with a `[:72]` slice and left the + indicator list unwrapped. Both halves are the same defect: a desk + acting on a prefix acts on a resource that is not the one reported, + and a prefix reads as complete because nothing says otherwise. + + So the value must survive intact, reassemblable by a reader, and + every line must still fit. Asserting only "the URL is in the text" + would pass on a long unwrapped line, and asserting only the column + limit would pass on a truncation; the two together admit neither. + """ + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [ + {"id": "ioc-1", "type": "url", "value": LONG_URL, + "origin": "body"}, + ] + manifest["contacts"] = [ + {"iocs": ["ioc-1"], "query": "example.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + text = report.text_part(manifest, destination, IDENTITY) + + for line in text.splitlines(): + self.assertLessEqual(len(line), 72, line) + # Rejoining the continuation lines must give back the exact value. + self.assertIn(LONG_URL, report.unwrap(text)) + + def test_a_long_header_value_is_not_truncated(self): + """A header is what the message DECLARED, and a cut one misstates it. + + The subject here is attacker-controlled free text of a length no + column limit accommodates. Truncating it publishes something the + message did not say. + """ + long_subject = ("Your account requires verification before " + "the end of the working day or it will be " + "suspended permanently") + manifest = copy.deepcopy(MANIFEST) + manifest["headers"] = [("Subject", long_subject)] + destination = report.email_destinations(manifest["contacts"])[0] + text = report.text_part(manifest, destination, IDENTITY) + + for line in text.splitlines(): + self.assertLessEqual(len(line), 72, line) + self.assertIn(long_subject, report.unwrap(text)) + + def test_a_long_reporter_identity_is_not_truncated(self): + """The identity is the one thing disclosed deliberately. + + A cut address is an address nobody can reply to, which defeats the + line's only purpose. The plan sliced it at 72. + """ + identity = {"name": "A Reporter With Rather A Long Name", + "org": "Example Consulting And Partners Limited", + "email": "a.reporter@consulting.example.org"} + destination = report.email_destinations(MANIFEST["contacts"])[0] + text = report.text_part(MANIFEST, destination, identity) + + for line in text.splitlines(): + self.assertLessEqual(len(line), 72, line) + joined = report.unwrap(text) + self.assertIn("A Reporter With Rather A Long Name", joined) + self.assertIn("a.reporter@consulting.example.org", joined) + + def test_headers_survive_a_json_round_trip(self): + """case.load() returns lists, not tuples: JSON has no tuple. + + The header block is the one place a pair is destructured, so it is + the one place the round-trip shape can break the report. + """ + manifest = json.loads(json.dumps(MANIFEST)) + self.assertIsInstance(manifest["headers"][0], list) + destination = report.email_destinations(manifest["contacts"])[0] + text = report.text_part(manifest, destination, IDENTITY) + self.assertIn("Your account requires verification", text) + + def test_an_origin_reads_as_english_not_as_an_internal_token(self): + """`header-list_unsubscribe` is a parser's word, not a desk's. + + A desk deciding whether to act needs to know where an indicator was + seen. An internal token with an underscore in it reads as debug + output and makes the whole report look machine-dumped. + """ + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [ + {"id": "ioc-1", "type": "url", "value": "http://a.invalid/x", + "origin": "header-list_unsubscribe"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + text = report.text_part(manifest, destination, IDENTITY) + self.assertNotIn("header-list_unsubscribe", text) + self.assertIn("List-Unsubscribe", text) + + def test_an_unknown_origin_is_shown_rather_than_dropped(self): + """A newer parse.py may invent an origin this table does not know. + + Dropping it would silently lose the one line saying where an + indicator came from, so an unknown token is shown as-is: ugly beats + absent, and it is visible enough to get the table updated. + """ + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [ + {"id": "ioc-1", "type": "url", "value": "http://a.invalid/x", + "origin": "some-future-origin"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + text = report.text_part(manifest, destination, IDENTITY) + self.assertIn("some-future-origin", text) + + def test_a_boundary_hop_is_described_as_the_sending_ip(self): + self.assertIn("sending IP", self.text) + + def test_an_ioc_id_a_destination_names_but_the_manifest_lacks(self): + """A manifest is a file the user edits, so the two can disagree. + + The report must still be produced for the indicators that do exist + rather than raising, and must not invent a row for the missing one. + """ + destination = {"id": "email-x", "kind": "email", + "target": "abuse@host.invalid", + "iocs": ["ioc-1", "ioc-404"], "body": None, + "status": "pending"} + text = report.text_part(MANIFEST, destination, IDENTITY) + self.assertIn("203.0.113.42", text) + self.assertNotIn("ioc-404", text) + + def test_a_manifest_with_no_auth_or_headers_still_reports(self): + """Both blocks are optional and an empty one must not print a + heading with nothing under it.""" + text = self._with(auth={}, headers=[]) + self.assertIn("203.0.113.42", text) + self.assertNotIn("Message as declared", text) + self.assertNotIn("Authentication results", text) + + +class BackslashRoundTrip(unittest.TestCase): + """A value's own backslash must never be read as a wrap marker. + + The continuation marker is a trailing "\\", and a URL path may legally + end in one. Until this was fixed the two were indistinguishable, so an + attacker who read this source could append a backslash and make their + own indicator garble itself in the report an abuse desk reads. That is + an adversarial trigger on attacker-supplied text, not an edge case. + + The property asserted throughout is the only one that closes it: + unwrap(text_part(...)) contains the value EXACTLY, for every value, + wrapped or not. Asserting "the value appears" without unwrap, or + asserting only on long values, both leave the short case open, and the + short case is the one that needs no wrapping to corrupt. + """ + + def _render_ioc(self, value: str) -> str: + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [{"id": "ioc-1", "type": "url", "value": value, + "origin": "body"}] + manifest["contacts"] = [ + {"iocs": ["ioc-1"], "query": "example.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + return report.text_part(manifest, destination, IDENTITY) + + def _assert_round_trips(self, value: str) -> None: + text = self._render_ioc(value) + for line in text.splitlines(): + self.assertLessEqual(len(line), 72, line) + self.assertIn(value, report.unwrap(text)) + + def test_a_short_value_ending_in_a_backslash_survives(self): + """The case that needs no wrapping at all to corrupt. + + Nothing is wrapped here, yet unwrap() used to eat the following + line, merging the indicator with its own origin annotation and + rendering "http://a.invalid/xseen in a link in the message body". + One corrupt line where there were two, and a wrong indicator. + """ + self._assert_round_trips("http://a.invalid/x\\") + + def test_a_long_value_ending_in_a_backslash_survives(self): + self._assert_round_trips("http://a.invalid/" + "b" * 90 + "\\") + + def test_a_value_with_an_interior_backslash_survives(self): + self._assert_round_trips("http://a.invalid/x\\y/z") + + def test_a_value_ending_in_two_backslashes_survives(self): + """Whatever escaping is chosen must not have its own off-by-one. + + Doubling every backslash makes a trailing pair into four, and a + decoder that consumes them greedily or in the wrong order gives + back one backslash or three. This is the test that catches that. + """ + self._assert_round_trips("http://a.invalid/x\\\\") + + def test_adversarial_values_round_trip_exactly(self): + """A handful of shapes chosen to sit on the seams. + + The two boundary values matter most: a value that exactly fills a + line and one a single character over it are where an off-by-one in + the wrap arithmetic lives, and a backslash landing exactly on the + break column is where escaping and wrapping interact. + """ + indent = 2 + room = 72 - indent + values = [ + "http://a.invalid/x\\", + "http://a.invalid/x\\y/z", + "http://a.invalid/x\\\\", + "\\" + "a" * 40, + "a" * 40 + "\\", + "http://a.invalid/" + "b" * 90 + "\\", + "a" * room, # exactly fills the line + "a" * (room + 1), # one character over + "a" * (room - 1) + "\\", # backslash at the break + "a" * room + "\\", + "\\\\" + "c" * 80 + "\\\\", + ] + for value in values: + with self.subTest(value=value): + self._assert_round_trips(value) + + def test_a_backslash_landing_on_the_break_column_survives(self): + """The case that a passing suite still missed. + + Escaping doubles each backslash, and a break falling BETWEEN the + two halves of a pair splits the run unwrap() counts the parity of. + Both halves are then misread, a real marker reads as content, the + continuation line is orphaned and the tail of the value is silently + dropped. It needs a backslash at exactly the break column, so no + hand-written case found it; a randomised sweep failed 454 of 3538. + + Walking the backslash across every position around the boundary is + what makes this deterministic rather than luck. + """ + room = 72 - 2 # indent is two spaces for an indicator line + for offset in range(-4, 5): + position = room + offset + if position < 1: + continue + value = "a" * position + "\\" + "b" * 30 + with self.subTest(offset=offset): + self._assert_round_trips(value) + + def test_a_run_of_backslashes_across_the_break_survives(self): + """A run is where an off-by-one in the back-off hides. + + Backing off one character is correct only if the character it lands + on is the first half of a pair; a run of three or four exercises + whether the parity test looks at the run rather than at one + character. + """ + room = 72 - 2 + for length in range(1, 6): + for offset in range(-3, 4): + position = room + offset + if position < 1: + continue + value = "a" * position + "\\" * length + "b" * 20 + with self.subTest(length=length, offset=offset): + self._assert_round_trips(value) + + def test_a_value_that_is_entirely_backslashes_survives(self): + """Escaping doubles the length, so this is the worst case for both + the wrap arithmetic and the parity test at once.""" + for length in (1, 2, 3, 34, 35, 36, 70, 71): + with self.subTest(length=length): + self._assert_round_trips("\\" * length) + + def test_an_attacker_subject_cannot_corrupt_the_header_block(self): + """Subject is attacker-controlled and sits beside headers it can eat. + + This is worse than the URL case: a trailing backslash on Subject + used to swallow the following line, rendering + "Subject: Verify nowDate: Mon, 07 Sep 2026 09:12:40 +0000". The + attacker's own text destroys a DIFFERENT field's value, so the + block misstates what the message declared, which is the one thing + that block exists to report faithfully. + """ + subject = "Verify now\\" + manifest = copy.deepcopy(MANIFEST) + manifest["headers"] = [ + ("Subject", subject), + ("Date", "Mon, 07 Sep 2026 09:12:40 +0000"), + ] + destination = report.email_destinations(manifest["contacts"])[0] + text = report.text_part(manifest, destination, IDENTITY) + + for line in text.splitlines(): + self.assertLessEqual(len(line), 72, line) + joined = report.unwrap(text) + self.assertIn(f"Subject: {subject}", joined) + # The Date must survive intact rather than being absorbed. + self.assertIn("Date: Mon, 07 Sep 2026 09:12:40 +0000", joined) + + def test_a_display_name_ending_in_a_backslash_survives(self): + """The From display name is attacker-controlled too, and a sweep + has already found a spoofed one.""" + value = '"Example Bank\\" <phish@sender.invalid>' + manifest = copy.deepcopy(MANIFEST) + manifest["headers"] = [("From", value), + ("Subject", "Your account requires check")] + destination = report.email_destinations(manifest["contacts"])[0] + text = report.text_part(manifest, destination, IDENTITY) + + joined = report.unwrap(text) + self.assertIn(f"From: {value}", joined) + self.assertIn("Subject: Your account requires check", joined) + + def test_unwrap_reads_a_marker_by_parity_not_by_a_trailing_backslash(self): + """unwrap() is PUBLIC, so its input is not only our own output. + + A case manifest is a file the user edits and a desk may script + against the text part, so unwrap() must decide correctly on a line + it did not generate. Inside generated text the wrap back-off means + a marker always follows an even run, so parity and a plain + endswith() agree and neither is distinguishable by a round-trip + test. They disagree here, on a line ending in an escaped pair and + nothing else: that is content, and the following line must NOT be + absorbed into it. + """ + # "a\\" escaped is a value ending in one literal backslash, whole + # on its line. endswith() reads the second half as a marker. + self.assertEqual(report.unwrap(" a\\\\\n next"), " a\\\n next") + # An odd run IS a marker: two escaped halves plus the marker. + self.assertEqual(report.unwrap(" a\\\\\\\n next"), " a\\next") + + def test_an_identity_containing_a_backslash_survives(self): + identity = {"name": "A Reporter\\", "org": "Example Consulting", + "email": "reporter@example.org"} + destination = report.email_destinations(MANIFEST["contacts"])[0] + text = report.text_part(MANIFEST, destination, identity) + self.assertIn("A Reporter\\", report.unwrap(text)) + self.assertIn("Generated by abusectl.", report.unwrap(text)) + + +class FeedbackPart(unittest.TestCase): + def setUp(self): + destination = report.email_destinations(MANIFEST["contacts"])[0] + self.fields = report.feedback_fields(MANIFEST, destination) + self.lookup = dict(self.fields) + + def test_the_three_rfc5965_required_fields_are_present(self): + self.assertEqual(self.lookup["Feedback-Type"], "abuse") + self.assertEqual(self.lookup["Version"], "1") + self.assertTrue(self.lookup["User-Agent"].startswith("abusectl/")) + + def test_the_xarf_report_type_is_phishing(self): + self.assertEqual(self.lookup["Report-Type"], "phishing") + + def test_source_is_the_primary_indicator(self): + self.assertEqual(self.lookup["Source"], "203.0.113.42") + self.assertEqual(self.lookup["Source-IP"], "203.0.113.42") + + def test_every_url_appears_as_a_reported_uri(self): + uris = [value for name, value in self.fields if name == "Reported-Uri"] + self.assertEqual( + uris, ["http://login.sender.invalid/verify?id=REDACTED"] + ) + + def test_a_destination_with_no_ip_omits_source_ip(self): + contacts = [{"iocs": ["ioc-2"], "query": "sender.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}] + destination = report.email_destinations(contacts)[0] + lookup = dict(report.feedback_fields(MANIFEST, destination)) + self.assertNotIn("Source-IP", lookup) + self.assertEqual( + lookup["Source"], "http://login.sender.invalid/verify?id=REDACTED" + ) + + # --- the parts the plan got wrong ------------------------------------- + + def _fields_for(self, iocs: list[dict]) -> list[tuple[str, str]]: + """Render the machine part for a hand-built IOC list. + + Every IOC reaches one destination, so the field list is exactly what + those indicators produce and nothing is filtered out behind the test. + """ + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = iocs + manifest["contacts"] = [ + {"iocs": [entry["id"] for entry in iocs], "query": "x.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + return report.feedback_fields(manifest, destination) + + def test_source_ip_is_emitted_at_most_once(self): + """RFC 5965 says Source-IP appears "once maximum". + + The plan emitted one per IP. A strict parser meeting a repeated + single-occurrence field either rejects the part or keeps whichever + occurrence it saw last, so the field a repeat was meant to add is + the field that displaces the primary one. Every IP still travels, + in the text part and in Reported-Uri's sibling below. + """ + fields = self._fields_for([ + {"id": "ioc-1", "type": "ipv4", "value": "203.0.113.42", + "origin": "received-chain", "confidence": "boundary-hop"}, + {"id": "ioc-2", "type": "ipv4", "value": "203.0.113.43", + "origin": "received-chain"}, + ]) + ips = [v for n, v in fields if n == "Source-IP"] + self.assertEqual(ips, ["203.0.113.42"]) + self.assertEqual(dict(fields)["Source"], "203.0.113.42") + + def test_no_field_appears_twice_unless_the_rfc_allows_it(self): + """The invariant behind the test above, stated once for every field. + + Reported-Uri and Reported-Domain are "any number of times"; every + other field this module emits is once-maximum. Asserting only on + Source-IP would let the next repeated field ship unnoticed. + """ + fields = self._fields_for([ + {"id": "ioc-1", "type": "ipv4", "value": "203.0.113.42", + "origin": "received-chain"}, + {"id": "ioc-2", "type": "ipv6", "value": "2001:db8::1", + "origin": "received-chain"}, + {"id": "ioc-3", "type": "url", "value": "http://a.invalid/x", + "origin": "body"}, + {"id": "ioc-4", "type": "url", "value": "http://b.invalid/y", + "origin": "body"}, + {"id": "ioc-5", "type": "domain", "value": "a.invalid", + "origin": "header-from"}, + {"id": "ioc-6", "type": "domain", "value": "b.invalid", + "origin": "header-reply_to"}, + ]) + seen: dict[str, int] = {} + for name, _ in fields: + seen[name] = seen.get(name, 0) + 1 + repeatable = {"Reported-Uri", "Reported-Domain"} + for name, count in seen.items(): + if name not in repeatable: + self.assertEqual(count, 1, f"{name} appeared {count} times") + self.assertEqual(seen["Reported-Uri"], 2) + self.assertEqual(seen["Reported-Domain"], 2) + + def test_an_ipv6_indicator_fills_source_ip_too(self): + """"ipv6" is a distinct type string from parse.iocs(). + + A branch testing only for "ipv4" drops every IPv6 sender, and the + given tests use IPv4 throughout so none of them would notice. + """ + fields = dict(self._fields_for([ + {"id": "ioc-1", "type": "ipv6", "value": "2001:db8::1", + "origin": "received-chain"}, + ])) + self.assertEqual(fields["Source"], "2001:db8::1") + self.assertEqual(fields["Source-IP"], "2001:db8::1") + + def test_a_destination_with_no_typed_indicator_omits_source(self): + """An empty Source is worse than an absent one. + + "Source:" with nothing after it asserts that the thing being + reported is the empty string. A 5965 parser reading a present-but- + empty field has been told a value; reading no field it has been + told nothing, which is the truth. Only sha256 and observation + indicators reach a desk here, and both belong in the text part. + """ + fields = self._fields_for([ + {"id": "ioc-1", "type": "observation", + "value": "display-name-carries-address", + "origin": "display-name-from"}, + ]) + lookup = dict(fields) + self.assertNotIn("Source", lookup) + self.assertNotIn("Source-IP", lookup) + # The envelope is still well formed: a desk gets a valid part. + self.assertEqual(lookup["Feedback-Type"], "abuse") + self.assertEqual(lookup["Version"], "1") + + def test_a_type_this_module_does_not_place_is_not_invented_into_one(self): + """sha256 and observation have no 5965 or x-arf field. + + Neither is a Source, a Reported-Uri or a Reported-Domain, and + forcing one into the nearest-looking field would tell a desk that a + file hash is a URI. They travel in the human part, which is where a + desk reads what an attachment was. + """ + fields = self._fields_for([ + {"id": "ioc-1", "type": "ipv4", "value": "203.0.113.42", + "origin": "received-chain"}, + {"id": "ioc-2", "type": "sha256", "value": "a" * 64, + "origin": "attachment", "filename": "invoice.zip"}, + {"id": "ioc-3", "type": "observation", + "value": "display-name-carries-address", + "origin": "display-name-from"}, + ]) + blob = repr(fields) + self.assertNotIn("a" * 64, blob) + self.assertNotIn("display-name-carries-address", blob) + self.assertEqual(dict(fields)["Source"], "203.0.113.42") + + def test_an_ioc_id_the_manifest_lacks_is_skipped_not_raised(self): + """A manifest is a file the user edits, so the two can disagree. + + text_part() already tolerates this; the machine part indexed with + destination["iocs"] straight into a dict would raise instead, and + the two parts of one document must not disagree about whether the + case can be reported at all. + """ + destination = {"id": "email-x", "kind": "email", + "target": "abuse@host.invalid", + "iocs": ["ioc-1", "ioc-404"], "body": None, + "status": "pending"} + lookup = dict(report.feedback_fields(MANIFEST, destination)) + self.assertEqual(lookup["Source"], "203.0.113.42") + self.assertNotIn("ioc-404", repr(lookup)) + + def test_a_destination_with_no_iocs_key_still_renders(self): + """destination.get("iocs"), not destination["iocs"].""" + destination = {"id": "email-x", "kind": "email", + "target": "abuse@host.invalid", "body": None, + "status": "pending"} + lookup = dict(report.feedback_fields(MANIFEST, destination)) + self.assertEqual(lookup["Feedback-Type"], "abuse") + + def test_arrival_date_is_not_taken_from_the_senders_date_header(self): + """RFC 5965: Arrival-Date is when the generating ADMD's MTA received + the message. The Date header is when the SENDER CLAIMS it was sent. + + The plan copied Date into Arrival-Date. On a phishing message that + header is attacker-controlled free text, so the report would assert + as our own observation a timestamp the attacker chose, and a desk + correlating it against their own logs would look in the wrong place + or find nothing and discount the report. + + The honest source is the boundary Received hop's own timestamp, + which parse.report_headers() already publishes. Parsing one is a + date parser this task does not need, so the field is OMITTED: 5965 + makes it optional, and an absent optional field misstates nothing. + """ + lookup = dict(self.fields) + self.assertNotIn("Arrival-Date", lookup) + self.assertNotIn("Mon, 07 Sep 2026 09:12:40 +0000", repr(lookup)) + + +class FeedbackInjection(unittest.TestCase): + """A field value carrying a line break forges a field in the report. + + This is not hypothetical and it is not stopped upstream. redact.py + URL-DECODES a redirector's destination parameter to recover it as an + indicator, so a message body carrying + + http://r.invalid/go?next=http%3A%2F%2Fa.invalid%2Fx%0AFeedback-Type... + + produces, through parse.iocs() on a real .eml, an IOC whose value is + "http://a.invalid/x\\nFeedback-Type: not-abuse". Emitted verbatim, the + abuse desk's parser reads a Feedback-Type this tool never asserted, on a + report that carries the reporter's identity. That is an attacker writing + fields into mail sent under our name. + + The answer here is to PERCENT-ENCODE the control characters rather than + to drop the indicator or strip them. Dropping loses a real redirect + target; stripping silently rewrites an indicator into a different one a + desk would then act on. Percent-encoding is the URL's own native + encoding, is exactly reversible, and leaves the value visibly altered + rather than quietly wrong. + """ + + def _value_out(self, value: str) -> str | None: + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [{"id": "ioc-1", "type": "url", "value": value, + "origin": "redirect-target"}] + manifest["contacts"] = [ + {"iocs": ["ioc-1"], "query": "x.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + fields = report.feedback_fields(manifest, destination) + for name, out in fields: + if name == "Reported-Uri": + return out + return None + + def _assert_no_break(self, fields: list[tuple[str, str]]) -> None: + for name, value in fields: + for bad in ("\r", "\n", "
", "
", "\v", "\f", + "\x1c", "\x1d", "\x1e", "\x85"): + self.assertNotIn(bad, name) + self.assertNotIn(bad, value) + + def test_a_newline_in_a_value_cannot_forge_a_field(self): + out = self._value_out("http://a.invalid/x\nFeedback-Type: not-abuse") + self.assertNotIn("\n", out) + self.assertIn("%0A", out) + # The forged field name must not survive as a line of its own, but + # the text of the indicator is still legible and reversible. + self.assertEqual(out, + "http://a.invalid/x%0AFeedback-Type: not-abuse") + + def test_the_real_parse_output_that_makes_this_reachable(self): + """End to end from an .eml, not from a hand-written IOC. + + A test that only feeds feedback_fields() a crafted string proves the + encoder works; it does not prove the encoder is needed. This runs + the actual redirector through parse.iocs() so the fixture and the + defence cannot drift apart. + """ + from abusectl import parse + raw = ( + "Received: from evil.invalid ([203.0.113.9]) by mx.example.org; " + "Mon, 07 Sep 2026 09:12:40 +0000\r\n" + "From: <phish@sender.invalid>\r\n" + "Subject: verify\r\n" + "Date: Mon, 07 Sep 2026 09:12:40 +0000\r\n" + "Content-Type: text/plain\r\n\r\n" + "http://r.invalid/go?next=http%3A%2F%2Fa.invalid%2Fx%0A" + "Feedback-Type%3A%20not-abuse\r\n" + ).encode() + iocs = parse.iocs(raw, trusted=["192.0.2.0/24"]) + injected = [e for e in iocs if "\n" in e["value"]] + self.assertTrue(injected, "the injection vector itself has changed") + + manifest = {"format": 1, "iocs": iocs, "headers": [], "auth": {}} + destination = {"id": "email-x", "kind": "email", + "target": "abuse@host.invalid", + "iocs": [e["id"] for e in iocs], "body": None, + "status": "pending"} + fields = report.feedback_fields(manifest, destination) + self._assert_no_break(fields) + + def test_every_line_breaking_shape_is_neutralised(self): + """The adversarial sweep, not a handful of cases. + + U+2028 and U+2029 are in here because Python's own email module + raises on them: str.splitlines() treats them as breaks, so a value + carrying one would make the whole document fail to assemble in + Task 6 rather than merely render oddly. + """ + breaks = ["\n", "\r", "\r\n", "\n\r", "
", "
", + "\v", "\f", "\x1c", "\x1d", "\x1e", "\x85"] + shapes = [] + for brk in breaks: + shapes += [ + brk, + "http://a.invalid/x" + brk, + brk + "http://a.invalid/x", + "http://a.invalid/x" + brk + "Feedback-Type: not-abuse", + "http://a.invalid/" + brk * 3 + "Source: 192.0.2.1", + ] + for value in shapes: + with self.subTest(value=repr(value)): + out = self._value_out(value) + self.assertIsNotNone(out) + for bad in breaks: + if len(bad) == 1: + self.assertNotIn(bad, out) + + def test_a_value_that_is_only_a_newline_still_yields_a_field(self): + """It must not become an empty value or vanish silently.""" + out = self._value_out("\n") + self.assertEqual(out, "%0A") + + def test_encoding_is_reversible_so_the_indicator_is_not_misstated(self): + """The property that makes encoding honest rather than a strip. + + A desk, or a later submit path, must be able to recover exactly what + the message declared. Stripping the character would pass every + assertion above and hand the desk a DIFFERENT URL. + """ + from urllib.parse import unquote + for value in ("http://a.invalid/x\nFeedback-Type: not-abuse", + "http://a.invalid/\r\n\r\n", + "http://a.invalid/x
y"): + with self.subTest(value=repr(value)): + self.assertEqual(unquote(self._value_out(value)), value) + + def test_a_literal_percent_is_encoded_so_the_reversal_is_unambiguous(self): + """Without this, "%0A" typed by the attacker decodes to a newline. + + A redacted URL legitimately contains percent signs, and an encoder + that leaves them alone produces text that unquote() turns into the + very control character the encoder existed to remove. The reversal + must be a true inverse or it is a second injection one step later. + """ + from urllib.parse import unquote + value = "http://a.invalid/x?a=%0AFeedback-Type: not-abuse" + out = self._value_out(value) + self.assertNotIn("\n", out) + self.assertEqual(unquote(out), value) + + def test_an_injected_field_name_in_a_domain_is_neutralised_too(self): + """Reported-Domain and Source take the same path as Reported-Uri. + + The defence must not live in one branch. That is the exact shape of + the fourth property's three leaks: validation applied per branch + gets forgotten on the next branch. + """ + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [ + {"id": "ioc-1", "type": "domain", + "value": "a.invalid\nSource: 192.0.2.1", + "origin": "header-from"}, + {"id": "ioc-2", "type": "ipv4", + "value": "203.0.113.42\nVersion: 9", + "origin": "received-chain"}, + ] + manifest["contacts"] = [ + {"iocs": ["ioc-1", "ioc-2"], "query": "x.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + fields = report.feedback_fields(manifest, destination) + self._assert_no_break(fields) + self.assertEqual(dict(fields)["Version"], "1") + lookup = dict(fields) + self.assertEqual(lookup["Source"], "203.0.113.42%0AVersion: 9") + self.assertEqual(lookup["Reported-Domain"], + "a.invalid%0ASource: 192.0.2.1") + + def test_the_rendered_part_survives_pythons_own_header_setter(self): + """The end the whole defence is for: Task 6 assembles with email. + + EmailMessage raises ValueError on a header value containing a break, + so an unencoded value does not merely render oddly, it aborts the + document. Asserting through the real setter is what makes this a + test of the outcome rather than of my own notion of a break. + """ + from email.message import EmailMessage + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [ + {"id": "ioc-1", "type": "url", + "value": "http://a.invalid/x\r\nFeedback-Type: not-abuse
z", + "origin": "redirect-target"}, + ] + manifest["contacts"] = [ + {"iocs": ["ioc-1"], "query": "x.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + part = EmailMessage() + for name, value in report.feedback_fields(manifest, destination): + part[name] = value + rendered = part.as_string() + # The property is per LINE, not per substring: the attacker's text + # may legitimately appear INSIDE a value, and asserting it absent + # would forbid reporting a URL that merely contains the words. What + # must not exist is a line a parser reads as a field of its own. + names = [line.split(":", 1)[0] for line in rendered.splitlines() + if line and not line[0].isspace() and ":" in line] + self.assertEqual(names.count("Feedback-Type"), 1) + self.assertEqual( + [n for n in names if n == "Feedback-Type"], ["Feedback-Type"]) + for line in rendered.splitlines(): + self.assertNotEqual(line.strip(), "Feedback-Type: not-abuse") + + def test_a_field_name_is_never_taken_from_data(self): + """Names are literals in this module, so no input can invent one. + + Pinned because the obvious "generalise it" refactor is a table + mapping an IOC's own type string to a field name, and a manifest is + a file the user edits: a type of "x: y\\nFeedback-Type" would then + BE a field name. The set is closed on purpose. + """ + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [ + {"id": "ioc-1", "type": "url\nFeedback-Type", "value": "x", + "origin": "body"}, + ] + manifest["contacts"] = [ + {"iocs": ["ioc-1"], "query": "x.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + fields = report.feedback_fields(manifest, destination) + self.assertEqual( + {name for name, _ in fields}, + {"Feedback-Type", "User-Agent", "Version", "Report-Type"}) + + def test_a_non_string_value_does_not_crash_the_report(self): + """A manifest is edited by hand and JSON has numbers. + + Not a security property, but a report that raises produces nothing + at all, and this is the one module standing between a reviewed case + and a sent mail. + """ + manifest = copy.deepcopy(MANIFEST) + manifest["iocs"] = [ + {"id": "ioc-1", "type": "ipv4", "value": 42, + "origin": "received-chain"}, + ] + manifest["contacts"] = [ + {"iocs": ["ioc-1"], "query": "x.invalid", + "abuse": ["abuse@host.invalid"], "source": "rdap"}, + ] + destination = report.email_destinations(manifest["contacts"])[0] + lookup = dict(report.feedback_fields(manifest, destination)) + self.assertEqual(lookup["Source"], "42") + + +class Document(unittest.TestCase): + def setUp(self): + self.destination = report.email_destinations(MANIFEST["contacts"])[0] + self.raw = report.build(MANIFEST, self.destination, IDENTITY) + self.parsed = email.message_from_string( + self.raw, policy=email.policy.default + ) + + def test_it_is_a_feedback_report_with_three_parts(self): + self.assertEqual(self.parsed.get_content_type(), "multipart/report") + self.assertEqual(self.parsed.get_param("report-type"), + "feedback-report") + parts = list(self.parsed.iter_parts()) + self.assertEqual( + [part.get_content_type() for part in parts], + ["text/plain", "message/feedback-report", "text/rfc822-headers"], + ) + + def test_the_envelope_is_addressed_and_identified(self): + self.assertEqual(self.parsed["To"], "abuse@host.invalid") + self.assertIn("reporter@example.org", self.parsed["From"]) + self.assertTrue(self.parsed["Subject"]) + + def test_the_source_message_is_never_attached(self): + self.assertNotIn("message/rfc822", self.raw) + + def test_the_headers_part_carries_what_the_manifest_declared(self): + """The third part is the whitelisted headers, unmangled. + + Asserted by PARSING the part as headers rather than by looking for + substrings, because what a desk does with this part is parse it. + """ + headers = self._headers_of(self.parsed) + self.assertEqual(headers.keys(), ["From", "Subject", "Date"]) + self.assertEqual(headers["Subject"], + "Your account requires verification") + self.assertEqual(headers["Date"], "Mon, 07 Sep 2026 09:12:40 +0000") + # The From is asserted twice over: on the wire text, which is what + # is actually published, and on the address a desk would act on. + # A structured header re-renders the display name's quoting on + # read-back, so only the raw text pins what was written. + self.assertIn('From: "Example Bank" <phish@sender.invalid>', self.raw) + self.assertEqual([a.addr_spec for a in headers["From"].addresses], + ["phish@sender.invalid"]) + + @staticmethod + def _headers_of(parsed): + """Parse the third part's body the way a desk's parser would.""" + body = list(parsed.iter_parts())[2].get_content() + return email.message_from_string(body, policy=email.policy.default) + + +class DocumentHeaders(unittest.TestCase): + """The third part: what a hand-edited manifest can and cannot publish.""" + + def _build(self, headers): + manifest = copy.deepcopy(MANIFEST) + manifest["headers"] = headers + destination = report.email_destinations(manifest["contacts"])[0] + return report.build(manifest, destination, IDENTITY) + + def _part(self, raw, index=2): + parsed = email.message_from_string(raw, policy=email.policy.default) + return list(parsed.iter_parts())[index] + + def test_a_recipient_header_in_the_manifest_is_not_published(self): + """A manifest is a FILE THE USER EDITS, so it can carry a "To". + + parse.report_headers() would never produce one, which is exactly + why asserting on its output proves nothing: the property has to + survive a manifest nobody generated. The victim of getting this + wrong is the recipient, whose address reaches the abuse desk and + through it the attacker. + """ + raw = self._build([ + ("To", "victim@example.org"), + ("Cc", "other@example.org"), + ("Delivered-To", "victim@example.org"), + ("X-Original-To", "victim@example.org"), + ("Subject", "kept"), + ]) + body = self._part(raw).get_content() + self.assertNotIn("victim", raw) + self.assertNotIn("other@example.org", raw) + published = email.message_from_string(body, + policy=email.policy.default) + self.assertEqual(published.keys(), ["Subject"]) + + def test_a_newline_in_a_value_cannot_forge_a_header(self): + """Subject is attacker-controlled free text and is kept deliberately. + + Emitted verbatim it forges a header line in a part whose entire + content is read as headers. The value must survive intact and the + header list must not grow. + """ + raw = self._build([ + ("Subject", "lure\nFrom: forged@attacker.invalid"), + ]) + published = email.message_from_string( + self._part(raw).get_content(), policy=email.policy.default) + self.assertEqual(published.keys(), ["Subject"]) + self.assertEqual(published["Subject"], + "lure From: forged@attacker.invalid") + self.assertEqual(published["From"], None) + + def test_every_break_character_is_neutralised(self): + """Not only LF. Python's own parsers break on more than RFC 5322 does. + + One header in, one header out, for each character in turn. + """ + for ch in ("\r", "\n", "\r\n", "\v", "\f", "\x1c", "\x1d", "\x1e", + "\x85", "
", "
"): + with self.subTest(ch=repr(ch)): + raw = self._build([("Subject", f"a{ch}From: forged@x.invalid")]) + published = email.message_from_string( + self._part(raw).get_content(), + policy=email.policy.default) + self.assertEqual(published.keys(), ["Subject"]) + self.assertEqual(published["From"], None) + + def test_an_ordinary_value_is_left_legible(self): + """RFC 2047 is applied only when it is needed. + + Encoding every header would render an ordinary Subject as + "=?utf-8?q?..." and cost the desk the legibility this part is for. + """ + raw = self._build([("Subject", "Your account requires verification")]) + # Asserted on the THIRD PART's own text, not on the whole document. + # The text part prints the same header under "Message as declared", + # so a whole-document substring passes even when this part is + # entirely encoded, which a mutation confirmed. + body = self._part(raw).get_content() + self.assertEqual(body.splitlines(), + ["Subject: Your account requires verification"]) + self.assertNotIn("=?utf-8?", body) + + def test_a_case_with_no_headers_omits_the_part(self): + """An empty third part claims the message declared no headers. + + That is never true of a real message. Absent says the report + carries none, which is the truth for a case parsed before the + headers block existed. + """ + for headers in ([], None): + with self.subTest(headers=headers): + raw = self._build(headers) + parsed = email.message_from_string( + raw, policy=email.policy.default) + self.assertEqual( + [p.get_content_type() for p in parsed.iter_parts()], + ["text/plain", "message/feedback-report"]) + self.assertNotIn("rfc822-headers", raw) + + def test_a_manifest_carrying_only_unpublishable_headers_omits_the_part( + self): + """The filter must not leave an empty part behind either.""" + raw = self._build([("To", "victim@example.org")]) + parsed = email.message_from_string(raw, policy=email.policy.default) + self.assertEqual( + [p.get_content_type() for p in parsed.iter_parts()], + ["text/plain", "message/feedback-report"]) + self.assertNotIn("victim", raw) + + def test_a_header_name_is_matched_case_insensitively(self): + """A header name is case-insensitive, and a hand edit will not match. + + Both directions matter: a lowercased "subject" must still publish, + and an uppercased "TO" must still be refused. + """ + raw = self._build([("subject", "lower"), ("SUBJECT", "upper"), + ("Subject", "mixed"), ("TO", "victim@example.org")]) + published = email.message_from_string( + self._part(raw).get_content(), policy=email.policy.default) + # All three spellings publish. The whitelist is stored lowercase, so + # a case-SENSITIVE comparison would still admit "subject" and the + # assertion would pass while the other two vanished; a mutation + # found exactly that. The capitalised spellings are what pin it. + self.assertEqual(published.keys(), ["subject", "SUBJECT", "Subject"]) + self.assertNotIn("victim", raw) + + +class DocumentIdentity(unittest.TestCase): + """The From header: the one identifier disclosed deliberately.""" + + def _from(self, identity): + destination = report.email_destinations(MANIFEST["contacts"])[0] + raw = report.build(MANIFEST, destination, identity) + parsed = email.message_from_string(raw, policy=email.policy.default) + return parsed["From"].addresses + + def test_a_comma_in_the_org_name_does_not_split_the_address(self): + """"Example Consulting, Ltd" is a legitimate name, and a comma is + the address-list separator. + + Formatted into an f-string it parses back as TWO addresses, the + first a bogus addr-spec with no domain, and a desk replying to the + report replies to nobody. + """ + addresses = self._from({"name": "Example Consulting, Ltd", + "org": "Example Consulting", + "email": "reporter@example.org"}) + self.assertEqual(len(addresses), 1) + self.assertEqual(addresses[0].addr_spec, "reporter@example.org") + self.assertEqual(addresses[0].display_name, "Example Consulting, Ltd") + + def test_awkward_names_still_yield_one_reachable_address(self): + for name in ('A "Quoted" Reporter', "Angle <brackets>", "Dänilo Ü", + "Back\\slash", "semi;colon", "at@sign"): + with self.subTest(name=name): + addresses = self._from({"name": name, "org": "o", + "email": "reporter@example.org"}) + self.assertEqual(len(addresses), 1) + self.assertEqual(addresses[0].addr_spec, + "reporter@example.org") + self.assertEqual(addresses[0].display_name, name) + + +if __name__ == "__main__": + unittest.main() |
