mirror of
https://github.com/domainaware/parsedmarc.git
synced 2026-07-27 02:44:57 +00:00
Revert overanalyzed changes
This commit is contained in:
@@ -34,17 +34,6 @@
|
||||
`policy_strings` / `mx_host_patterns` from an earlier policy were reused for
|
||||
a later policy that did not define them, because the row template dict was
|
||||
built once per report instead of once per policy.
|
||||
- **The `xml_schema` field of a parsed aggregate report can no longer be
|
||||
empty.** parsedmarc reports the schema revision of an aggregate report in a
|
||||
synthesized `xml_schema` field (surfaced in the JSON/CSV output for every
|
||||
record), defaulting to `"draft"` when the report omits the optional
|
||||
`<version>` element (RFC 7489 Appendix C). That default did not cover an
|
||||
empty `<version/>` or a whitespace-only value, which produced
|
||||
`xml_schema=None` — an empty string in the flat JSON/CSV output — so
|
||||
consumers that identify aggregate records by a non-empty `xml_schema`
|
||||
(such as the Google SecOps parser's report-type detection) dropped those
|
||||
records. The `"draft"` fallback now applies unless `<version>` contains
|
||||
non-empty text.
|
||||
|
||||
## 10.2.1
|
||||
|
||||
|
||||
@@ -49,7 +49,7 @@ detects them by a field unique to each and maps them as follows:
|
||||
|
||||
| parsedmarc report | Detected by | UDM `metadata.event_type` |
|
||||
| --- | --- | --- |
|
||||
| DMARC aggregate | `xml_schema` (the serialized marker of the `<feedback>` XML root) | `EMAIL_TRANSACTION` |
|
||||
| DMARC aggregate | `xml_schema` | `EMAIL_TRANSACTION` |
|
||||
| DMARC failure | `feedback_type` or `arrival_date_utc` | `EMAIL_TRANSACTION` |
|
||||
| SMTP TLS (RFC 8460) | `policy_type` or `result_type` | `GENERIC_EVENT` |
|
||||
|
||||
@@ -59,14 +59,6 @@ field (parsedmarc emits no `feedback_type` key for them, but always computes
|
||||
older than this parser lack `policy_type` (every RFC 8460 failure detail has a
|
||||
`result_type`).
|
||||
|
||||
`xml_schema` needs no fallback because it is not something the report
|
||||
supplies. What actually identifies an aggregate report is its `<feedback>`
|
||||
XML root element — but the raw XML never reaches SecOps: parsedmarc parses
|
||||
it and only the flattened JSON rows go over syslog. `xml_schema` is the
|
||||
field parsedmarc synthesizes on every row that came from a `<feedback>`
|
||||
document ("draft" whenever `<version>` is missing, empty, or whitespace), so
|
||||
testing it is one-to-one equivalent to testing for the `<feedback>` root.
|
||||
|
||||
`EMAIL_TRANSACTION` and `GENERIC_EVENT` are both valid `metadata.event_type`
|
||||
values. Note that **`GENERIC_EVENT` events only appear in raw-log and UDM
|
||||
search**, not in the curated SecOps views — that is the documented behaviour for
|
||||
|
||||
@@ -241,12 +241,9 @@ filter {
|
||||
# policy_type alone is not enough for streams from parsedmarc versions
|
||||
# that omitted policy_domain/policy_type on SMTP TLS failure-detail rows;
|
||||
# result_type is required in every RFC 8460 failure detail.
|
||||
# xml_schema is aggregate-only and guaranteed non-empty: the <feedback>
|
||||
# XML root element is what identifies an aggregate report, and parsedmarc
|
||||
# falls back to "draft" whenever <version> is missing, empty, or
|
||||
# whitespace (parsedmarc/__init__.py), so every JSON line serialized
|
||||
# from an aggregate report carries a usable value. It is preferred
|
||||
# over: header_from (can be empty when a
|
||||
# xml_schema is aggregate-only and parsedmarc defaults it to "draft" when
|
||||
# the report omits <version> (parsedmarc/__init__.py), so it survives a
|
||||
# missing version. It is preferred over: header_from (can be empty when a
|
||||
# record carries no identifiers), adkim (a defaulted policy field), domain
|
||||
# (a generic name), and dmarc_aligned (a boolean that only becomes testable
|
||||
# after the convert in step 1b -- detection should not depend on that).
|
||||
|
||||
+1
-12
@@ -848,20 +848,9 @@ def parse_aggregate_report_xml(
|
||||
report_metadata["email"],
|
||||
)
|
||||
report_metadata["email"] = unwrapped
|
||||
# The <feedback> root element (anchored above) is what identifies an
|
||||
# aggregate report; <version> is only metadata about the schema
|
||||
# revision. Reports in the wild omit <version> entirely or send an
|
||||
# empty <version/> (which xmltodict parses as None) or one carrying
|
||||
# attributes (a dict). Only a non-empty text value may override the
|
||||
# "draft" default: xml_schema must never be empty, because consumers
|
||||
# of the flattened JSON/CSV output (e.g. the Google SecOps parser's
|
||||
# report-type detection) rely on it being non-empty on every line
|
||||
# serialized from an aggregate report.
|
||||
schema = "draft"
|
||||
if "version" in report:
|
||||
version = _text(report["version"])
|
||||
if isinstance(version, str) and version.strip():
|
||||
schema = version.strip()
|
||||
schema = report["version"]
|
||||
new_report: dict[str, Any] = {
|
||||
"xml_schema": schema,
|
||||
"xml_namespace": xml_namespace,
|
||||
|
||||
@@ -1497,73 +1497,6 @@ class Test(unittest.TestCase):
|
||||
self.assertIsNone(pp["testing"])
|
||||
self.assertIsNone(pp["discovery_method"])
|
||||
|
||||
def testEmptyVersionElementFallsBackToDraftSchema(self):
|
||||
"""An empty <version/> must not blank out xml_schema.
|
||||
|
||||
The <feedback> root element is what identifies an aggregate report;
|
||||
<version> is optional metadata (RFC 7489 Appendix C) that reporters
|
||||
omit or leave empty in the wild. xmltodict parses an empty
|
||||
<version/> as None, which previously became xml_schema=None and an
|
||||
empty string in the flat rows — breaking downstream consumers that
|
||||
detect aggregate rows by a non-empty xml_schema (the Google SecOps
|
||||
parser's report-type cascade).
|
||||
"""
|
||||
xml = """<?xml version="1.0"?>
|
||||
<feedback>
|
||||
<version/>
|
||||
<report_metadata>
|
||||
<org_name>TestOrg</org_name>
|
||||
<email>test@example.com</email>
|
||||
<report_id>test-empty-version</report_id>
|
||||
<date_range><begin>1704067200</begin><end>1704153599</end></date_range>
|
||||
</report_metadata>
|
||||
<policy_published>
|
||||
<domain>example.com</domain>
|
||||
<p>none</p>
|
||||
</policy_published>
|
||||
<record>
|
||||
<row>
|
||||
<source_ip>192.0.2.1</source_ip>
|
||||
<count>1</count>
|
||||
<policy_evaluated><disposition>none</disposition><dkim>pass</dkim><spf>pass</spf></policy_evaluated>
|
||||
</row>
|
||||
<identifiers><header_from>example.com</header_from></identifiers>
|
||||
<auth_results><spf><domain>example.com</domain><result>pass</result></spf></auth_results>
|
||||
</record>
|
||||
</feedback>"""
|
||||
report = parsedmarc.parse_aggregate_report_xml(xml, offline=True)
|
||||
self.assertEqual(report["xml_schema"], "draft")
|
||||
rows = parsedmarc.parsed_aggregate_reports_to_csv_rows(report)
|
||||
self.assertEqual(rows[0]["xml_schema"], "draft")
|
||||
|
||||
def testWhitespaceVersionElementFallsBackToDraftSchema(self):
|
||||
"""A whitespace-only <version> value also falls back to 'draft'."""
|
||||
xml = """<?xml version="1.0"?>
|
||||
<feedback>
|
||||
<version> </version>
|
||||
<report_metadata>
|
||||
<org_name>TestOrg</org_name>
|
||||
<email>test@example.com</email>
|
||||
<report_id>test-ws-version</report_id>
|
||||
<date_range><begin>1704067200</begin><end>1704153599</end></date_range>
|
||||
</report_metadata>
|
||||
<policy_published>
|
||||
<domain>example.com</domain>
|
||||
<p>none</p>
|
||||
</policy_published>
|
||||
<record>
|
||||
<row>
|
||||
<source_ip>192.0.2.1</source_ip>
|
||||
<count>1</count>
|
||||
<policy_evaluated><disposition>none</disposition><dkim>pass</dkim><spf>pass</spf></policy_evaluated>
|
||||
</row>
|
||||
<identifiers><header_from>example.com</header_from></identifiers>
|
||||
<auth_results><spf><domain>example.com</domain><result>pass</result></spf></auth_results>
|
||||
</record>
|
||||
</feedback>"""
|
||||
report = parsedmarc.parse_aggregate_report_xml(xml, offline=True)
|
||||
self.assertEqual(report["xml_schema"], "draft")
|
||||
|
||||
def testMagicXmlTagDetection(self):
|
||||
"""XML without declaration (starting with '<') is extracted"""
|
||||
xml_no_decl = b"<feedback><report_metadata><org_name>T</org_name><email>a@b.com</email><report_id>r1</report_id><date_range><begin>1704067200</begin><end>1704153599</end></date_range></report_metadata><policy_published><domain>example.com</domain><p>none</p></policy_published><record><row><source_ip>192.0.2.1</source_ip><count>1</count><policy_evaluated><disposition>none</disposition><dkim>pass</dkim><spf>pass</spf></policy_evaluated></row><identifiers><header_from>example.com</header_from></identifiers><auth_results><spf><domain>example.com</domain><result>pass</result></spf></auth_results></record></feedback>"
|
||||
|
||||
Reference in New Issue
Block a user