pymilter-suspicious-from/main.py

import sys
import logging

import Milter

from email.header import decode_header
from email.utils import getaddresses

import re

import config

# Basic logger that also logs to stdout
# TODO: Improve this a lot.
logger = logging.getLogger(__name__)
logger.setLevel(config.log_level)

handler = logging.StreamHandler(sys.stdout)
handler.setLevel(config.log_level)
formatter = logging.Formatter('%(asctime)s - %(levelname)s - %(message)s')
handler.setFormatter(formatter)

logger.addHandler(handler)

# Rough regex to fetch domain values from address-like text
# Not matching the part in front of @ because greedy and stuff :(
address_domain_regex = re.compile('\@(?P<domain>[\.\w-]+)')


def get_decoded_header(value):
    """Use python builtins to decode encoding stuff from header properly."""
    decoded_header_items = decode_header(value)
    decoded_header_value = ''
    for item in decoded_header_items:
        decoded_item = item[0].decode(item[1], 'replace') if item[1] is not None else item[0]
        if isinstance(decoded_item, bytes):
            decoded_item = decoded_item.decode('ascii', 'replace')
        decoded_header_value += decoded_item
    return decoded_header_value


def normalizeRawFromHeader(value):
    """Clean up linebreaks and spaces that are not needed."""
    return value.replace('\n', '').replace('\r', '').strip()


def getDomainFromValue(value):
    """ Check whether given 'From:' header label contains something that looks like an email address."""
    match = address_domain_regex.match(value)
    return match.group('domain').strip() if match is not None else None


class SuspiciousFrom(Milter.Base):
    def __init__(self):
        self.id = Milter.uniqueID()
        self.reset()
        logger.debug(f"({self.id}) Instanciated.")

    def reset(self):
        """It looks like one milter instance can reach eom hook multiple times.
           This allows to re-use an instance in a more clean way."""
        self.final_result = Milter.ACCEPT
        self.new_headers = []

    def set_suspicious_headers(self, is_suspicious, reason):
        str_okay = "PASS" if not is_suspicious else "FAIL"
        str_suspicious = "YES" if is_suspicious else "NO"
        self.new_headers.append({'name': 'X-From-Checked', 'value': f"{str_okay} - {reason}"})
        self.new_headers.append({'name': 'X-From-Suspicious', 'value': str_suspicious})

    def header(self, field, value):
        """Header hook gets called for every header within the email processed."""
        if field.lower() == 'from':
            logger.debug(f"({self.id}) \"From:\" raw: '{value}'")
            value = normalizeRawFromHeader(value)
            logger.info(f"({self.id}) \"From:\" cleaned: '{value}'")
            if value == '':
                logger.warning(f"\"From:\" header empty! WTF, but nothing to do. OK for now.")
                self.set_suspicious_headers(False, "EMPTY FROM HEADER - WTF")
            else:
                decoded_from = get_decoded_header(value)
                logger.debug(f"({self.id}) \"From:\" decoded raw: '{value}'")
                decoded_from = normalizeRawFromHeader(decoded_from)
                logger.info(f"({self.id}) \"From:\" decoded cleaned: '{decoded_from}'")
                all_domains = address_domain_regex.findall(decoded_from)
                all_domains = [a.lower() for a in all_domains]
                if len(all_domains) == 0:
                    logger.warning(f"({self.id}) No domain in decoded \"From:\" - WTF! OK, though")
                    self.set_suspicious_headers(False, "No domains in decoded FROM")
                elif len(all_domains) == 1:
                    logger.debug(f"({self.id}) Only one domain in decoded \"From:\": '{all_domains[0]}' - OK")
                    self.set_suspicious_headers(False, "Only one domain in decoded FROM")
                else:
                    logger.info(f"({self.id}) Raw decoded from header contains multiple domains: '{all_domains}' - Checking")
                    if len(set(all_domains)) > 1:
                        logger.info(f"({self.id}) Multiple different domains in decoded \"From:\". - NOT OK")
                        self.set_suspicious_headers(True, "Multiple domains in decoded FROM are different")
                    else:
                        logger.info(f"({self.id}) All domains in decoded \"From:\" are identical - OK")
                        self.set_suspicious_headers(False, "Multiple domains in decoded FROM match properly")
        # CONTINUE so we reach eom hook.
        # TODO: Log and react if multiple From-headers are found?
        return Milter.CONTINUE

    def eom(self):
        """EOM hook gets called at the end of message processed. Headers and final verdict are applied only here."""
        logger.info(f"({self.id}) EOM: Final verdict is {self.final_result}. New headers: {self.new_headers}")
        for new_header in self.new_headers:
            self.addheader(new_header['name'], new_header['value'])
        logger.debug(f"({self.id}) EOM: Reseting self.")
        self.reset()
        return self.final_result


def main():
    # TODO: Move this into configuration of some sort.
    Milter.factory = SuspiciousFrom
    logger.info(f"Starting Milter.")
    # This call blocks the main thread.
    # TODO: Improve handling CTRL+C
    Milter.runmilter("SuspiciousFromMilter", config.milter_socket, config.milter_timeout, rmsock=False)
    logger.info(f"Milter finished running.")


if __name__ == "__main__":
    main()
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00			`import sys`
			`import logging`

			`import Milter`

Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`from email.header import decode_header`
			`from email.utils import getaddresses`

Allow rough configuration to take place 2019-12-19 11:43:21 +01:00			`import re`

			`import config`
Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00			`# Basic logger that also logs to stdout`
			`# TODO: Improve this a lot.`
			`logger = logging.getLogger(__name__)`
Allow rough configuration to take place 2019-12-19 11:43:21 +01:00			`logger.setLevel(config.log_level)`
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00
			`handler = logging.StreamHandler(sys.stdout)`
Allow rough configuration to take place 2019-12-19 11:43:21 +01:00			`handler.setLevel(config.log_level)`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`formatter = logging.Formatter('%(asctime)s - %(levelname)s - %(message)s')`
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00			`handler.setFormatter(formatter)`

			`logger.addHandler(handler)`

Allow rough configuration to take place 2019-12-19 11:43:21 +01:00			`# Rough regex to fetch domain values from address-like text`
Big refactoring, getaddresses() is unreliable. :-( 2019-12-20 12:16:26 +01:00			`# Not matching the part in front of @ because greedy and stuff :(`
			`address_domain_regex = re.compile('\@(?P<domain>[\.\w-]+)')`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00
Introduce rough feature skeleton 2019-12-18 15:12:33 +01:00
Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`def get_decoded_header(value):`
Restructure due to unexpected, new edgecase 2019-12-20 11:10:07 +01:00			`"""Use python builtins to decode encoding stuff from header properly."""`
Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`decoded_header_items = decode_header(value)`
			`decoded_header_value = ''`
			`for item in decoded_header_items:`
Fix another decoding issue, too 2020-01-30 15:11:47 +01:00			`decoded_item = item[0].decode(item[1], 'replace') if item[1] is not None else item[0]`
Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`if isinstance(decoded_item, bytes):`
Fix error handling for decoding errors 2020-01-30 15:04:24 +01:00			`decoded_item = decoded_item.decode('ascii', 'replace')`
Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`decoded_header_value += decoded_item`
Restructure due to unexpected, new edgecase 2019-12-20 11:10:07 +01:00			`return decoded_header_value`
Fix bug that occurs with linebreaks within from header value 2019-12-18 16:41:55 +01:00

Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`def normalizeRawFromHeader(value):`
Allow rough configuration to take place 2019-12-19 11:43:21 +01:00			`"""Clean up linebreaks and spaces that are not needed."""`
Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`return value.replace('\n', '').replace('\r', '').strip()`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00

Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`def getDomainFromValue(value):`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`""" Check whether given 'From:' header label contains something that looks like an email address."""`
Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`match = address_domain_regex.match(value)`
Proof of concept level prototype - emails are only marked at this stage 2019-12-18 16:33:54 +01:00			`return match.group('domain').strip() if match is not None else None`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00
Introduce rough feature skeleton 2019-12-18 15:12:33 +01:00
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00			`class SuspiciousFrom(Milter.Base):`
			`def __init__(self):`
			`self.id = Milter.uniqueID()`
Proof of concept level prototype - emails are only marked at this stage 2019-12-18 16:33:54 +01:00			`self.reset()`
Handle issues with parse errors 2019-12-18 16:43:59 +01:00			`logger.debug(f"({self.id}) Instanciated.")`
Proof of concept level prototype - emails are only marked at this stage 2019-12-18 16:33:54 +01:00
			`def reset(self):`
Allow rough configuration to take place 2019-12-19 11:43:21 +01:00			`"""It looks like one milter instance can reach eom hook multiple times.`
			`This allows to re-use an instance in a more clean way."""`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`self.final_result = Milter.ACCEPT`
Introduce rough feature skeleton 2019-12-18 15:12:33 +01:00			`self.new_headers = []`
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00
Big refactoring, getaddresses() is unreliable. :-( 2019-12-20 12:16:26 +01:00			`def set_suspicious_headers(self, is_suspicious, reason):`
			`str_okay = "PASS" if not is_suspicious else "FAIL"`
			`str_suspicious = "YES" if is_suspicious else "NO"`
			`self.new_headers.append({'name': 'X-From-Checked', 'value': f"{str_okay} - {reason}"})`
			`self.new_headers.append({'name': 'X-From-Suspicious', 'value': str_suspicious})`

Initial import of basic skeleton 2019-12-18 14:25:09 +01:00			`def header(self, field, value):`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`"""Header hook gets called for every header within the email processed."""`
Introduce rough feature skeleton 2019-12-18 15:12:33 +01:00			`if field.lower() == 'from':`
Big refactoring, getaddresses() is unreliable. :-( 2019-12-20 12:16:26 +01:00			`logger.debug(f"({self.id}) \"From:\" raw: '{value}'")`
Fix bug that occurs with linebreaks within from header value 2019-12-18 16:41:55 +01:00			`value = normalizeRawFromHeader(value)`
Big refactoring, getaddresses() is unreliable. :-( 2019-12-20 12:16:26 +01:00			`logger.info(f"({self.id}) \"From:\" cleaned: '{value}'")`
Proof of concept level prototype - emails are only marked at this stage 2019-12-18 16:33:54 +01:00			`if value == '':`
Big refactoring, getaddresses() is unreliable. :-( 2019-12-20 12:16:26 +01:00			`logger.warning(f"\"From:\" header empty! WTF, but nothing to do. OK for now.")`
			`self.set_suspicious_headers(False, "EMPTY FROM HEADER - WTF")`
Introduce rough feature skeleton 2019-12-18 15:12:33 +01:00			`else:`
Big refactoring, getaddresses() is unreliable. :-( 2019-12-20 12:16:26 +01:00			`decoded_from = get_decoded_header(value)`
			`logger.debug(f"({self.id}) \"From:\" decoded raw: '{value}'")`
			`decoded_from = normalizeRawFromHeader(decoded_from)`
			`logger.info(f"({self.id}) \"From:\" decoded cleaned: '{decoded_from}'")`
			`all_domains = address_domain_regex.findall(decoded_from)`
Do not be case sensitive when comparing domains 2019-12-20 12:45:04 +01:00			`all_domains = [a.lower() for a in all_domains]`
Big refactoring, getaddresses() is unreliable. :-( 2019-12-20 12:16:26 +01:00			`if len(all_domains) == 0:`
			`logger.warning(f"({self.id}) No domain in decoded \"From:\" - WTF! OK, though")`
			`self.set_suspicious_headers(False, "No domains in decoded FROM")`
			`elif len(all_domains) == 1:`
			`logger.debug(f"({self.id}) Only one domain in decoded \"From:\": '{all_domains[0]}' - OK")`
			`self.set_suspicious_headers(False, "Only one domain in decoded FROM")`
Use builtin python tools to do address/header parsing 2019-12-19 11:34:18 +01:00			`else:`
Big refactoring, getaddresses() is unreliable. :-( 2019-12-20 12:16:26 +01:00			`logger.info(f"({self.id}) Raw decoded from header contains multiple domains: '{all_domains}' - Checking")`
			`if len(set(all_domains)) > 1:`
			`logger.info(f"({self.id}) Multiple different domains in decoded \"From:\". - NOT OK")`
			`self.set_suspicious_headers(True, "Multiple domains in decoded FROM are different")`
			`else:`
			`logger.info(f"({self.id}) All domains in decoded \"From:\" are identical - OK")`
			`self.set_suspicious_headers(False, "Multiple domains in decoded FROM match properly")`
			`# CONTINUE so we reach eom hook.`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`# TODO: Log and react if multiple From-headers are found?`
			`return Milter.CONTINUE`
Introduce rough feature skeleton 2019-12-18 15:12:33 +01:00
			`def eom(self):`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`"""EOM hook gets called at the end of message processed. Headers and final verdict are applied only here."""`
Proof of concept level prototype - emails are only marked at this stage 2019-12-18 16:33:54 +01:00			`logger.info(f"({self.id}) EOM: Final verdict is {self.final_result}. New headers: {self.new_headers}")`
Introduce rough feature skeleton 2019-12-18 15:12:33 +01:00			`for new_header in self.new_headers:`
			`self.addheader(new_header['name'], new_header['value'])`
Handle issues with parse errors 2019-12-18 16:43:59 +01:00			`logger.debug(f"({self.id}) EOM: Reseting self.")`
Proof of concept level prototype - emails are only marked at this stage 2019-12-18 16:33:54 +01:00			`self.reset()`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`return self.final_result`
Introduce rough feature skeleton 2019-12-18 15:12:33 +01:00
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00
			`def main():`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`# TODO: Move this into configuration of some sort.`
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00			`Milter.factory = SuspiciousFrom`
			`logger.info(f"Starting Milter.")`
			`# This call blocks the main thread.`
Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00			`# TODO: Improve handling CTRL+C`
Allow rough configuration to take place 2019-12-19 11:43:21 +01:00			`Milter.runmilter("SuspiciousFromMilter", config.milter_socket, config.milter_timeout, rmsock=False)`
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00			`logger.info(f"Milter finished running.")`

Lots of restructuring and basic implementing 2019-12-18 15:35:23 +01:00
Initial import of basic skeleton 2019-12-18 14:25:09 +01:00			`if __name__ == "__main__":`
			`main()`