Olof Mogren

42 papers B 4C 1Misc 1Journal 22Unranked 14
YearRankTypeTitle / Venue / Authors
2025 conf
EUSIPCO
Richard Lindholm, Oscar Marklund, Olof Mogren, John Martinsson
2025 J jnl
CoRR
Richard Lindholm, Oscar Marklund, Olof Mogren, John Martinsson
2025 J jnl
CoRR
John Martinsson, Olof Mogren, Tuomas Virtanen, Maria Sandsten
2025 J jnl
Trans. Mach. Learn. Res.
John Martinsson, Tuomas Virtanen, Maria Sandsten, Olof Mogren
2024 conf
NLDL
Edvin Listo Zec, Johan Östman, Olof Mogren, Daniel Gillblad
2024 J jnl
CoRR
Maria Bånkestad, Olof Mogren, Aleksis Pirinen
2024 conf
EUSIPCO
John Martinsson, Olof Mogren, Maria Sandsten, Tuomas Virtanen
2024 J jnl
CoRR
John Martinsson, Olof Mogren, Maria Sandsten, Tuomas Virtanen
2024 J jnl
CoRR
Martin Willbo, Aleksis Pirinen, John Martinsson, Edvin Listo Zec, Olof Mogren, Mikael Nilsson
2023 J jnl
CoRR
Marcus Toftås, Emilie Klefbom, Edvin Listo Zec, Martin Willbo, Olof Mogren
2023 conf
SAIS
Aleksis Pirinen, Olof Mogren, Mårten Västerdal
2023 J jnl
CoRR
Aleksis Pirinen, Olof Mogren, Mårten Västerdal
2023 J jnl
CoRR
Edvin Listo Zec, Olof Mogren
2023 J jnl
CoRR
Edvin Listo Zec, Johan Östman, Olof Mogren, Daniel Gillblad
2022 conf
FL@IJCAI
Edvin Listo Zec, Ebba Ekblom, Martin Willbo, Olof Mogren, Sarunas Girdzijauskas
2022 J jnl
CoRR
Edvin Listo Zec, Ebba Ekblom, Martin Willbo, Olof Mogren, Sarunas Girdzijauskas
2022 B conf
IEEE Big Data
Ebba Ekblom, Edvin Listo Zec, Olof Mogren
2022 J jnl
CoRR
Ebba Ekblom, Edvin Listo Zec, Olof Mogren
2022 conf
DCASE
John Martinsson, Martin Willbo, Aleksis Pirinen, Olof Mogren, Maria Sandsten
2022 B conf
IEEE Big Data
Edvin Listo Zec, Olof Mogren, Ann-Charlotte Mellquist, Sarah Fallahi, Peter Algurén
2021 conf
IEEE BigData
John Martinsson, Edvin Listo Zec, Daniel Gillblad, Olof Mogren
2021 J jnl
CoRR
Noa Onoszko, Gustav Karlsson, Olof Mogren, Edvin Listo Zec
2021 C conf
NLDB
Agrin Hilmkil, Sebastian Callh, Matteo Barbieri, Leon René Sütfeld, Edvin Listo Zec, Olof Mogren
2021 J jnl
CoRR
Agrin Hilmkil, Sebastian Callh, Matteo Barbieri, Leon René Sütfeld, Edvin Listo Zec, Olof Mogren
2020 J jnl
CoRR
David Ericsson, Adam Östberg, Edvin Listo Zec, John Martinsson, Olof Mogren
2020 J jnl
CoRR
John Martinsson, Edvin Listo Zec, Daniel Gillblad, Olof Mogren
2020 J jnl
J. Heal. Informatics Res.
John Martinsson, Alexander Schliep, Björn Eliasson, Olof Mogren
2020 J jnl
CoRR
Edvin Listo Zec, Olof Mogren, John Martinsson, Leon René Sütfeld, Daniel Gillblad
2019 conf
ICCV Workshops
Marie Korneliusson, John Martinsson, Olof Mogren
2019 conf
ICCV Workshops
John Martinsson, Olof Mogren
2018 conf
KDH@IJCAI
John Martinsson, Alexander Schliep, Björn Eliasson, Christian Meijner, Simon Persson, Olof Mogren
2017 conf
SWCN@EMNLP
Olof Mogren, Richard Johansson
2016 conf
Rep4NLP@ACL
Jacob Hagstedt P. Suorra, Olof Mogren
2016 J jnl
CoRR
Olof Mogren
2016 conf
BioTxtM@COLING 2016
Simon Almgren, Sean Pavlov, Olof Mogren
2016 B conf
SOFSEM
Azam Sheikh Muhammad, Peter Damaschke, Olof Mogren
2015 Misc conf
RANLP
Olof Mogren, Mikael Kågebäck, Devdatt P. Dubhashi
2015 J jnl
Int. J. Digit. Libr.
Nina Tahmasebi, Lars Borin, Gabriele Capannini, Devdatt P. Dubhashi, Peter Exner, Markus Forsberg, Gerhard Gossen, Fredrik D. Johansson, Richard Johansson, Mikael Kågebäck, Olof Mogren, Pierre Nugues, Thomas Risse
2014 J jnl
J. Graph Algorithms Appl.
Peter Damaschke, Olof Mogren
2014 B conf
WALCOM
Peter Damaschke, Olof Mogren
2014 conf
CVSC@EACL
Mikael Kågebäck, Olof Mogren, Nina Tahmasebi, Devdatt P. Dubhashi
2008 J jnl
CoRR
Olof Mogren, Oskar Sandberg, Vilhelm Verendel, Devdatt P. Dubhashi
redb/extractors/pe_extractors/pe_extra_findings.py
← Index redb/extractors/pe_extractors/pe_extra_findings.py python
from hashlib import md5, sha1, sha256
import inspect
import pefile
from magika import Magika
import json
from datetime import datetime, timezone
from typing import List, Optional, Any
from asn1crypto import cms, pem, x509, core
from dataclasses import asdict

from redb.extractors.enum import Tag
from redb.extractors.pe_extractor import PEExtractor
from redb.models.dataclasses import PEExtraFinding, PECertificate


class PEExtraFindings(PEExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        pe=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            pe,
        )
        self.elastic_index = self.index_prefix + "-pe_extrafindings"
        self.log.debug(inspect.currentframe().f_code.co_name)
        self.has_binary_overlay = False
        self.has_binary_resource = False
        self.findings = []

    def tag(self):
        return Tag.PE_EMBEDDED_EXTRAS.value

    """
    The section below triggers in case of a binary which is not signed.
    It trigger the scan for Certificates inside the binary, for example
    inside a resource or an overlay.
    """

    def _scan_for_certificates(self):
        """
        Comprehensive scan for certificate data in PE file.
        Returns list of tuples (location_description, certificate_data)
        """
        self.log.debug(inspect.currentframe().f_code.co_name)

        # 1. Scan resources for embedded PE files
        self._scan_resources_for_certificates()

        # 3. Scan overlay
        overlay_data = self.pe.get_overlay()
        if overlay_data:
            self._scan_overlay_for_certificates(overlay_data)


    def _extract_certificate_info(self, cert_data):
        self.log.debug(inspect.currentframe().f_code.co_name)

        certificate = x509.Certificate.load(cert_data)
        certificate_serial_number = format(certificate.serial_number, "x").upper()
        grouped_serial_number = ":".join(
            [
                certificate_serial_number[i : i + 2]
                for i in range(0, len(certificate_serial_number), 2)
            ]
        )
        return PECertificate(
            certificate_serial_number=grouped_serial_number,
            certificate_issuer=certificate.issuer.human_friendly,
            certificate_subject=certificate.subject.human_friendly,
            certificate_valid_from=certificate["tbs_certificate"]["validity"][
                "not_before"
            ].native.strftime("%Y-%m-%d %H:%M:%S"),
            certificate_valid_to=certificate["tbs_certificate"]["validity"][
                "not_after"
            ].native.strftime("%Y-%m-%d %H:%M:%S"),
            certificate_thumbprint=sha1(cert_data).hexdigest().upper(),
            certificate_algorithm=None,
        )

    def _extract_standard_certificates(self, pe):
        self.log.debug(inspect.currentframe().f_code.co_name)
        address = pe.OPTIONAL_HEADER.DATA_DIRECTORY[
            pefile.DIRECTORY_ENTRY["IMAGE_DIRECTORY_ENTRY_SECURITY"]
        ].VirtualAddress
        size = pe.OPTIONAL_HEADER.DATA_DIRECTORY[
            pefile.DIRECTORY_ENTRY["IMAGE_DIRECTORY_ENTRY_SECURITY"]
        ].Size

        addr_8 = address + 8
        signature_data = bytes(
            pe.write()[addr_8 : addr_8 + size]  # noqa E203
        )  # Ensure this is a bytes object
        if pem.detect(signature_data):
            signature_data = pem.unarmor(signature_data)

        content_info = cms.ContentInfo.load(signature_data)
        signed_data = content_info["content"]
        certificates = signed_data["certificates"]

        certificates_list = []
        for cert in certificates:
            cert_info = self._extract_certificate_info(cert.chosen.dump())
            certificates_list.append(cert_info)

        return certificates_list

    def _is_signed(self, pedata):
        address = pedata.OPTIONAL_HEADER.DATA_DIRECTORY[
            pefile.DIRECTORY_ENTRY["IMAGE_DIRECTORY_ENTRY_SECURITY"]
        ].VirtualAddress
        if address == 0:
            return False
        return True

    def _scan_resources_for_certificates(self):
        """Scan resources for embedded PE files and their certificates"""
        self.log.debug(inspect.currentframe().f_code.co_name)

        try:
            if hasattr(self.pe, "DIRECTORY_ENTRY_RESOURCE"):
                for entry in self.pe.DIRECTORY_ENTRY_RESOURCE.entries:
                    for resource in self._get_resource_data(entry):
                        # Check if resource data might be a PE file
                        resource_magika = Magika().identify_bytes(resource).output.label
                        if resource.startswith(b"MZ") or resource_magika == "pebin":
                            try:
                                embedded_pe = pefile.PE(data=resource)
                                if self._is_signed(embedded_pe):
                                    cert_data = self._extract_standard_certificates(
                                        embedded_pe
                                    )
                                    if cert_data:
                                        # certificates.append(
                                        #     (f"Resource_{entry.id}", cert_data)
                                        # )
                                        self.findings.append(
                                            PEExtraFinding(
                                                context=f"Resource ({sha256(resource).hexdigest()}) is a PE file. Found Code Signing Certificates information inside.",
                                                finding=cert_data,
                                            )
                                        )
                            except Exception as e:
                                self.log.warning(
                                    f"Resource {sha256(resource).hexdigest()} is not a valid PE: {e}"
                                )

                        # Direct certificate pattern matching in resource
                        cert_data = self._find_certificate_patterns(resource)
                        if cert_data:
                            # certificates.append((f"Resource_{entry.id}", cert_data))
                            self.findings.append(
                                PEExtraFinding(
                                    context=f"Resource ({sha256(resource).hexdigest()}) is a PE file.\
                                    Found Code Signing Certificates information inside.",
                                    finding=cert_data,
                                )
                            )

        except Exception as e:
            self.log.error(f"Failed to scan resources: {e}")


    def _scan_overlay_for_certificates(self, overlay_data) -> List[bytes]:
        """Scan PE overlay for certificate data"""
        self.log.debug(inspect.currentframe().f_code.co_name)
        try:
            # Check if overlay is a PE file
            overlay_magika = Magika().identify_bytes(overlay_data).output.ct_label
            if overlay_data.startswith(b"MZ") or overlay_magika == "pebin":
                try:
                    overlay_pe = pefile.PE(data=overlay_data)
                    if self._is_signed(overlay_pe):
                        cert_data = self._extract_standard_certificates(overlay_data)
                        if cert_data:
                            self.findings.append(
                                PEExtraFinding(
                                    context=f"Overlay data ({sha256(overlay_data).hexdigest()}) is a PE file.\
                                    Found Code Signing Certificates information inside.",
                                    finding=cert_data,
                                )
                            )
                except Exception as e:
                    self.log.warning(f"Error looking for certificates in Overlay: {e}")
                    pass

            # Direct certificate pattern matching in overlay
            cert_data = self._find_certificate_patterns(overlay_data)
            if cert_data:
                self.findings.append(
                    PEExtraFinding(
                        context=f"Overlay data ({sha256(overlay_data).hexdigest()}) is a PE file.\
                        Found Code Signing Certificates information inside.",
                        finding=cert_data,
                    )
                )
        except Exception as e:
            self.log.error(f"Failed to scan overlay: {e}")


    def _get_resource_data(self, entry) -> List[bytes]:
        """Recursively get resource data"""
        resources = []
        try:
            if hasattr(entry, "directory"):
                for subentry in entry.directory.entries:
                    resources.extend(self._get_resource_data(subentry))
            else:
                data = self.pe.get_data(
                    entry.data.struct.OffsetToData, entry.data.struct.Size
                )
                resources.append(data)
        except Exception as e:
            self.log.error(f"Failed to get resource data: {e}")
        return resources

    def _find_certificate_patterns(self, data: bytes) -> Optional[bytes]:
        """Find certificate patterns in binary data"""
        # PKCS#7 SignedData OID pattern
        PKCS7_PATTERN = b"\x06\x09\x2A\x86\x48\x86\xF7\x0D\x01\x07\x02"

        try:
            idx = data.find(PKCS7_PATTERN)
            if idx != -1:
                # Search backwards for ASN.1 sequence marker
                for i in range(idx, max(0, idx - 50), -1):
                    if data[i] == 0x30:  # ASN.1 SEQUENCE
                        try:
                            # Try to parse as X.509 certificate
                            cert = x509.load_der_x509_certificate(
                                data[i:], default_backend()
                            )
                            return data[i:]
                        except Exception:
                            continue
        except Exception as e:
            self.log.error(f"Failed to find certificate patterns: {e}")
        return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        self.log.debug(inspect.currentframe().f_code.co_name)
        if exporter_type == "ElasticsearchExporter":
            return self.findings
        elif exporter_type == "ClickHouseExporter":
            current_time = datetime.now(timezone.utc)
            
            data = [
                [
                    self.sha256,
                    self.md5,
                    self.sha1,
                    finding.context,
                    [asdict(cert) for cert in finding.finding] if finding.finding else None,
                    current_time
                ]
                for finding in self.findings
            ]
            
            column_names = [
                'sha256', 'md5', 'sha1', 'context', 'finding', 'analysis_date'
            ]
            
            if not data:
                return None

            column_type_names = [
                'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                'String', 'Array(JSON)', 'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_extra_findings"

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)
            self._scan_for_certificates()
            if len(self.findings) > 0:
                return self.findings
            return None
        except Exception as e:
            self.log.error(f"Error extracting PE Extras: {e}")
            return None