Olof Mogren

42 papers B 4C 1Misc 1Journal 22Unranked 14
YearRankTypeTitle / Venue / Authors
2025 conf
EUSIPCO
Richard Lindholm, Oscar Marklund, Olof Mogren, John Martinsson
2025 J jnl
CoRR
Richard Lindholm, Oscar Marklund, Olof Mogren, John Martinsson
2025 J jnl
CoRR
John Martinsson, Olof Mogren, Tuomas Virtanen, Maria Sandsten
2025 J jnl
Trans. Mach. Learn. Res.
John Martinsson, Tuomas Virtanen, Maria Sandsten, Olof Mogren
2024 conf
NLDL
Edvin Listo Zec, Johan Östman, Olof Mogren, Daniel Gillblad
2024 J jnl
CoRR
Maria Bånkestad, Olof Mogren, Aleksis Pirinen
2024 conf
EUSIPCO
John Martinsson, Olof Mogren, Maria Sandsten, Tuomas Virtanen
2024 J jnl
CoRR
John Martinsson, Olof Mogren, Maria Sandsten, Tuomas Virtanen
2024 J jnl
CoRR
Martin Willbo, Aleksis Pirinen, John Martinsson, Edvin Listo Zec, Olof Mogren, Mikael Nilsson
2023 J jnl
CoRR
Marcus Toftås, Emilie Klefbom, Edvin Listo Zec, Martin Willbo, Olof Mogren
2023 conf
SAIS
Aleksis Pirinen, Olof Mogren, Mårten Västerdal
2023 J jnl
CoRR
Aleksis Pirinen, Olof Mogren, Mårten Västerdal
2023 J jnl
CoRR
Edvin Listo Zec, Olof Mogren
2023 J jnl
CoRR
Edvin Listo Zec, Johan Östman, Olof Mogren, Daniel Gillblad
2022 conf
FL@IJCAI
Edvin Listo Zec, Ebba Ekblom, Martin Willbo, Olof Mogren, Sarunas Girdzijauskas
2022 J jnl
CoRR
Edvin Listo Zec, Ebba Ekblom, Martin Willbo, Olof Mogren, Sarunas Girdzijauskas
2022 B conf
IEEE Big Data
Ebba Ekblom, Edvin Listo Zec, Olof Mogren
2022 J jnl
CoRR
Ebba Ekblom, Edvin Listo Zec, Olof Mogren
2022 conf
DCASE
John Martinsson, Martin Willbo, Aleksis Pirinen, Olof Mogren, Maria Sandsten
2022 B conf
IEEE Big Data
Edvin Listo Zec, Olof Mogren, Ann-Charlotte Mellquist, Sarah Fallahi, Peter Algurén
2021 conf
IEEE BigData
John Martinsson, Edvin Listo Zec, Daniel Gillblad, Olof Mogren
2021 J jnl
CoRR
Noa Onoszko, Gustav Karlsson, Olof Mogren, Edvin Listo Zec
2021 C conf
NLDB
Agrin Hilmkil, Sebastian Callh, Matteo Barbieri, Leon René Sütfeld, Edvin Listo Zec, Olof Mogren
2021 J jnl
CoRR
Agrin Hilmkil, Sebastian Callh, Matteo Barbieri, Leon René Sütfeld, Edvin Listo Zec, Olof Mogren
2020 J jnl
CoRR
David Ericsson, Adam Östberg, Edvin Listo Zec, John Martinsson, Olof Mogren
2020 J jnl
CoRR
John Martinsson, Edvin Listo Zec, Daniel Gillblad, Olof Mogren
2020 J jnl
J. Heal. Informatics Res.
John Martinsson, Alexander Schliep, Björn Eliasson, Olof Mogren
2020 J jnl
CoRR
Edvin Listo Zec, Olof Mogren, John Martinsson, Leon René Sütfeld, Daniel Gillblad
2019 conf
ICCV Workshops
Marie Korneliusson, John Martinsson, Olof Mogren
2019 conf
ICCV Workshops
John Martinsson, Olof Mogren
2018 conf
KDH@IJCAI
John Martinsson, Alexander Schliep, Björn Eliasson, Christian Meijner, Simon Persson, Olof Mogren
2017 conf
SWCN@EMNLP
Olof Mogren, Richard Johansson
2016 conf
Rep4NLP@ACL
Jacob Hagstedt P. Suorra, Olof Mogren
2016 J jnl
CoRR
Olof Mogren
2016 conf
BioTxtM@COLING 2016
Simon Almgren, Sean Pavlov, Olof Mogren
2016 B conf
SOFSEM
Azam Sheikh Muhammad, Peter Damaschke, Olof Mogren
2015 Misc conf
RANLP
Olof Mogren, Mikael Kågebäck, Devdatt P. Dubhashi
2015 J jnl
Int. J. Digit. Libr.
Nina Tahmasebi, Lars Borin, Gabriele Capannini, Devdatt P. Dubhashi, Peter Exner, Markus Forsberg, Gerhard Gossen, Fredrik D. Johansson, Richard Johansson, Mikael Kågebäck, Olof Mogren, Pierre Nugues, Thomas Risse
2014 J jnl
J. Graph Algorithms Appl.
Peter Damaschke, Olof Mogren
2014 B conf
WALCOM
Peter Damaschke, Olof Mogren
2014 conf
CVSC@EACL
Mikael Kågebäck, Olof Mogren, Nina Tahmasebi, Devdatt P. Dubhashi
2008 J jnl
CoRR
Olof Mogren, Oskar Sandberg, Vilhelm Verendel, Devdatt P. Dubhashi
redb/extractors/elf_extractors/elf_sections.py
← Index redb/extractors/elf_extractors/elf_sections.py python
import inspect
import hashlib
import math
from collections import Counter
from datetime import datetime, timezone
from typing import Any, List, Dict

from elftools.elf.elffile import ELFFile
from elftools.common.exceptions import ELFError

from redb.extractors.enum import Tag
from redb.extractors.elf_extractor import ELFExtractor
from redb.models.dataclasses import ELFSection


class ELFSectionExtractor(ELFExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        elf=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            elf,
        )
        self.elf_sections = []
        self.elastic_index = self.index_prefix + "-elf_sections"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def _is_empty_result(self, extracted_data) -> bool:
        """
        Override: Empty sections is an ERROR, not a valid empty case.
        A valid ELF file must have sections (at minimum a null section).
        """
        # Always return False - empty sections should be treated as an error
        return False

    def _calculate_entropy(self, data: bytes) -> float:
        """Calculate Shannon entropy of data."""
        if not data:
            return 0.0

        try:
            # Count frequency of each byte
            byte_counts = Counter(data)
            data_len = len(data)

            # Calculate entropy
            entropy = 0.0
            for count in byte_counts.values():
                if count > 0:
                    frequency = count / data_len
                    entropy -= frequency * math.log2(frequency)

            return entropy
        except Exception as e:
            self.log.error(f"Error calculating entropy: {e}")
            return 0.0

    def _map_section_type(self, sh_type_str: str) -> int:
        """Map section type string to enum value."""
        type_map = {
            'SHT_NULL': 0,
            'SHT_PROGBITS': 1,
            'SHT_SYMTAB': 2,
            'SHT_STRTAB': 3,
            'SHT_RELA': 4,
            'SHT_HASH': 5,
            'SHT_DYNAMIC': 6,
            'SHT_NOTE': 7,
            'SHT_NOBITS': 8,
            'SHT_REL': 9,
            'SHT_DYNSYM': 11
        }
        return type_map.get(sh_type_str, 0)

    def _decode_section_flags(self, flags: int) -> List[str]:
        """Decode section flags to human-readable strings."""
        flag_strings = []

        # Common ELF section flags
        if flags & 0x1:  # SHF_WRITE
            flag_strings.append('WRITE')
        if flags & 0x2:  # SHF_ALLOC
            flag_strings.append('ALLOC')
        if flags & 0x4:  # SHF_EXECINSTR
            flag_strings.append('EXECINSTR')
        if flags & 0x10:  # SHF_MERGE
            flag_strings.append('MERGE')
        if flags & 0x20:  # SHF_STRINGS
            flag_strings.append('STRINGS')
        if flags & 0x40:  # SHF_INFO_LINK
            flag_strings.append('INFO_LINK')
        if flags & 0x80:  # SHF_LINK_ORDER
            flag_strings.append('LINK_ORDER')
        if flags & 0x100:  # SHF_OS_NONCONFORMING
            flag_strings.append('OS_NONCONFORMING')
        if flags & 0x200:  # SHF_GROUP
            flag_strings.append('GROUP')
        if flags & 0x400:  # SHF_TLS
            flag_strings.append('TLS')

        return flag_strings if flag_strings else ['NONE']

    def _extract_section_data(self, section) -> Dict:
        """Extract data from a single section."""
        try:
            header = section.header

            # Get section name (handle empty names)
            section_name = section.name if section.name else f"<unnamed_{section.header.get('sh_name', 0)}>"

            # Get section type and map to enum
            sh_type_str = header.get('sh_type', 'SHT_NULL')
            section_type_enum = self._map_section_type(sh_type_str)
            section_type_str = sh_type_str.replace('SHT_', '') if sh_type_str.startswith('SHT_') else sh_type_str

            # Get section properties
            section_flags = header.get('sh_flags', 0)
            section_flags_str = self._decode_section_flags(section_flags)
            section_addr = header.get('sh_addr', 0)
            section_offset = header.get('sh_offset', 0)
            section_size = header.get('sh_size', 0)
            section_link = header.get('sh_link', 0)
            section_info = header.get('sh_info', 0)
            section_addralign = header.get('sh_addralign', 0)
            section_entsize = header.get('sh_entsize', 0)

            # Calculate entropy and hashes for section data
            section_entropy = 0.0
            section_sha256 = ""
            section_md5 = ""

            try:
                if section_size > 0 and section_type_str != 'NOBITS':
                    section_data = section.data()
                    if section_data:
                        # Calculate entropy
                        section_entropy = self._calculate_entropy(section_data)

                        # Calculate hashes
                        section_sha256 = hashlib.sha256(section_data).hexdigest()
                        section_md5 = hashlib.md5(section_data).hexdigest()
            except Exception as e:
                self.log.warning(f"Could not read section '{section_name}' data: {e}")
                section_entropy = 0.0
                section_sha256 = ""
                section_md5 = ""

            return ELFSection(
                section_name=section_name,
                section_type=section_type_enum,
                section_type_str=section_type_str,
                section_flags=section_flags,
                section_flags_str=section_flags_str,
                section_addr=section_addr,
                section_offset=section_offset,
                section_size=section_size,
                section_link=section_link,
                section_info=section_info,
                section_addralign=section_addralign,
                section_entsize=section_entsize,
                section_entropy=section_entropy,
                section_sha256=section_sha256,
                section_md5=section_md5
            )

        except Exception as e:
            self.log.error(f"Error extracting section data: {e}")
            return None

    def tag(self):
        return Tag.ELF_SECTIONS.value if hasattr(Tag, 'ELF_SECTIONS') else "elf_sections"

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)

            def extract_data(elf):
                sections_data = []

                # Iterate through all sections with per-section error handling
                for section_index, section in enumerate(elf.iter_sections()):
                    try:
                        section_data = self._extract_section_data(section)
                        if section_data:
                            sections_data.append(section_data)
                        else:
                            self.log.warning(f"Failed to extract data for section {section_index}")
                    except Exception as e:
                        self.log.warning(f"Error processing section {section_index}: {e}")
                        # Continue processing other sections

                return sections_data

            if not self._is_elf_file():
                return None

            result = self._with_elf_file(extract_data)
            if result is None:
                return None

            self.elf_sections = result
            return self.elf_sections

        except Exception as e:
            self.log.error(f"Error extracting ELF sections {self.hash.sha256}: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        self.log.debug(inspect.currentframe().f_code.co_name)

        if exporter_type == "ElasticsearchExporter":
            return self.elf_sections
        elif exporter_type == "ClickHouseExporter":
            try:
                if not self.elf_sections:
                    return None

                # Prepare data arrays for all sections
                data = []
                current_time = datetime.now(timezone.utc)
                for section in self.elf_sections:
                    row = [
                        self.sha256,
                        self.md5,
                        self.sha1,
                        section.section_name,
                        section.section_type,
                        section.section_type_str,
                        section.section_flags,
                        section.section_flags_str,
                        section.section_addr,
                        section.section_offset,
                        section.section_size,
                        section.section_link,
                        section.section_info,
                        section.section_addralign,
                        section.section_entsize,
                        section.section_entropy,
                        section.section_sha256,
                        section.section_md5,
                        current_time
                    ]
                    data.append(row)

                column_names = [
                    'sha256', 'md5', 'sha1',
                    'section_name', 'section_type', 'section_type_str',
                    'section_flags', 'section_flags_str',
                    'section_addr', 'section_offset', 'section_size',
                    'section_link', 'section_info', 'section_addralign', 'section_entsize',
                    'section_entropy', 'section_sha256', 'section_md5',
                    'analysis_date'
                ]

                column_type_names = [
                    'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                    'LowCardinality(String)',
                    "Enum8('NULL'=0, 'PROGBITS'=1, 'SYMTAB'=2, 'STRTAB'=3, 'RELA'=4, 'HASH'=5, 'DYNAMIC'=6, 'NOTE'=7, 'NOBITS'=8, 'REL'=9, 'DYNSYM'=11)",
                    'LowCardinality(String)',
                    'UInt64',
                    'Array(LowCardinality(String))',
                    'UInt64', 'UInt64', 'UInt64',
                    'UInt32', 'UInt32', 'UInt64', 'UInt64',
                    'Float64',
                    'FixedString(64)', 'FixedString(32)',
                    'DateTime64(3, \'UTC\')'
                ]

                if not data:
                    return None

                return (data, column_names, column_type_names)

            except Exception as e:
                self.log.error(f"Error preparing export data: {e}")
                raise

    def get_clickhouse_table(self) -> str:
        return "redb_elf_sections"