Olivier Perrotin

39 papers A 9Misc 1Journal 12Unranked 14
YearRankTypeTitle / Venue / Authors
2026 J jnl
Comput. Speech Lang.
Martin Lenglet, Olivier Perrotin, Gérard Bailly
2026 J jnl
Speech Commun.
Delphine Charuau, Nathalie Henrich Bernardoni, Silvain Gerber, Olivier Perrotin
2026 J jnl
Comput. Speech Lang.
Ihab Asaad, Maxime Jacquelin, Olivier Perrotin, Laurent Girin, Thomas Hueber
2025 A conf
INTERSPEECH
Maxime Jacquelin, Maëva Garnier, Laurent Girin, Rémy Vincent, Olivier Perrotin
2025 J jnl
Comput. Speech Lang.
Olivier Perrotin, Brooke Stephenson, Silvain Gerber, Gérard Bailly, Simon King
2025 J jnl
CoRR
Shree Harsha Bokkahalli Satish, Harm Lameris, Olivier Perrotin, Gustav Eje Henter, Éva Székely
2024 conf
LREC/COLING
Gérard Bailly, Romain Legrand, Martin Lenglet, Frédéric Elisei, Maëva Hueber, Olivier Perrotin
2024 conf
TALN (JEP)
Maxime Jacquelin, Maëva Garnier, Laurent Girin, Rémy Vincent, Olivier Perrotin
2024 conf
ICASSP Workshops
Maxime Jacquelin, Maëva Garnier, Laurent Girin, Rémy Vincent, Olivier Perrotin
2024 A conf
INTERSPEECH
Martin Lenglet, Olivier Perrotin, Gérard Bailly
2024 J jnl
CoRR
Ihab Asaad, Maxime Jacquelin, Olivier Perrotin, Laurent Girin, Thomas Hueber
2024 conf
TALN (JEP)
Delphine Charuau, Nathalie Henrich Bernardoni, Silvain Gerber, Olivier Perrotin
2023 ed.
SSW
Gérard Bailly, Thomas Hueber, Damien Lolive, Nicolas Obin, Olivier Perrotin
2023 ed.
Blizzard Challenge
Olivier Perrotin, Gérard Bailly, Simon King
2023 conf
SSW
Gérard Bailly, Martin Lenglet, Olivier Perrotin, Esther Klabbers
2023 conf
SSW
Maxime Jacquelin, Maeva Garnier, Laurent Girin, Rémy Vincent, Olivier Perrotin
2023 A conf
INTERSPEECH
Sanjana Sankar, Denis Beautemps, Frédéric Elisei, Olivier Perrotin, Thomas Hueber
2023 J jnl
CoRR
Sanjana Sankar, Denis Beautemps, Frédéric Elisei, Olivier Perrotin, Thomas Hueber
2023 conf
SSW
Martin Lenglet, Olivier Perrotin, Gérard Bailly
2023 conf
Blizzard Challenge
Olivier Perrotin, Brooke Stephenson, Silvain Gerber, Gérard Bailly
2023 conf
Blizzard Challenge
Martin Lenglet, Olivier Perrotin, Gérard Bailly
2022 conf
SPECOM
Maria-Loulou Hajj, Martin Lenglet, Olivier Perrotin, Gérard Bailly
2022 A conf
INTERSPEECH
Martin Lenglet, Olivier Perrotin, Gérard Bailly
2022 A conf
INTERSPEECH
Luc Ardaillon, Nathalie Henrich Bernardoni, Olivier Perrotin
2021 A conf
Interspeech
Olivier Perrotin, Hussein El Amouri, Gérard Bailly, Thomas Hueber
2021 conf
SSW
Martin Lenglet, Olivier Perrotin, Gérard Bailly
2021 conf
AHFE (1)
Franck Tarpin-Bernard, Joan Fruitet, Jean-Philippe Vigne, Patrick Constant, Hanna Chainay, Olivier Koenig, Fabien Ringeval, Béatrice Bouchot, Gérard Bailly, François Portet, Sina Alisamir, Yongxin Zhou, Jean Serre, Vincent Delerue, Hippolyte Fournier, Kévin Berenger, Isabella Zsoldos, Olivier Perrotin, Frédéric Elisei, Martin Lenglet, Charles Puaux, Léo Pacheco, Mélodie Fouillen, Didier Ghenassia
2020 J jnl
IEEE ACM Trans. Audio Speech Lang. Process.
Olivier Perrotin, Ian Vince McLoughlin
2020 A conf
INTERSPEECH
Jacob J. Webber, Olivier Perrotin, Simon King
2019 Misc conf
ICASSP
Olivier Perrotin, Ian McLoughlin
2019 A conf
INTERSPEECH
Olivier Perrotin, Ian McLoughlin
2017 J jnl
EURASIP J. Audio Speech Music. Process.
Lionel Feugère, Christophe d'Alessandro, Boris Doval, Olivier Perrotin
2017 J jnl
CoRR
Olivier Perrotin, Ian McLoughlin
2017 J jnl
ACM Trans. Appl. Percept.
Olivier Perrotin, Christophe d'Alessandro
2016 J jnl
ACM Trans. Comput. Hum. Interact.
Olivier Perrotin, Christophe d'Alessandro
2016 A conf
INTERSPEECH
Olivier Perrotin, Christophe d'Alessandro
2015
Olivier Perrotin
2014 conf
NIME
Olivier Perrotin, Christophe d'Alessandro
2013 conf
NIME
Olivier Perrotin, Christophe d'Alessandro
redb/extractors/elf_extractors/elf_sections.py
← Index redb/extractors/elf_extractors/elf_sections.py python
import inspect
import hashlib
import math
from collections import Counter
from datetime import datetime, timezone
from typing import Any, List, Dict

from elftools.elf.elffile import ELFFile
from elftools.common.exceptions import ELFError

from redb.extractors.enum import Tag
from redb.extractors.elf_extractor import ELFExtractor
from redb.models.dataclasses import ELFSection


class ELFSectionExtractor(ELFExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        elf=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            elf,
        )
        self.elf_sections = []
        self.elastic_index = self.index_prefix + "-elf_sections"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def _is_empty_result(self, extracted_data) -> bool:
        """
        Override: Empty sections is an ERROR, not a valid empty case.
        A valid ELF file must have sections (at minimum a null section).
        """
        # Always return False - empty sections should be treated as an error
        return False

    def _calculate_entropy(self, data: bytes) -> float:
        """Calculate Shannon entropy of data."""
        if not data:
            return 0.0

        try:
            # Count frequency of each byte
            byte_counts = Counter(data)
            data_len = len(data)

            # Calculate entropy
            entropy = 0.0
            for count in byte_counts.values():
                if count > 0:
                    frequency = count / data_len
                    entropy -= frequency * math.log2(frequency)

            return entropy
        except Exception as e:
            self.log.error(f"Error calculating entropy: {e}")
            return 0.0

    def _map_section_type(self, sh_type_str: str) -> int:
        """Map section type string to enum value."""
        type_map = {
            'SHT_NULL': 0,
            'SHT_PROGBITS': 1,
            'SHT_SYMTAB': 2,
            'SHT_STRTAB': 3,
            'SHT_RELA': 4,
            'SHT_HASH': 5,
            'SHT_DYNAMIC': 6,
            'SHT_NOTE': 7,
            'SHT_NOBITS': 8,
            'SHT_REL': 9,
            'SHT_DYNSYM': 11
        }
        return type_map.get(sh_type_str, 0)

    def _decode_section_flags(self, flags: int) -> List[str]:
        """Decode section flags to human-readable strings."""
        flag_strings = []

        # Common ELF section flags
        if flags & 0x1:  # SHF_WRITE
            flag_strings.append('WRITE')
        if flags & 0x2:  # SHF_ALLOC
            flag_strings.append('ALLOC')
        if flags & 0x4:  # SHF_EXECINSTR
            flag_strings.append('EXECINSTR')
        if flags & 0x10:  # SHF_MERGE
            flag_strings.append('MERGE')
        if flags & 0x20:  # SHF_STRINGS
            flag_strings.append('STRINGS')
        if flags & 0x40:  # SHF_INFO_LINK
            flag_strings.append('INFO_LINK')
        if flags & 0x80:  # SHF_LINK_ORDER
            flag_strings.append('LINK_ORDER')
        if flags & 0x100:  # SHF_OS_NONCONFORMING
            flag_strings.append('OS_NONCONFORMING')
        if flags & 0x200:  # SHF_GROUP
            flag_strings.append('GROUP')
        if flags & 0x400:  # SHF_TLS
            flag_strings.append('TLS')

        return flag_strings if flag_strings else ['NONE']

    def _extract_section_data(self, section) -> Dict:
        """Extract data from a single section."""
        try:
            header = section.header

            # Get section name (handle empty names)
            section_name = section.name if section.name else f"<unnamed_{section.header.get('sh_name', 0)}>"

            # Get section type and map to enum
            sh_type_str = header.get('sh_type', 'SHT_NULL')
            section_type_enum = self._map_section_type(sh_type_str)
            section_type_str = sh_type_str.replace('SHT_', '') if sh_type_str.startswith('SHT_') else sh_type_str

            # Get section properties
            section_flags = header.get('sh_flags', 0)
            section_flags_str = self._decode_section_flags(section_flags)
            section_addr = header.get('sh_addr', 0)
            section_offset = header.get('sh_offset', 0)
            section_size = header.get('sh_size', 0)
            section_link = header.get('sh_link', 0)
            section_info = header.get('sh_info', 0)
            section_addralign = header.get('sh_addralign', 0)
            section_entsize = header.get('sh_entsize', 0)

            # Calculate entropy and hashes for section data
            section_entropy = 0.0
            section_sha256 = ""
            section_md5 = ""

            try:
                if section_size > 0 and section_type_str != 'NOBITS':
                    section_data = section.data()
                    if section_data:
                        # Calculate entropy
                        section_entropy = self._calculate_entropy(section_data)

                        # Calculate hashes
                        section_sha256 = hashlib.sha256(section_data).hexdigest()
                        section_md5 = hashlib.md5(section_data).hexdigest()
            except Exception as e:
                self.log.warning(f"Could not read section '{section_name}' data: {e}")
                section_entropy = 0.0
                section_sha256 = ""
                section_md5 = ""

            return ELFSection(
                section_name=section_name,
                section_type=section_type_enum,
                section_type_str=section_type_str,
                section_flags=section_flags,
                section_flags_str=section_flags_str,
                section_addr=section_addr,
                section_offset=section_offset,
                section_size=section_size,
                section_link=section_link,
                section_info=section_info,
                section_addralign=section_addralign,
                section_entsize=section_entsize,
                section_entropy=section_entropy,
                section_sha256=section_sha256,
                section_md5=section_md5
            )

        except Exception as e:
            self.log.error(f"Error extracting section data: {e}")
            return None

    def tag(self):
        return Tag.ELF_SECTIONS.value if hasattr(Tag, 'ELF_SECTIONS') else "elf_sections"

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)

            def extract_data(elf):
                sections_data = []

                # Iterate through all sections with per-section error handling
                for section_index, section in enumerate(elf.iter_sections()):
                    try:
                        section_data = self._extract_section_data(section)
                        if section_data:
                            sections_data.append(section_data)
                        else:
                            self.log.warning(f"Failed to extract data for section {section_index}")
                    except Exception as e:
                        self.log.warning(f"Error processing section {section_index}: {e}")
                        # Continue processing other sections

                return sections_data

            if not self._is_elf_file():
                return None

            result = self._with_elf_file(extract_data)
            if result is None:
                return None

            self.elf_sections = result
            return self.elf_sections

        except Exception as e:
            self.log.error(f"Error extracting ELF sections {self.hash.sha256}: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        self.log.debug(inspect.currentframe().f_code.co_name)

        if exporter_type == "ElasticsearchExporter":
            return self.elf_sections
        elif exporter_type == "ClickHouseExporter":
            try:
                if not self.elf_sections:
                    return None

                # Prepare data arrays for all sections
                data = []
                current_time = datetime.now(timezone.utc)
                for section in self.elf_sections:
                    row = [
                        self.sha256,
                        self.md5,
                        self.sha1,
                        section.section_name,
                        section.section_type,
                        section.section_type_str,
                        section.section_flags,
                        section.section_flags_str,
                        section.section_addr,
                        section.section_offset,
                        section.section_size,
                        section.section_link,
                        section.section_info,
                        section.section_addralign,
                        section.section_entsize,
                        section.section_entropy,
                        section.section_sha256,
                        section.section_md5,
                        current_time
                    ]
                    data.append(row)

                column_names = [
                    'sha256', 'md5', 'sha1',
                    'section_name', 'section_type', 'section_type_str',
                    'section_flags', 'section_flags_str',
                    'section_addr', 'section_offset', 'section_size',
                    'section_link', 'section_info', 'section_addralign', 'section_entsize',
                    'section_entropy', 'section_sha256', 'section_md5',
                    'analysis_date'
                ]

                column_type_names = [
                    'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                    'LowCardinality(String)',
                    "Enum8('NULL'=0, 'PROGBITS'=1, 'SYMTAB'=2, 'STRTAB'=3, 'RELA'=4, 'HASH'=5, 'DYNAMIC'=6, 'NOTE'=7, 'NOBITS'=8, 'REL'=9, 'DYNSYM'=11)",
                    'LowCardinality(String)',
                    'UInt64',
                    'Array(LowCardinality(String))',
                    'UInt64', 'UInt64', 'UInt64',
                    'UInt32', 'UInt32', 'UInt64', 'UInt64',
                    'Float64',
                    'FixedString(64)', 'FixedString(32)',
                    'DateTime64(3, \'UTC\')'
                ]

                if not data:
                    return None

                return (data, column_names, column_type_names)

            except Exception as e:
                self.log.error(f"Error preparing export data: {e}")
                raise

    def get_clickhouse_table(self) -> str:
        return "redb_elf_sections"