Rami Albatal

35 papers A* 1A 2B 1C 2Journal 1Unranked 27
YearRankTypeTitle / Venue / Authors
2022 conf
NTCIR
Liting Zhou, Cathal Gurrin, Graham Healy, Hideo Joho, Binh T. Nguyen, Rami Albatal, Frank Hopfgartner, Duc-Tien Dang-Nguyen
2019 conf
MMM (1)
Cathal Gurrin, Klaus Schoeffmann, Hideo Joho, Bernd Münzer, Rami Albatal, Frank Hopfgartner, Liting Zhou, Duc-Tien Dang-Nguyen
2019 conf
NTCIR (Revised Selected Papers)
Cathal Gurrin, Hideo Joho, Frank Hopfgartner, Liting Zhou, Van-Tu Ninh, Tu-Khiem Le, Rami Albatal, Duc-Tien Dang-Nguyen, Graham Healy
2019 conf
NTCIR
Cathal Gurrin, Hideo Joho, Frank Hopfgartner, Liting Zhou, Van-Tu Ninh, Tu-Khiem Le, Rami Albatal, Duc-Tien Dang-Nguyen, Graham Healy
2017 J jnl
Neurocomputing
Zhenxing Zhang, Rami Albatal, Cathal Gurrin, Alan F. Smeaton
2017 conf
NTCIR
Cathal Gurrin, Hideo Joho, Frank Hopfgartner, Liting Zhou, Duc-Tien Dang-Nguyen, Rashmi Gupta, Rami Albatal
2017 conf
MMM (1)
Zaher Hinbarji, Rami Albatal, Cathal Gurrin
2016 conf
MMM (2)
Jiang Zhou, Rami Albatal, Cathal Gurrin
2016 conf
MMM (1)
Zhenxing Zhang, Rami Albatal, Cathal Gurrin, Alan F. Smeaton
2016 conf
MMM (2)
Zaher Hinbarji, Rami Albatal, Noel E. O'Connor, Cathal Gurrin
2016 A* conf
SIGIR
Cathal Gurrin, Hideo Joho, Frank Hopfgartner, Liting Zhou, Rami Albatal
2016 conf
NTCIR
Cathal Gurrin, Hideo Joho, Frank Hopfgartner, Liting Zhou, Rami Albatal
2016 conf
LTA@MM
Zaher Hinbarji, Moohamad Hinbarji, Rami Albatal, Cathal Gurrin
2016 B conf
ICIP
Ramya Hebbalaguppe, Kevin McGuinness, Jogile Kuklyte, Rami Albatal, Cem Direkoglu, Noel E. O'Connor
2015 conf
MMM (2)
Zaher Hinbarji, Rami Albatal, Cathal Gurrin
2015 conf
MMM (2)
Stefan Terziyski, Rami Albatal, Cathal Gurrin
2015 conf
MMM (2)
Zhenxing Zhang, Rami Albatal, Cathal Gurrin, Alan F. Smeaton
2015 conf
MMM (2)
Jiang Zhou, Aaron Duane, Rami Albatal, Cathal Gurrin, Dag Johansen
2014 conf
MMM (2)
David Scott, Zhenxing Zhang, Rami Albatal, Kevin McGuinness, Esra Acar, Frank Hopfgartner, Cathal Gurrin, Noel E. O'Connor, Alan F. Smeaton
2014 conf
MMM (2)
Rami Albatal, Suzanne Little
2014 conf
VL@COLING
Kevin McGuinness, Feiyan Hu, Rami Albatal, Alan F. Smeaton
2014 conf
TRECVID
Kevin McGuinness, Eva Mohedano, Zhenxing Zhang, Feiyan Hu, Rami Albatal, Cathal Gurrin, Noel E. O'Connor, Alan F. Smeaton, Amaia Salvador, Xavier Giró-i-Nieto, Carles Ventura
2014 A conf
CIKM
TengQi Ye, Brian Moynagh, Rami Albatal, Cathal Gurrin
2013 conf
TRECVID
Suzanne Little, Iveel Jargalsaikhan, Rami Albatal, Cem Direkoglu, Noel E. O'Connor, Alan F. Smeaton, Kathy M. Clawson, Min Jing, Bryan W. Scotney, Hui Wang, Jun Liu, Marcos Nieto, Juan Diego Ortega, Aitor Rodriguez, Iñigo Aramburu, Emmanouil Kafetzakis
2013 C conf
ISTAS
Rami Albatal, Cathal Gurrin, Jiang Zhou, Yang Yang, Denise Carthy, Na Li
2013 conf
TRECVID
Zhenxing Zhang, Rami Albatal, Cathal Gurrin, Alan F. Smeaton
2011 A conf
ICME
Rami Albatal, Philippe Mulhem, Yves Chiaramella
2011 conf
CLEF (Notebook Papers/Labs/Workshop)
Rami Albatal, Bahjat Safadi, Georges Quénot, Philippe Mulhem
2011 conf
CORIA
Rami Albatal, Philippe Mulhem, Yves Chiaramella
2010
Rami Albatal
2010 conf
CLEF (Notebook Papers/LABs/Workshops)
Rami Albatal, Philippe Mulhem
2010 conf
CORIA
Rami Albatal, Philippe Mulhem, Yves Chiaramella
2010 C conf
CBMI
Rami Albatal, Philippe Mulhem, Yves Chiaramella
2009 conf
CLEF (Working Notes)
Philippe Mulhem, Jean-Pierre Chevallet, Georges Quénot, Rami Albatal
2009 conf
CLEF (2)
Trong-Ton Pham, Loïc Maisonnasse, Philippe Mulhem, Jean-Pierre Chevallet, Georges Quénot, Rami Albatal
redb/extractors/pe_extractors/pe_resources.py
← Index redb/extractors/pe_extractors/pe_resources.py python
from hashlib import sha256
import inspect
from datetime import datetime, timezone
from typing import Any

import magic
from magika import Magika
import pefile
from pefile import UnicodeStringWrapperPostProcessor

from redb.extractors.enum import Tag
from redb.extractors.pe_extractor import PEExtractor
from redb.models.dataclasses import PEResource


class PEResourceExtractor(PEExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        pe=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            pe,
        )
        self.elastic_index = self.index_prefix + "-pe_resources"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.PE_RESOURCE.value

    def _extract_resources(self):
        """
        Returns:
        resources: a list of dictionaries, one per each resources type found.
                    each dictionary the key represents the name of the content,
                    which is the value itself.
                    Empty list if no resources present.
        """
        self.log.debug(inspect.currentframe().f_code.co_name)
        resources_list = []
        try:
            if hasattr(self.pe, "DIRECTORY_ENTRY_RESOURCE"):
                for resource_type in self.pe.DIRECTORY_ENTRY_RESOURCE.entries:
                    # if resource_type.name is not None:
                    #     name = resource_type.name
                    # else:
                    #     name = pefile.RESOURCE_TYPE.get(resource_type.struct.Id)
                    # if not name:
                    #     name = resource_type.struct.Id
                    name = (
                        resource_type.name
                        if resource_type.name is not None
                        else pefile.RESOURCE_TYPE.get(resource_type.struct.Id)
                    )
                    if isinstance(name, UnicodeStringWrapperPostProcessor):
                        name = name.decode()
                    try:
                        if hasattr(resource_type, "directory"):
                            for resource_id in resource_type.directory.entries:
                                if hasattr(resource_id, "directory"):
                                    for resource_lang in resource_id.directory.entries:
                                        rsrc_data = self.pe.get_data(
                                            resource_lang.data.struct.OffsetToData,
                                            resource_lang.data.struct.Size,
                                        )
                                        file_type = magic.from_buffer(rsrc_data)
                                        magik = Magika().identify_bytes(rsrc_data).output.label

                                        rsrc_entropy = (
                                            "%.2f"
                                            % pefile.SectionStructure.entropy_H(
                                                self.pe, rsrc_data
                                            )
                                        )
                                        rsrc_sha256 = sha256(rsrc_data).hexdigest()
                                        lang = pefile.LANG.get(
                                            resource_lang.data.lang, "*unknown*"
                                        )
                                        sublang = pefile.get_sublang_name_for_lang(
                                            resource_lang.data.lang,
                                            resource_lang.data.sublang,
                                        )
                                        pe_resource = PEResource(
                                            _id=rsrc_sha256,
                                            resource_type=name,
                                            resource_entropy=rsrc_entropy,
                                            resource_sha256=rsrc_sha256,
                                            resource_filetype=file_type,
                                            resource_magika=magik,
                                            resource_language=lang,
                                            resource_rva=resource_lang.data.struct.OffsetToData,
                                            resource_size=resource_lang.data.struct.Size,
                                            resource_sub_lang=sublang,
                                        )
                                        resources_list.append(pe_resource)
                    except Exception as e:
                        self.log.warning(
                            f"Continue after Error in {self.hash.sha256}: {resource_type.name} "
                            f"Exception: {e}",
                            stack_info=True,
                        )
                        # resources_list.append({f"{e} - {resource_type.name}"})
                        continue
        except Exception as e:
            self.log.exception(
                f"Extract exports error {self.hash.sha256} Exception: {e}"
            )
        self.log.debug(f"Resource list {resources_list}")
        return resources_list

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)
            resources = self._extract_resources()
            # self.export_to_elastic(resources)  # Let the exporters handle this
            return resources
        except Exception as e:
            self.log.error(f"Extract resources error {self.hash.sha256} Exception: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ElasticsearchExporter":
            return self.extract()
        elif exporter_type == "ClickHouseExporter":
            resources = self.extract()
            if resources is None:
                return None
            
            data = []
            current_time = datetime.now(timezone.utc)
            
            for resource in resources:
                data.append([
                    self.sha256,                    # sha256
                    self.md5,                       # md5
                    self.sha1,                      # sha1
                    resource.resource_type,         # resource_type
                    resource.resource_entropy,      # resource_entropy
                    resource.resource_sha256,       # resource_sha256
                    resource.resource_filetype,     # resource_filetype
                    resource.resource_magika,       # resource_magika
                    resource.resource_language,     # resource_language
                    resource.resource_sub_lang,     # resource_sub_lang
                    resource.resource_size,         # resource_size
                    resource.resource_rva,          # resource_rva
                    current_time                    # analysis_date
                ])
            
            column_names = [
                'sha256', 'md5', 'sha1', 'resource_type', 'resource_entropy',
                'resource_sha256', 'resource_filetype', 'resource_magika',
                'resource_language', 'resource_sub_lang', 'resource_size',
                'resource_rva', 'analysis_date'
            ]
            
            if not data:
                return None

            column_type_names = [
                'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                'LowCardinality(Nullable(String))', 'Float64',
                'FixedString(64)', 'LowCardinality(Nullable(String))', 'LowCardinality(Nullable(String))',
                'LowCardinality(Nullable(String))', 'LowCardinality(Nullable(String))', 'UInt64',
                'UInt64', 'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_pe_resources"