Viet-Anh Tran

43 papers A* 5A 14Journal 21Unranked 2
YearRankTypeTitle / Venue / Authors
2025 A conf
RecSys
Viet-Anh Tran, Bruno Sguerra, Gabriel Meseguer-Brocal, Léa Briand, Manuel Moussallam
2025 J jnl
CoRR
Viet-Anh Tran, Bruno Sguerra, Gabriel Meseguer-Brocal, Léa Briand, Manuel Moussallam
2025 J jnl
CoRR
Yuexuan Kong, Viet-Anh Tran, Romain Hennequin
2025 A conf
UMAP
Bruno Sguerra, Viet-Anh Tran, Romain Hennequin, Manuel Moussallam
2025 J jnl
CoRR
Bruno Sguerra, Viet-Anh Tran, Romain Hennequin, Manuel Moussallam
2024 A conf
INTERSPEECH
Yuexuan Kong, Viet-Anh Tran, Romain Hennequin
2024 J jnl
CoRR
Yuexuan Kong, Viet-Anh Tran, Romain Hennequin
2024 A conf
RecSys
Viet-Anh Tran, Guillaume Salha-Galvan, Bruno Sguerra, Romain Hennequin
2024 J jnl
CoRR
Viet-Anh Tran, Guillaume Salha-Galvan, Bruno Sguerra, Romain Hennequin
2023 A* conf
SIGIR
Viet-Anh Tran, Guillaume Salha-Galvan, Bruno Sguerra, Romain Hennequin
2023 J jnl
CoRR
Viet-Anh Tran, Guillaume Salha-Galvan, Bruno Sguerra, Romain Hennequin
2023 conf
ACL (1)
Noé Durandard, Viet-Anh Tran, Gaspard Michel, Elena V. Epure
2023 J jnl
CoRR
Noé Durandard, Viet-Anh Tran, Gaspard Michel, Elena V. Epure
2023 A conf
RecSys
Bruno Sguerra, Viet-Anh Tran, Romain Hennequin
2023 J jnl
CoRR
Bruno Sguerra, Viet-Anh Tran, Romain Hennequin
2022 A conf
RecSys
Bruno Sguerra, Viet-Anh Tran, Romain Hennequin
2022 J jnl
CoRR
Bruno Sguerra, Viet-Anh Tran, Romain Hennequin
2022 A conf
INTERSPEECH
Manh Luong, Viet-Anh Tran
2022 A* conf
ICML
Tam Minh Nguyen, Tan Minh Nguyen, Dung D. Le, Duy Khuong Nguyen, Viet-Anh Tran, Richard G. Baraniuk, Nhat Ho, Stanley J. Osher
2021 A* conf
KDD
Léa Briand, Guillaume Salha-Galvan, Walid Bendada, Mathieu Morlon, Viet-Anh Tran
2021 J jnl
CoRR
Léa Briand, Guillaume Salha-Galvan, Walid Bendada, Mathieu Morlon, Viet-Anh Tran
2021 A conf
RecSys
Guillaume Salha-Galvan, Romain Hennequin, Benjamin Chapus, Viet-Anh Tran, Michalis Vazirgiannis
2021 J jnl
CoRR
Guillaume Salha-Galvan, Romain Hennequin, Benjamin Chapus, Viet-Anh Tran, Michalis Vazirgiannis
2021 J jnl
CoRR
Manh Luong, Viet-Anh Tran
2021 A conf
RecSys
Viet-Anh Tran, Guillaume Salha-Galvan, Romain Hennequin, Manuel Moussallam
2021 J jnl
CoRR
Viet-Anh Tran, Guillaume Salha-Galvan, Romain Hennequin, Manuel Moussallam
2021 A conf
Interspeech
Manh Luong, Viet-Anh Tran
2021 J jnl
CoRR
Manh Luong, Viet-Anh Tran
2021 J jnl
CoRR
Manh-Ha Bui, Viet-Anh Tran, Cuong Pham
2019 A* conf
IJCAI
Guillaume Salha, Romain Hennequin, Viet-Anh Tran, Michalis Vazirgiannis
2019 J jnl
CoRR
Guillaume Salha, Romain Hennequin, Viet-Anh Tran, Michalis Vazirgiannis
2019 A conf
CIKM
Guillaume Salha, Stratis Limnios, Romain Hennequin, Viet-Anh Tran, Michalis Vazirgiannis
2019 J jnl
CoRR
Guillaume Salha, Stratis Limnios, Romain Hennequin, Viet-Anh Tran, Michalis Vazirgiannis
2019 A* conf
SIGIR
Viet-Anh Tran, Romain Hennequin, Jimena Royo-Letelier, Manuel Moussallam
2019 J jnl
CoRR
Viet-Anh Tran, Romain Hennequin, Jimena Royo-Letelier, Manuel Moussallam
2018 conf
ISMIR
Jimena Royo-Letelier, Romain Hennequin, Viet-Anh Tran, Manuel Moussallam
2018 J jnl
CoRR
Jimena Royo-Letelier, Romain Hennequin, Viet-Anh Tran, Manuel Moussallam
2011 A conf
INTERSPEECH
Viet-Anh Tran, Viet Bac Le, Claude Barras, Lori Lamel
2010 J jnl
Speech Commun.
Viet-Anh Tran, Gérard Bailly, Hélène Loevenbruck, Tomoki Toda
2010
Viet-Anh Tran
2009 J jnl
IEICE Electron. Express
Panikos Heracleous, Denis Beautemps, Viet-Anh Tran, Hélène Loevenbruck, Gérard Bailly
2009 A conf
INTERSPEECH
Viet-Anh Tran, Gérard Bailly, Hélène Loevenbruck, Tomoki Toda
2008 A conf
INTERSPEECH
Viet-Anh Tran, Gérard Bailly, Hélène Loevenbruck, Christian Jutten
redb/extractors/pe_extractors/pe_resources.py
← Index redb/extractors/pe_extractors/pe_resources.py python
from hashlib import sha256
import inspect
from datetime import datetime, timezone
from typing import Any

import magic
from magika import Magika
import pefile
from pefile import UnicodeStringWrapperPostProcessor

from redb.extractors.enum import Tag
from redb.extractors.pe_extractor import PEExtractor
from redb.models.dataclasses import PEResource


class PEResourceExtractor(PEExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        pe=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            pe,
        )
        self.elastic_index = self.index_prefix + "-pe_resources"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.PE_RESOURCE.value

    def _extract_resources(self):
        """
        Returns:
        resources: a list of dictionaries, one per each resources type found.
                    each dictionary the key represents the name of the content,
                    which is the value itself.
                    Empty list if no resources present.
        """
        self.log.debug(inspect.currentframe().f_code.co_name)
        resources_list = []
        try:
            if hasattr(self.pe, "DIRECTORY_ENTRY_RESOURCE"):
                for resource_type in self.pe.DIRECTORY_ENTRY_RESOURCE.entries:
                    # if resource_type.name is not None:
                    #     name = resource_type.name
                    # else:
                    #     name = pefile.RESOURCE_TYPE.get(resource_type.struct.Id)
                    # if not name:
                    #     name = resource_type.struct.Id
                    name = (
                        resource_type.name
                        if resource_type.name is not None
                        else pefile.RESOURCE_TYPE.get(resource_type.struct.Id)
                    )
                    if isinstance(name, UnicodeStringWrapperPostProcessor):
                        name = name.decode()
                    try:
                        if hasattr(resource_type, "directory"):
                            for resource_id in resource_type.directory.entries:
                                if hasattr(resource_id, "directory"):
                                    for resource_lang in resource_id.directory.entries:
                                        rsrc_data = self.pe.get_data(
                                            resource_lang.data.struct.OffsetToData,
                                            resource_lang.data.struct.Size,
                                        )
                                        file_type = magic.from_buffer(rsrc_data)
                                        magik = Magika().identify_bytes(rsrc_data).output.label

                                        rsrc_entropy = (
                                            "%.2f"
                                            % pefile.SectionStructure.entropy_H(
                                                self.pe, rsrc_data
                                            )
                                        )
                                        rsrc_sha256 = sha256(rsrc_data).hexdigest()
                                        lang = pefile.LANG.get(
                                            resource_lang.data.lang, "*unknown*"
                                        )
                                        sublang = pefile.get_sublang_name_for_lang(
                                            resource_lang.data.lang,
                                            resource_lang.data.sublang,
                                        )
                                        pe_resource = PEResource(
                                            _id=rsrc_sha256,
                                            resource_type=name,
                                            resource_entropy=rsrc_entropy,
                                            resource_sha256=rsrc_sha256,
                                            resource_filetype=file_type,
                                            resource_magika=magik,
                                            resource_language=lang,
                                            resource_rva=resource_lang.data.struct.OffsetToData,
                                            resource_size=resource_lang.data.struct.Size,
                                            resource_sub_lang=sublang,
                                        )
                                        resources_list.append(pe_resource)
                    except Exception as e:
                        self.log.warning(
                            f"Continue after Error in {self.hash.sha256}: {resource_type.name} "
                            f"Exception: {e}",
                            stack_info=True,
                        )
                        # resources_list.append({f"{e} - {resource_type.name}"})
                        continue
        except Exception as e:
            self.log.exception(
                f"Extract exports error {self.hash.sha256} Exception: {e}"
            )
        self.log.debug(f"Resource list {resources_list}")
        return resources_list

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)
            resources = self._extract_resources()
            # self.export_to_elastic(resources)  # Let the exporters handle this
            return resources
        except Exception as e:
            self.log.error(f"Extract resources error {self.hash.sha256} Exception: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ElasticsearchExporter":
            return self.extract()
        elif exporter_type == "ClickHouseExporter":
            resources = self.extract()
            if resources is None:
                return None
            
            data = []
            current_time = datetime.now(timezone.utc)
            
            for resource in resources:
                data.append([
                    self.sha256,                    # sha256
                    self.md5,                       # md5
                    self.sha1,                      # sha1
                    resource.resource_type,         # resource_type
                    resource.resource_entropy,      # resource_entropy
                    resource.resource_sha256,       # resource_sha256
                    resource.resource_filetype,     # resource_filetype
                    resource.resource_magika,       # resource_magika
                    resource.resource_language,     # resource_language
                    resource.resource_sub_lang,     # resource_sub_lang
                    resource.resource_size,         # resource_size
                    resource.resource_rva,          # resource_rva
                    current_time                    # analysis_date
                ])
            
            column_names = [
                'sha256', 'md5', 'sha1', 'resource_type', 'resource_entropy',
                'resource_sha256', 'resource_filetype', 'resource_magika',
                'resource_language', 'resource_sub_lang', 'resource_size',
                'resource_rva', 'analysis_date'
            ]
            
            if not data:
                return None

            column_type_names = [
                'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                'LowCardinality(Nullable(String))', 'Float64',
                'FixedString(64)', 'LowCardinality(Nullable(String))', 'LowCardinality(Nullable(String))',
                'LowCardinality(Nullable(String))', 'LowCardinality(Nullable(String))', 'UInt64',
                'UInt64', 'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_pe_resources"