Xiaofei Wang

28 papers A 6C 1Misc 2Journal 13Unranked 6
YearRankTypeTitle / Venue / Authors
2025 A conf
INTERSPEECH
Aswin Shanmugam Subramanian, Amit Das, Naoyuki Kanda, Jinyu Li, Xiaofei Wang, Yifan Gong
2024 J jnl
IEEE ACM Trans. Audio Speech Lang. Process.
Xiaofei Wang, Manthan Thakker, Zhuo Chen, Naoyuki Kanda, Sefik Emre Eskimez, Sanyuan Chen, Min Tang, Shujie Liu, Jinyu Li, Takuya Yoshioka
2020 Misc conf
ICASSP
Ruizhi Li, Gregory Sell, Xiaofei Wang, Shinji Watanabe, Hynek Hermansky
2020 J jnl
IEEE ACM Trans. Audio Speech Lang. Process.
Ruizhi Li, Xiaofei Wang, Sri Harish Mallidi, Shinji Watanabe, Takaaki Hori, Hynek Hermansky
2019 C conf
ASRU
Shigeki Karita, Xiaofei Wang, Shinji Watanabe, Takenori Yoshimura, Wangyou Zhang, Nanxin Chen, Tomoki Hayashi, Takaaki Hori, Hirofumi Inaguma, Ziyan Jiang, Masao Someki, Nelson Enrique Yalta Soplin, Ryuichi Yamamoto
2019 J jnl
CoRR
Shigeki Karita, Nanxin Chen, Tomoki Hayashi, Takaaki Hori, Hirofumi Inaguma, Ziyan Jiang, Masao Someki, Nelson Enrique Yalta Soplin, Ryuichi Yamamoto, Xiaofei Wang, Shinji Watanabe, Takenori Yoshimura, Wangyou Zhang
2019 J jnl
CoRR
Ruizhi Li, Gregory Sell, Xiaofei Wang, Shinji Watanabe, Hynek Hermansky
2019 J jnl
CoRR
Aswin Shanmugam Subramanian, Xiaofei Wang, Shinji Watanabe, Toru Taniguchi, Dung T. Tran, Yuya Fujita
2019 A conf
INTERSPEECH
Xiaofei Wang, Jinyi Yang, Ruizhi Li, Samik Sadhu, Hynek Hermansky
2019 J jnl
CoRR
Xiaofei Wang, Jinyi Yang, Ruizhi Li, Samik Sadhu, Hynek Hermansky
2019 conf
WASPAA
Toru Taniguchi, Aswin Shanmugam Subramanian, Xiaofei Wang, Dung T. Tran, Yuya Fujita, Shinji Watanabe
2019 J jnl
CoRR
Ruizhi Li, Xiaofei Wang, Sri Harish Mallidi, Shinji Watanabe, Takaaki Hori, Hynek Hermansky
2019 conf
WASPAA
Aswin Shanmugam Subramanian, Xiaofei Wang, Murali Karthick Baskar, Shinji Watanabe, Toru Taniguchi, Dung T. Tran, Yuya Fujita
2019 Misc conf
ICASSP
Xiaofei Wang, Ruizhi Li, Sri Harish Mallidi, Takaaki Hori, Shinji Watanabe, Hynek Hermansky
2018 J jnl
CoRR
Ruizhi Li, Xiaofei Wang, Sri Harish Reddy Mallidi, Takaaki Hori, Shinji Watanabe, Hynek Hermansky
2018 A conf
INTERSPEECH
Xiaofei Wang, Ruizhi Li, Hynek Hermansky
2018 J jnl
CoRR
Xiaofei Wang, Ruizhi Li, Sri Harish Mallidi, Takaaki Hori, Shinji Watanabe, Hynek Hermansky
2017 J jnl
CoRR
Xiaofei Wang, Yonghong Yan, Hynek Hermansky
2016 A conf
INTERSPEECH
Ziteng Wang, Xu Li, Xiaofei Wang, Qiang Fu, Yonghong Yan
2016 A conf
INTERSPEECH
Yanmeng Guo, Xiaofei Wang, Chao Wu, Qiang Fu, Ning Ma, Guy J. Brown
2016 A conf
INTERSPEECH
Xu Li, Ziteng Wang, Xiaofei Wang, Qiang Fu, Yonghong Yan
2016 conf
IWAENC
Xu Li, Xiaofei Wang, Qiang Fu, Yonghong Yan
2016 conf
IWAENC
Ziteng Wang, Xiaofei Wang, Xu Li, Qiang Fu, Yonghong Yan
2016 J jnl
Circuits Syst. Signal Process.
Chao Wu, Xiaofei Wang, Yanmeng Guo, Qiang Fu, Yonghong Yan
2015 J jnl
Speech Commun.
Xiaofei Wang, Yanmeng Guo, Chao Wu, Qiang Fu, Yonghong Yan
2015 J jnl
CoRR
Xiaofei Wang, Chao Wu, Pengyuan Zhang, Ziteng Wang, Yong Liu, Xu Li, Qiang Fu, Yonghong Yan
2015 conf
ChinaSIP
Xiaofei Wang, Xu Li, Yanmeng Guo, Qiang Fu, Yonghong Yan
2014 conf
ChinaSIP
Xiaofei Wang, Yanmeng Guo, Qiang Fu, Yonghong Yan
redb/extractors/pe_extractors/pe_resources.py
← Index redb/extractors/pe_extractors/pe_resources.py python
from hashlib import sha256
import inspect
from datetime import datetime, timezone
from typing import Any

import magic
from magika import Magika
import pefile
from pefile import UnicodeStringWrapperPostProcessor

from redb.extractors.enum import Tag
from redb.extractors.pe_extractor import PEExtractor
from redb.models.dataclasses import PEResource


class PEResourceExtractor(PEExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        pe=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            pe,
        )
        self.elastic_index = self.index_prefix + "-pe_resources"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.PE_RESOURCE.value

    def _extract_resources(self):
        """
        Returns:
        resources: a list of dictionaries, one per each resources type found.
                    each dictionary the key represents the name of the content,
                    which is the value itself.
                    Empty list if no resources present.
        """
        self.log.debug(inspect.currentframe().f_code.co_name)
        resources_list = []
        try:
            if hasattr(self.pe, "DIRECTORY_ENTRY_RESOURCE"):
                for resource_type in self.pe.DIRECTORY_ENTRY_RESOURCE.entries:
                    # if resource_type.name is not None:
                    #     name = resource_type.name
                    # else:
                    #     name = pefile.RESOURCE_TYPE.get(resource_type.struct.Id)
                    # if not name:
                    #     name = resource_type.struct.Id
                    name = (
                        resource_type.name
                        if resource_type.name is not None
                        else pefile.RESOURCE_TYPE.get(resource_type.struct.Id)
                    )
                    if isinstance(name, UnicodeStringWrapperPostProcessor):
                        name = name.decode()
                    try:
                        if hasattr(resource_type, "directory"):
                            for resource_id in resource_type.directory.entries:
                                if hasattr(resource_id, "directory"):
                                    for resource_lang in resource_id.directory.entries:
                                        rsrc_data = self.pe.get_data(
                                            resource_lang.data.struct.OffsetToData,
                                            resource_lang.data.struct.Size,
                                        )
                                        file_type = magic.from_buffer(rsrc_data)
                                        magik = Magika().identify_bytes(rsrc_data).output.label

                                        rsrc_entropy = (
                                            "%.2f"
                                            % pefile.SectionStructure.entropy_H(
                                                self.pe, rsrc_data
                                            )
                                        )
                                        rsrc_sha256 = sha256(rsrc_data).hexdigest()
                                        lang = pefile.LANG.get(
                                            resource_lang.data.lang, "*unknown*"
                                        )
                                        sublang = pefile.get_sublang_name_for_lang(
                                            resource_lang.data.lang,
                                            resource_lang.data.sublang,
                                        )
                                        pe_resource = PEResource(
                                            _id=rsrc_sha256,
                                            resource_type=name,
                                            resource_entropy=rsrc_entropy,
                                            resource_sha256=rsrc_sha256,
                                            resource_filetype=file_type,
                                            resource_magika=magik,
                                            resource_language=lang,
                                            resource_rva=resource_lang.data.struct.OffsetToData,
                                            resource_size=resource_lang.data.struct.Size,
                                            resource_sub_lang=sublang,
                                        )
                                        resources_list.append(pe_resource)
                    except Exception as e:
                        self.log.warning(
                            f"Continue after Error in {self.hash.sha256}: {resource_type.name} "
                            f"Exception: {e}",
                            stack_info=True,
                        )
                        # resources_list.append({f"{e} - {resource_type.name}"})
                        continue
        except Exception as e:
            self.log.exception(
                f"Extract exports error {self.hash.sha256} Exception: {e}"
            )
        self.log.debug(f"Resource list {resources_list}")
        return resources_list

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)
            resources = self._extract_resources()
            # self.export_to_elastic(resources)  # Let the exporters handle this
            return resources
        except Exception as e:
            self.log.error(f"Extract resources error {self.hash.sha256} Exception: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ElasticsearchExporter":
            return self.extract()
        elif exporter_type == "ClickHouseExporter":
            resources = self.extract()
            if resources is None:
                return None
            
            data = []
            current_time = datetime.now(timezone.utc)
            
            for resource in resources:
                data.append([
                    self.sha256,                    # sha256
                    self.md5,                       # md5
                    self.sha1,                      # sha1
                    resource.resource_type,         # resource_type
                    resource.resource_entropy,      # resource_entropy
                    resource.resource_sha256,       # resource_sha256
                    resource.resource_filetype,     # resource_filetype
                    resource.resource_magika,       # resource_magika
                    resource.resource_language,     # resource_language
                    resource.resource_sub_lang,     # resource_sub_lang
                    resource.resource_size,         # resource_size
                    resource.resource_rva,          # resource_rva
                    current_time                    # analysis_date
                ])
            
            column_names = [
                'sha256', 'md5', 'sha1', 'resource_type', 'resource_entropy',
                'resource_sha256', 'resource_filetype', 'resource_magika',
                'resource_language', 'resource_sub_lang', 'resource_size',
                'resource_rva', 'analysis_date'
            ]
            
            if not data:
                return None

            column_type_names = [
                'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                'LowCardinality(Nullable(String))', 'Float64',
                'FixedString(64)', 'LowCardinality(Nullable(String))', 'LowCardinality(Nullable(String))',
                'LowCardinality(Nullable(String))', 'LowCardinality(Nullable(String))', 'UInt64',
                'UInt64', 'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_pe_resources"