Xiangyu Zhao

16 papers Journal 13Unranked 3
YearRankTypeTitle / Venue / Authors
2025 J jnl
Comput. Electron. Agric.
Kaiyi Wang, Feng Yang, Wenqin Bai, Shun Wang, Yong Zhang, Xiangyu Zhao, Zhongqiang Liu, Qiusi Zhang, Jinlong Li, Dongfeng Zhang
2023 J jnl
Comput. Electron. Agric.
Qiusi Zhang, Xiangyu Zhao, Yanyun Han, Feng Yang, Shouhui Pan, Zhongqiang Liu, Kaiyi Wang, Chunjiang Zhao
2022 J jnl
Comput. Electron. Agric.
Qi Zhang, Kaiyi Wang, Yanyun Han, Zhongqiang Liu, Feng Yang, Shufeng Wang, Xiangyu Zhao, Chunjiang Zhao
2022 J jnl
Frontiers Neuroinformatics
Xiangyu Zhao, Xueping Peng, Ke Niu, Hailong Li, Lili He, Feng Yang, Ting Wu, Duo Chen, Qiusi Zhang, Menglin Ouyang, Jiayang Guo, Yijie Pan
2022 J jnl
Comput. Electron. Agric.
Xiangyu Zhao, Shouhui Pan, Zhongqiang Liu, Yanyun Han, Qi Zhang, Kaiyi Wang
2022 J jnl
Comput. Electron. Agric.
Shouhui Pan, Xiangyu Zhao, Yanyun Han, Xiaofeng Wang, Feng Yang, Shu-feng Wang, Zhongqiang Liu, Qiusi Zhang, Qi Zhang, Kaiyi Wang
2020 J jnl
IEEE Access
Xiangyu Zhao, Yanyun Han, Zhongqiang Liu, Shouhui Pan, Kaiyi Wang
2019 J jnl
IEEE Access
Ke Niu, Xiangyu Zhao, Fangfang Li, Ning Li, Xueping Peng, Wei Chen
2018 J jnl
Comput. Electron. Agric.
Yan-yun Han, Kai-yi Wang, Zhongqiang Liu, Shouhui Pan, Xiangyu Zhao, Shu-feng Wang
2017 J jnl
Comput. Electron. Agric.
Yan-yun Han, Kai-yi Wang, Zhongqiang Liu, Qi Zhang, Shouhui Pan, Xiangyu Zhao, Shu-feng Wang
2016 conf
DASFAA Workshops
Xiangyu Zhao, Wei Chen, Feng Yang, Zhongqiang Liu
2016 conf
DASFAA Workshops
Wei Chen, Xiangyu Zhao
2015 J jnl
J. Intell. Inf. Syst.
Xiangyu Zhao, Zhendong Niu, Wei Chen, Chongyang Shi, Ke Niu, Donglei Liu
2014 J jnl
World Wide Web
Wei Chen, Zhendong Niu, Xiangyu Zhao, Yi Li
2013 J jnl
Knowl. Based Syst.
Xiangyu Zhao, Zhendong Niu, Wei Chen
2013 conf
DEXA (2)
Xiangyu Zhao, Zhendong Niu, Wei Chen
redb/extractors/pe_extractors/pe_resources.py
← Index redb/extractors/pe_extractors/pe_resources.py python
from hashlib import sha256
import inspect
from datetime import datetime, timezone
from typing import Any

import magic
from magika import Magika
import pefile
from pefile import UnicodeStringWrapperPostProcessor

from redb.extractors.enum import Tag
from redb.extractors.pe_extractor import PEExtractor
from redb.models.dataclasses import PEResource


class PEResourceExtractor(PEExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        pe=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            pe,
        )
        self.elastic_index = self.index_prefix + "-pe_resources"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.PE_RESOURCE.value

    def _extract_resources(self):
        """
        Returns:
        resources: a list of dictionaries, one per each resources type found.
                    each dictionary the key represents the name of the content,
                    which is the value itself.
                    Empty list if no resources present.
        """
        self.log.debug(inspect.currentframe().f_code.co_name)
        resources_list = []
        try:
            if hasattr(self.pe, "DIRECTORY_ENTRY_RESOURCE"):
                for resource_type in self.pe.DIRECTORY_ENTRY_RESOURCE.entries:
                    # if resource_type.name is not None:
                    #     name = resource_type.name
                    # else:
                    #     name = pefile.RESOURCE_TYPE.get(resource_type.struct.Id)
                    # if not name:
                    #     name = resource_type.struct.Id
                    name = (
                        resource_type.name
                        if resource_type.name is not None
                        else pefile.RESOURCE_TYPE.get(resource_type.struct.Id)
                    )
                    if isinstance(name, UnicodeStringWrapperPostProcessor):
                        name = name.decode()
                    try:
                        if hasattr(resource_type, "directory"):
                            for resource_id in resource_type.directory.entries:
                                if hasattr(resource_id, "directory"):
                                    for resource_lang in resource_id.directory.entries:
                                        rsrc_data = self.pe.get_data(
                                            resource_lang.data.struct.OffsetToData,
                                            resource_lang.data.struct.Size,
                                        )
                                        file_type = magic.from_buffer(rsrc_data)
                                        magik = Magika().identify_bytes(rsrc_data).output.label

                                        rsrc_entropy = (
                                            "%.2f"
                                            % pefile.SectionStructure.entropy_H(
                                                self.pe, rsrc_data
                                            )
                                        )
                                        rsrc_sha256 = sha256(rsrc_data).hexdigest()
                                        lang = pefile.LANG.get(
                                            resource_lang.data.lang, "*unknown*"
                                        )
                                        sublang = pefile.get_sublang_name_for_lang(
                                            resource_lang.data.lang,
                                            resource_lang.data.sublang,
                                        )
                                        pe_resource = PEResource(
                                            _id=rsrc_sha256,
                                            resource_type=name,
                                            resource_entropy=rsrc_entropy,
                                            resource_sha256=rsrc_sha256,
                                            resource_filetype=file_type,
                                            resource_magika=magik,
                                            resource_language=lang,
                                            resource_rva=resource_lang.data.struct.OffsetToData,
                                            resource_size=resource_lang.data.struct.Size,
                                            resource_sub_lang=sublang,
                                        )
                                        resources_list.append(pe_resource)
                    except Exception as e:
                        self.log.warning(
                            f"Continue after Error in {self.hash.sha256}: {resource_type.name} "
                            f"Exception: {e}",
                            stack_info=True,
                        )
                        # resources_list.append({f"{e} - {resource_type.name}"})
                        continue
        except Exception as e:
            self.log.exception(
                f"Extract exports error {self.hash.sha256} Exception: {e}"
            )
        self.log.debug(f"Resource list {resources_list}")
        return resources_list

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)
            resources = self._extract_resources()
            # self.export_to_elastic(resources)  # Let the exporters handle this
            return resources
        except Exception as e:
            self.log.error(f"Extract resources error {self.hash.sha256} Exception: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ElasticsearchExporter":
            return self.extract()
        elif exporter_type == "ClickHouseExporter":
            resources = self.extract()
            if resources is None:
                return None
            
            data = []
            current_time = datetime.now(timezone.utc)
            
            for resource in resources:
                data.append([
                    self.sha256,                    # sha256
                    self.md5,                       # md5
                    self.sha1,                      # sha1
                    resource.resource_type,         # resource_type
                    resource.resource_entropy,      # resource_entropy
                    resource.resource_sha256,       # resource_sha256
                    resource.resource_filetype,     # resource_filetype
                    resource.resource_magika,       # resource_magika
                    resource.resource_language,     # resource_language
                    resource.resource_sub_lang,     # resource_sub_lang
                    resource.resource_size,         # resource_size
                    resource.resource_rva,          # resource_rva
                    current_time                    # analysis_date
                ])
            
            column_names = [
                'sha256', 'md5', 'sha1', 'resource_type', 'resource_entropy',
                'resource_sha256', 'resource_filetype', 'resource_magika',
                'resource_language', 'resource_sub_lang', 'resource_size',
                'resource_rva', 'analysis_date'
            ]
            
            if not data:
                return None

            column_type_names = [
                'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                'LowCardinality(Nullable(String))', 'Float64',
                'FixedString(64)', 'LowCardinality(Nullable(String))', 'LowCardinality(Nullable(String))',
                'LowCardinality(Nullable(String))', 'LowCardinality(Nullable(String))', 'UInt64',
                'UInt64', 'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_pe_resources"