Haifan Gong

33 papers A* 1A 1B 1Journal 22Unranked 8
YearRankTypeTitle / Venue / Authors
2026 J jnl
Expert Syst. Appl.
Senmao Wang, Haifan Gong, Runmeng Cui, Boyao Wan, Zhonglin Hu, Haiqing Yang, Jingyang Zhou, Haiyue Jiang, Lin Lin
2025 J jnl
IEEE Trans. Medical Imaging
Wenhao Huang, Haifan Gong, Huan Zhang, Yu Wang, Xiang Wan, Guanbin Li, Haofeng Li, Hong Shen
2025 J jnl
IEEE Trans. Medical Imaging
Haifan Gong, Boyao Wan, Luoyao Kang, Xiang Wan, Lingyan Zhang, Haofeng Li
2025 A* conf
AAAI
Haifan Gong, Yu Lu, Xiang Wan, Haofeng Li
2025 J jnl
IEEE J. Biomed. Health Informatics
Haifan Gong, Huixian Liu, Yitao Wang, Xiaoling Liu, Xiang Wan, Qiao Shi, Haofeng Li
2025 conf
ISBI
Haifan Gong, Luoyao Kang, Yitao Wang, Yihan Wang, Xiang Wan, Xusheng Wu, Haofeng Li
2025 J jnl
Patterns
Haifan Gong, Tianyu Han, Guanbin Li
2024 J jnl
CoRR
Senmao Wang, Haifan Gong, Runmeng Cui, Boyao Wan, Yicheng Liu, Zhonglin Hu, Haiqing Yang, Jingyang Zhou, Bo Pan, Lin Lin, Haiyue Jiang
2024 J jnl
CoRR
Haifan Gong, Yitao Wang, Yihan Wang, Jiashun Xiao, Xiang Wan, Haofeng Li
2024 A conf
ICME
Haifan Gong, Wenhao Huang, Huan Zhang, Yu Wang, Xiang Wan, Hong Shen, Guanbin Li, Haofeng Li
2024 J jnl
CoRR
Haifan Gong, Wenhao Huang, Huan Zhang, Yu Wang, Xiang Wan, Hong Shen, Guanbin Li, Haofeng Li
2024 J jnl
CoRR
Haifan Gong, Luoyao Kang, Yitao Wang, Xiang Wan, Haofeng Li
2023 J jnl
ACM Comput. Surv.
Chenhe Dong, Yinghui Li, Haifan Gong, Miaoxin Chen, Junxin Li, Ying Shen, Min Yang
2023 conf
MICCAI (7)
Zihang Xu, Haifan Gong, Xiang Wan, Haofeng Li
2023 J jnl
CoRR
Zihang Xu, Haifan Gong, Xiang Wan, Haofeng Li
2023 J jnl
Biomed. Signal Process. Control.
Liuliu Xu, Haifan Gong, Yun Zhong, Fan Wang, Shouxin Wang, Lu Lu, Jinru Ding, Chen Zhao, Wenchao Tang, Jie Xu
2023 J jnl
Comput. Biol. Medicine
Haifan Gong, Jiaxin Chen, Guanqi Chen, Haofeng Li, Guanbin Li, Fei Chen
2023 J jnl
Bioinform.
Haifan Gong, Yumeng Zhang, Chenhe Dong, Yue Wang, Guanqi Chen, Bilin Liang, Haofeng Li, Lanxuan Liu, Jie Xu, Guanbin Li
2023 conf
MICCAI (5)
Luoyao Kang, Haifan Gong, Xiang Wan, Haofeng Li
2023 J jnl
CoRR
Luoyao Kang, Haifan Gong, Xiang Wan, Haofeng Li
2022 J jnl
CoRR
Wenhao Huang, Haifan Gong, Huan Zhang, Yu Wang, Haofeng Li, Guanbin Li, Hong Shen
2022 conf
MICCAI (4)
Haifan Gong, Hui Cheng, Yifan Xie, Shuangyi Tan, Guanqi Chen, Fei Chen, Guanbin Li
2022 J jnl
CoRR
Haifan Gong, Hui Cheng, Yifan Xie, Shuangyi Tan, Guanqi Chen, Fei Chen, Guanbin Li
2022 J jnl
BMC Bioinform.
Bilin Liang, Haifan Gong, Lu Lu, Jie Xu
2022 J jnl
IEEE Trans. Medical Imaging
Haifan Gong, Guanqi Chen, Mingzhi Mao, Zhen Li, Guanbin Li
2021 J jnl
CoRR
Chenhe Dong, Yinghui Li, Haifan Gong, Miaoxin Chen, Junxin Li, Ying Shen, Min Yang
2021 B conf
ICMR
Haifan Gong, Guanqi Chen, Sishuo Liu, Yizhou Yu, Guanbin Li
2021 J jnl
CoRR
Haifan Gong, Guanqi Chen, Sishuo Liu, Yizhou Yu, Guanbin Li
2021 conf
ISBI
Haifan Gong, Guanqi Chen, Ranran Wang, Xiang Xie, Mingzhi Mao, Yizhou Yu, Fei Chen, Guanbin Li
2021 conf
CLEF (Working Notes)
Haifan Gong, Ricong Huang, Guanqi Chen, Guanbin Li
2020 conf
CLEF (Working Notes)
Guanqi Chen, Haifan Gong, Guanbin Li
2020 J jnl
IEEE Access
Haifan Gong, Limin Chen, Changhao Li, Jun Zeng, Xichen Tao, Yue Wang
2019 conf
ICMLT
Haifan Gong, Chaoqin Qian, Yue Wang, Jianfeng Yang, Sheng Yi, Zichen Xu
redb/extractors/elf_extractors/elf_notes.py
← Index redb/extractors/elf_extractors/elf_notes.py python
import inspect
import binascii
from datetime import datetime, timezone
from typing import Any, List, Dict

from elftools.elf.elffile import ELFFile
from elftools.common.exceptions import ELFError

from redb.extractors.enum import Tag
from redb.extractors.elf_extractor import ELFExtractor
from redb.models.dataclasses import ELFNote


class ELFNotesExtractor(ELFExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        elf=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            elf,
        )
        self.elf_notes = []
        self.elastic_index = self.index_prefix + "-elf_notes"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def _get_note_type_string(self, note_type: int, note_name: str) -> str:
        """Convert note type number to human-readable string."""

        # GNU-specific note types
        if note_name == "GNU":
            gnu_types = {
                1: "NT_GNU_ABI_TAG",
                2: "NT_GNU_HWCAP",
                3: "NT_GNU_BUILD_ID",
                4: "NT_GNU_GOLD_VERSION",
                5: "NT_GNU_PROPERTY_TYPE_0"
            }
            return gnu_types.get(note_type, f"NT_GNU_UNKNOWN_{note_type}")

        # Generic note types
        generic_types = {
            1: "NT_PRSTATUS",
            2: "NT_FPREGSET",
            3: "NT_PRPSINFO",
            4: "NT_TASKSTRUCT",
            5: "NT_AUXV",
            6: "NT_PSTATUS",
            7: "NT_FPREGS",
            8: "NT_PSINFO",
            9: "NT_PRCRED",
            10: "NT_UTSNAME",
            11: "NT_LWPSTATUS",
            12: "NT_LWPSINFO",
            13: "NT_PRFPXREG"
        }

        return generic_types.get(note_type, f"NT_UNKNOWN_{note_type}")

    def _format_note_description(self, note_desc, note_type: int, note_name: str) -> str:
        """Format note description based on type for human readability."""
        try:
            if not note_desc:
                return ""

            # Handle build ID specifically (common case)
            if note_name == "GNU" and note_type == 3:  # NT_GNU_BUILD_ID
                if isinstance(note_desc, bytes):
                    return binascii.hexlify(note_desc).decode('ascii')
                return str(note_desc)

            # Handle ABI tag
            if note_name == "GNU" and note_type == 1:  # NT_GNU_ABI_TAG
                if isinstance(note_desc, bytes) and len(note_desc) >= 16:
                    # ABI tag contains OS, major, minor, subminor
                    import struct
                    try:
                        os_val, major, minor, subminor = struct.unpack('<IIII', note_desc[:16])
                        os_names = {0: "Linux", 1: "GNU", 2: "Solaris", 3: "FreeBSD"}
                        os_name = os_names.get(os_val, f"OS_{os_val}")
                        return f"{os_name} {major}.{minor}.{subminor}"
                    except:
                        pass

            # For binary data, convert to hex
            if isinstance(note_desc, bytes):
                # Limit size for very large descriptions
                if len(note_desc) > 256:
                    return binascii.hexlify(note_desc[:256]).decode('ascii') + "..."
                return binascii.hexlify(note_desc).decode('ascii')

            # For string data
            if isinstance(note_desc, str):
                return note_desc

            # Fallback
            return str(note_desc)

        except Exception as e:
            self.log.error(f"Error formatting note description: {e}")
            return str(note_desc) if note_desc else ""

    def _extract_note_data(self, note, section_name: str) -> ELFNote:
        """Extract data from a single note entry."""
        try:
            # Get note properties
            note_name = note.get('n_name', '').rstrip('\x00') if note.get('n_name') else ""
            note_type_raw = note.get('n_type', 0)
            note_desc_raw = note.get('n_desc', b'')

            # Handle note_type - pyelftools may return string or int
            if isinstance(note_type_raw, str):
                # pyelftools returned the type as a string like 'NT_GNU_BUILD_ID'
                note_type_str = note_type_raw
                # Map known string types to integers
                note_type_map = {
                    'NT_GNU_ABI_TAG': 1,
                    'NT_GNU_HWCAP': 2,
                    'NT_GNU_BUILD_ID': 3,
                    'NT_GNU_GOLD_VERSION': 4,
                    'NT_GNU_PROPERTY_TYPE_0': 5,
                    'NT_PRSTATUS': 1,
                    'NT_FPREGSET': 2,
                    'NT_PRPSINFO': 3,
                    'NT_TASKSTRUCT': 4,
                    'NT_AUXV': 5,
                    'NT_PSTATUS': 6,
                    'NT_FPREGS': 7,
                    'NT_PSINFO': 8,
                    'NT_PRCRED': 9,
                    'NT_UTSNAME': 10,
                    'NT_LWPSTATUS': 11,
                    'NT_LWPSINFO': 12,
                    'NT_PRFPXREG': 13,
                }
                note_type = note_type_map.get(note_type_raw, 0)
            else:
                note_type = note_type_raw
                # Get human-readable type string
                note_type_str = self._get_note_type_string(note_type, note_name)

            # Format description
            note_desc = self._format_note_description(note_desc_raw, note_type, note_name)

            return ELFNote(
                note_name=note_name,
                note_type=note_type,
                note_type_str=note_type_str,
                note_desc=note_desc,
                note_section=section_name
            )

        except Exception as e:
            self.log.error(f"Error extracting note data: {e}")
            return None

    def _extract_notes_from_sections(self, elf) -> List[Dict]:
        """Extract notes from note sections."""
        notes = []

        try:
            # Look for note sections
            for section in elf.iter_sections():
                if (section.name and
                    section.name.startswith('.note') and
                    hasattr(section, 'iter_notes')):

                    section_name = section.name
                    try:
                        for note in section.iter_notes():
                            note_data = self._extract_note_data(note, section_name)
                            if note_data:
                                notes.append(note_data)
                    except Exception as e:
                        self.log.debug(f"Could not process notes in section {section_name}: {e}")

        except Exception as e:
            self.log.error(f"Error extracting notes from sections: {e}")

        return notes

    def _extract_notes_from_segments(self, elf) -> List[Dict]:
        """Extract notes from PT_NOTE segments."""
        notes = []

        try:
            # Look for PT_NOTE segments
            for segment in elf.iter_segments():
                if segment.header.get('p_type') == 'PT_NOTE':
                    segment_name = f"PT_NOTE_segment_{segment.header.get('p_offset', 0)}"

                    try:
                        if hasattr(segment, 'iter_notes'):
                            for note in segment.iter_notes():
                                note_data = self._extract_note_data(note, segment_name)
                                if note_data:
                                    notes.append(note_data)
                    except Exception as e:
                        self.log.debug(f"Could not process notes in segment: {e}")

        except Exception as e:
            self.log.error(f"Error extracting notes from segments: {e}")

        return notes

    def tag(self):
        return Tag.ELF_NOTES.value if hasattr(Tag, 'ELF_NOTES') else "elf_notes"

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)

            def extract_data(elf):
                all_notes = []

                # Extract notes from note sections
                section_notes = self._extract_notes_from_sections(elf)
                all_notes.extend(section_notes)

                # Extract notes from PT_NOTE segments
                segment_notes = self._extract_notes_from_segments(elf)
                all_notes.extend(segment_notes)

                # Remove duplicates (same note might appear in section and segment)
                unique_notes = []
                seen_notes = set()
                for note in all_notes:
                    note_key = (note.note_name, note.note_type, note.note_desc)
                    if note_key not in seen_notes:
                        seen_notes.add(note_key)
                        unique_notes.append(note)

                return unique_notes

            if not self._is_elf_file():
                return None

            result = self._with_elf_file(extract_data)
            if result is None:
                return None

            self.elf_notes = result
            return self.elf_notes

        except Exception as e:
            self.log.error(f"Error extracting ELF notes {self.hash.sha256}: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        self.log.debug(inspect.currentframe().f_code.co_name)

        if exporter_type == "ElasticsearchExporter":
            return self.elf_notes
        elif exporter_type == "ClickHouseExporter":
            try:
                # Return valid empty structure if no notes found
                # None is reserved for actual errors

                # Prepare data arrays for all notes
                data = []
                current_time = datetime.now(timezone.utc)
                for note in self.elf_notes:
                    row = [
                        self.sha256,
                        self.md5,
                        self.sha1,
                        note.note_name,
                        note.note_type,
                        note.note_type_str,
                        note.note_desc,
                        note.note_section,
                        current_time
                    ]
                    data.append(row)

                column_names = [
                    'sha256', 'md5', 'sha1',
                    'note_name', 'note_type', 'note_type_str',
                    'note_desc', 'note_section',
                    'analysis_date'
                ]

                if not data:
                    return None

                column_type_names = [
                    'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                    'LowCardinality(String)', 'UInt32', 'LowCardinality(String)',
                    'String CODEC(ZSTD(3))', 'LowCardinality(String)',
                    'DateTime64(3, \'UTC\')'
                ]

                return (data, column_names, column_type_names)

            except Exception as e:
                self.log.error(f"Error preparing export data: {e}")
                raise

    def get_clickhouse_table(self) -> str:
        return "redb_elf_notes"