James Chen

29 papers A 3B 2C 1Misc 2Journal 14Unranked 7
YearRankTypeTitle / Venue / Authors
2026 J jnl
CoRR
James Chen, Alan Edelman, John Urschel
2025 J jnl
CoRR
S. M. Sadrul Islam Asif, James Chen, Earl T. Barr, Mark Marron
2025 conf
ACL (1)
Harshit Joshi, Shicheng Liu, James Chen, Larsen Weigle, Monica S. Lam
2025 A conf
ICSME
Junayed Mahmud, James Chen, Terry Achille, Camilo Alvarez-Velez, Darren Dean Bansil, Patrick Ijieh, Samar Karanch, Nadeeshan De Silva, Oscar Chaparro, Andrian Marcus, Kevin Moran
2025 J jnl
CoRR
Junayed Mahmud, James Chen, Terry Achille, Camilo Alvarez-Velez, Darren Dean Bansil, Patrick Ijieh, Samar Karanch, Nadeeshan De Silva, Oscar Chaparro, Andrian Marcus, Kevin Moran
2024 J jnl
CoRR
James Chen, Sheehan Olver
2024 J jnl
CoRR
Harshit Joshi, Shicheng Liu, James Chen, Robert Weigle, Monica S. Lam
2023 conf
SIGCSE (2)
James Chen, Jonathan Calver
2022 B conf
IEEE Big Data
Clay Adams, Malvina Bozhidarova, James Chen, Andrew Gao, Zhengtong Liu, J. Hunter Priniski, Junyuan Lin, Rishi Sonthalia, Andrea L. Bertozzi, P. Jeffrey Brantingham
2021 J jnl
IEEE Trans. Circuits Syst. II Express Briefs
Taejun Lim, Akash Anand, James Chen, Xiaoguang Liu, Yongshik Lee
2019 conf
MIXDES
Arkadiusz Malinowski, James Chen, Shiv Kumar Mishra, Srikanth Samavedam, Dong Kyun Sohn
2018 conf
VLSI-DAT
John R. Hu, James Chen, Boon-khim Liew, Yanfeng Wang, Lianxi Shen, Lin Cong
2017 J jnl
IEEE J. Biomed. Health Informatics
M. Khalid Khan Niazi, Keluo Yao, Debra L. Zynger, Steven K. Clinton, James Chen, Mehmet Koyutürk, Thomas LaFramboise, Metin N. Gurcan
2014 conf
CSE
Yuhan Wang, Qiaochu Li, Tian Lan, James Chen
2014 J jnl
Comput. Animat. Virtual Worlds
Shamima Yasmin, Nan Du, James Chen, Yusheng Feng
2014 conf
CSE
Haiying Huang, Wuyi Zhang, Gaochao Deng, James Chen
2014 conf
CSE
Jingcong Wang, Shuo Zhang, James Chen
2013 Misc conf
AMIA
Arka Pattanayak, James Chen
2011 J jnl
J. Digit. Imaging
James Chen, John Bradshaw, Paul G. Nagy
2011 Misc conf
UbiComp
Udana Bandara, James Chen
2009 J jnl
IEEE Trans. Image Process.
Chun H. Yoon, Bjarni Bödvarsson, Soren Klim, Martin Mørkebjerg, Stig Mortensen, James Chen, Julian R. Maclaren, Pradeep K. Luther, John M. Squire, Philip J. Bones, Rick P. Millane
2008 J jnl
Image Vis. Comput.
Bjarni Bödvarsson, Soren Klim, Martin Mørkebjerg, Stig Mortensen, Chun H. Yoon, James Chen, Julian R. Maclaren, Pradeep K. Luther, John M. Squire, Philip J. Bones, Rick P. Millane
2008 J jnl
Computer
William J. Dally, James D. Balfour, David Black-Schaffer, James Chen, R. Curtis Harting, Vishal Parikh, Jongsoo Park, David Sheffield
2000 A conf
CIKM
James Chen, Shawn R. Wolfe, Stephen D. Wragg
2000 A conf
IUI
Daniel Billsus, Michael J. Pazzani, James Chen
1998 B conf
SSDBM
Joyce C. Niland, James Chen, Wendy Paul, Rebecca Ottesen
1995 J jnl
AT&T Tech. J.
Albert D. Baker, James Chen, Nancy T. Radics, Lee A. Vallone
1993 J jnl
User Model. User Adapt. Interact.
Craig A. Kaplan, Justine Fenwick, James Chen
1992 C conf
IAAI
R. Greg Arbon, Laurie Atkinson, James Chen, Chris A. Guida
redb/extractors/elf_extractors/elf_sections.py
← Index redb/extractors/elf_extractors/elf_sections.py python
import inspect
import hashlib
import math
from collections import Counter
from datetime import datetime, timezone
from typing import Any, List, Dict

from elftools.elf.elffile import ELFFile
from elftools.common.exceptions import ELFError

from redb.extractors.enum import Tag
from redb.extractors.elf_extractor import ELFExtractor
from redb.models.dataclasses import ELFSection


class ELFSectionExtractor(ELFExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        elf=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            elf,
        )
        self.elf_sections = []
        self.elastic_index = self.index_prefix + "-elf_sections"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def _is_empty_result(self, extracted_data) -> bool:
        """
        Override: Empty sections is an ERROR, not a valid empty case.
        A valid ELF file must have sections (at minimum a null section).
        """
        # Always return False - empty sections should be treated as an error
        return False

    def _calculate_entropy(self, data: bytes) -> float:
        """Calculate Shannon entropy of data."""
        if not data:
            return 0.0

        try:
            # Count frequency of each byte
            byte_counts = Counter(data)
            data_len = len(data)

            # Calculate entropy
            entropy = 0.0
            for count in byte_counts.values():
                if count > 0:
                    frequency = count / data_len
                    entropy -= frequency * math.log2(frequency)

            return entropy
        except Exception as e:
            self.log.error(f"Error calculating entropy: {e}")
            return 0.0

    def _map_section_type(self, sh_type_str: str) -> int:
        """Map section type string to enum value."""
        type_map = {
            'SHT_NULL': 0,
            'SHT_PROGBITS': 1,
            'SHT_SYMTAB': 2,
            'SHT_STRTAB': 3,
            'SHT_RELA': 4,
            'SHT_HASH': 5,
            'SHT_DYNAMIC': 6,
            'SHT_NOTE': 7,
            'SHT_NOBITS': 8,
            'SHT_REL': 9,
            'SHT_DYNSYM': 11
        }
        return type_map.get(sh_type_str, 0)

    def _decode_section_flags(self, flags: int) -> List[str]:
        """Decode section flags to human-readable strings."""
        flag_strings = []

        # Common ELF section flags
        if flags & 0x1:  # SHF_WRITE
            flag_strings.append('WRITE')
        if flags & 0x2:  # SHF_ALLOC
            flag_strings.append('ALLOC')
        if flags & 0x4:  # SHF_EXECINSTR
            flag_strings.append('EXECINSTR')
        if flags & 0x10:  # SHF_MERGE
            flag_strings.append('MERGE')
        if flags & 0x20:  # SHF_STRINGS
            flag_strings.append('STRINGS')
        if flags & 0x40:  # SHF_INFO_LINK
            flag_strings.append('INFO_LINK')
        if flags & 0x80:  # SHF_LINK_ORDER
            flag_strings.append('LINK_ORDER')
        if flags & 0x100:  # SHF_OS_NONCONFORMING
            flag_strings.append('OS_NONCONFORMING')
        if flags & 0x200:  # SHF_GROUP
            flag_strings.append('GROUP')
        if flags & 0x400:  # SHF_TLS
            flag_strings.append('TLS')

        return flag_strings if flag_strings else ['NONE']

    def _extract_section_data(self, section) -> Dict:
        """Extract data from a single section."""
        try:
            header = section.header

            # Get section name (handle empty names)
            section_name = section.name if section.name else f"<unnamed_{section.header.get('sh_name', 0)}>"

            # Get section type and map to enum
            sh_type_str = header.get('sh_type', 'SHT_NULL')
            section_type_enum = self._map_section_type(sh_type_str)
            section_type_str = sh_type_str.replace('SHT_', '') if sh_type_str.startswith('SHT_') else sh_type_str

            # Get section properties
            section_flags = header.get('sh_flags', 0)
            section_flags_str = self._decode_section_flags(section_flags)
            section_addr = header.get('sh_addr', 0)
            section_offset = header.get('sh_offset', 0)
            section_size = header.get('sh_size', 0)
            section_link = header.get('sh_link', 0)
            section_info = header.get('sh_info', 0)
            section_addralign = header.get('sh_addralign', 0)
            section_entsize = header.get('sh_entsize', 0)

            # Calculate entropy and hashes for section data
            section_entropy = 0.0
            section_sha256 = ""
            section_md5 = ""

            try:
                if section_size > 0 and section_type_str != 'NOBITS':
                    section_data = section.data()
                    if section_data:
                        # Calculate entropy
                        section_entropy = self._calculate_entropy(section_data)

                        # Calculate hashes
                        section_sha256 = hashlib.sha256(section_data).hexdigest()
                        section_md5 = hashlib.md5(section_data).hexdigest()
            except Exception as e:
                self.log.warning(f"Could not read section '{section_name}' data: {e}")
                section_entropy = 0.0
                section_sha256 = ""
                section_md5 = ""

            return ELFSection(
                section_name=section_name,
                section_type=section_type_enum,
                section_type_str=section_type_str,
                section_flags=section_flags,
                section_flags_str=section_flags_str,
                section_addr=section_addr,
                section_offset=section_offset,
                section_size=section_size,
                section_link=section_link,
                section_info=section_info,
                section_addralign=section_addralign,
                section_entsize=section_entsize,
                section_entropy=section_entropy,
                section_sha256=section_sha256,
                section_md5=section_md5
            )

        except Exception as e:
            self.log.error(f"Error extracting section data: {e}")
            return None

    def tag(self):
        return Tag.ELF_SECTIONS.value if hasattr(Tag, 'ELF_SECTIONS') else "elf_sections"

    def extract(self):
        try:
            self.log.debug(inspect.currentframe().f_code.co_name)

            def extract_data(elf):
                sections_data = []

                # Iterate through all sections with per-section error handling
                for section_index, section in enumerate(elf.iter_sections()):
                    try:
                        section_data = self._extract_section_data(section)
                        if section_data:
                            sections_data.append(section_data)
                        else:
                            self.log.warning(f"Failed to extract data for section {section_index}")
                    except Exception as e:
                        self.log.warning(f"Error processing section {section_index}: {e}")
                        # Continue processing other sections

                return sections_data

            if not self._is_elf_file():
                return None

            result = self._with_elf_file(extract_data)
            if result is None:
                return None

            self.elf_sections = result
            return self.elf_sections

        except Exception as e:
            self.log.error(f"Error extracting ELF sections {self.hash.sha256}: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        self.log.debug(inspect.currentframe().f_code.co_name)

        if exporter_type == "ElasticsearchExporter":
            return self.elf_sections
        elif exporter_type == "ClickHouseExporter":
            try:
                if not self.elf_sections:
                    return None

                # Prepare data arrays for all sections
                data = []
                current_time = datetime.now(timezone.utc)
                for section in self.elf_sections:
                    row = [
                        self.sha256,
                        self.md5,
                        self.sha1,
                        section.section_name,
                        section.section_type,
                        section.section_type_str,
                        section.section_flags,
                        section.section_flags_str,
                        section.section_addr,
                        section.section_offset,
                        section.section_size,
                        section.section_link,
                        section.section_info,
                        section.section_addralign,
                        section.section_entsize,
                        section.section_entropy,
                        section.section_sha256,
                        section.section_md5,
                        current_time
                    ]
                    data.append(row)

                column_names = [
                    'sha256', 'md5', 'sha1',
                    'section_name', 'section_type', 'section_type_str',
                    'section_flags', 'section_flags_str',
                    'section_addr', 'section_offset', 'section_size',
                    'section_link', 'section_info', 'section_addralign', 'section_entsize',
                    'section_entropy', 'section_sha256', 'section_md5',
                    'analysis_date'
                ]

                column_type_names = [
                    'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                    'LowCardinality(String)',
                    "Enum8('NULL'=0, 'PROGBITS'=1, 'SYMTAB'=2, 'STRTAB'=3, 'RELA'=4, 'HASH'=5, 'DYNAMIC'=6, 'NOTE'=7, 'NOBITS'=8, 'REL'=9, 'DYNSYM'=11)",
                    'LowCardinality(String)',
                    'UInt64',
                    'Array(LowCardinality(String))',
                    'UInt64', 'UInt64', 'UInt64',
                    'UInt32', 'UInt32', 'UInt64', 'UInt64',
                    'Float64',
                    'FixedString(64)', 'FixedString(32)',
                    'DateTime64(3, \'UTC\')'
                ]

                if not data:
                    return None

                return (data, column_names, column_type_names)

            except Exception as e:
                self.log.error(f"Error preparing export data: {e}")
                raise

    def get_clickhouse_table(self) -> str:
        return "redb_elf_sections"