Kai Cong

23 papers A* 2A 6B 4C 2Journal 4Unranked 5
YearRankTypeTitle / Venue / Authors
2025 B conf
IEEE Big Data
Songwen Pei, Yunfeng Chen, Kai Cong, Xing Jia
2025 B conf
IEEE Big Data
Junyi Wang, Songwen Pei, Kai Cong, Joel Rodrigues
2025 B conf
IEEE Big Data
Songwen Pei, Jian Zhang, Kai Cong
2022 J jnl
Knowl. Based Syst.
Kai Cong, Jin Yang, Hongjun Wang, Li Tao
2020 A conf
SANER
Bo Chen, Zhenkun Yang, Li Lei, Kai Cong, Fei Xie
2019 conf
NTCIR
Kai Cong, Wai Lam
2019 conf
ICESS
Bo Chen, Kai Cong, Zhenkun Yang, Qin Wang, Jialu Wang, Li Lei, Fei Xie
2019 J jnl
CoRR
Li Lei, Kai Cong, Zhenkun Yang, Bo Chen, Fei Xie
2018 B conf
FASE
Bo Chen, Christopher Havlicek, Zhenkun Yang, Kai Cong, Raghudeep Kannavara, Fei Xie
2018 conf
ISQED
Bin Lin, Kai Cong, Zhenkun Yang, Zhi-gang Liao, Tao Zhan, Christopher Havlicek, Fei Xie
2017 J jnl
CoRR
Gorka Irazoqui, Kai Cong, Xiaofei Guo, Hareesh Khattri, Arun K. Kanuparthi, Thomas Eisenbarth, Berk Sunar
2016 conf
ASP-DAC
Bin Lin, Zhenkun Yang, Kai Cong, Fei Xie
2016 J jnl
CoRR
Kai Cong, Li Lei, Zhenkun Yang, Fei Xie
2016 A conf
DATE
Zhenkun Yang, Kecheng Hao, Kai Cong, Li Lei, Sandip Ray, Fei Xie
2015 A conf
ISSTA
Kai Cong, Li Lei, Zhenkun Yang, Fei Xie
2014 A conf
DATE
Kai Cong, Li Lei, Zhenkun Yang, Fei Xie
2014 A* conf
DAC
Zhenkun Yang, Kecheng Hao, Kai Cong, Li Lei, Sandip Ray, Fei Xie
2014 A conf
ICCAD
Li Lei, Kai Cong, Zhenkun Yang, Fei Xie
2013 A conf
ICCAD
Kai Cong, Fei Xie, Li Lei
2013 C conf
ICCD
Zhenkun Yang, Kecheng Hao, Kai Cong, Sandip Ray, Fei Xie
2013 C conf
ICCD
Li Lei, Kai Cong, Fei Xie
2013 A* conf
DAC
Li Lei, Fei Xie, Kai Cong
2013 conf
QSIC
Kai Cong, Fei Xie, Li Lei
redb/extractors/macho_extractors/macho_exports.py
← Index redb/extractors/macho_extractors/macho_exports.py python
import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.macho_extractor import MachOExtractor
from redb.models.dataclasses import MachOExport


class MachOExportExtractor(MachOExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        macho=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            macho,
        )
        self.elastic_index = self.index_prefix + "-macho_exports"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.MACHO_EXPORT.value

    def _extract_exports(self, arch_name=None):
        """Extract export information from the MachO binary for a specific architecture."""
        self.log.debug(inspect.currentframe().f_code.co_name)

        if not self.macho:
            return None

        try:
            # Get exported symbols using new API for specific architecture
            exported_symbols = self.macho.get_exported_symbols(arch=arch_name)

            # Extract all exported symbols (may be empty for some binaries)
            # Handle None or empty dict
            if not exported_symbols:
                exported_symbols = {}
            all_symbols = []
            for dylib_name, symbols in exported_symbols.items():
                # Process symbol names
                for symbol in symbols:
                    if isinstance(symbol, bytes):
                        symbol = symbol.decode('utf-8', errors='replace')
                    all_symbols.append(symbol)

            # Create export dataclass
            macho_export = MachOExport(
                macho_exports_total=len(all_symbols),
                macho_export_symbols=all_symbols  # Empty list is fine, but None is not allowed for Array type
            )

            return macho_export

        except Exception as e:
            self.log.error(f"Error extracting MachO exports for arch {arch_name}: {e}")
            return None

    def extract(self):
        self.log.debug(inspect.currentframe().f_code.co_name)
        try:
            # Get architectures (macho is already parsed in base class)
            architectures = self.macho.get_architectures()
            if len(architectures) > 1:
                # FAT binary - return list of exports for each architecture
                results = []
                for arch_name in architectures:
                    exports = self._extract_exports(arch_name)
                    if exports:
                        exports.arch_identifier = arch_name
                        results.append(exports)
                return results
            else:
                # Single architecture - return single result
                return self._extract_exports(architectures[0] if architectures else None)
        except Exception as e:
            self.log.error(f"Error extracting MachO exports: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ElasticsearchExporter":
            return self.extract()
        elif exporter_type == "ClickHouseExporter":
            if not self.macho:
                return None

            # Get architectures (macho is already parsed in base class)
            try:
                architectures = self.macho.get_architectures()
                is_fat = len(architectures) > 1
            except Exception as e:
                self.log.error(f"Could not get architectures: {e}")
                return None

            data = []
            current_time = datetime.now(timezone.utc)

            # Loop through each architecture (1 for single, multiple for FAT)
            for arch_name in architectures:
                # Get architecture-specific sha256
                try:
                    arch_general_info = self.macho.get_general_info(arch=arch_name)
                    arch_header_raw = self.macho.get_macho_header(arch=arch_name)
                    arch_sha256 = arch_general_info.get('SHA256', self.sha256)
                    arch_cputype_raw = arch_header_raw.get('cputype', 0) if arch_header_raw else 0
                except Exception as e:
                    self.log.warning(f"Could not get arch-specific data for {arch_name}: {e}")
                    arch_sha256 = self.sha256
                    arch_cputype_raw = 0

                # Get exports for this architecture
                macho_export = self._extract_exports(arch_name)
                if not macho_export:
                    continue

                data.append([
                    arch_sha256,                          # sha256 (arch-specific)
                    macho_export.macho_exports_total,     # macho_exports_total
                    macho_export.macho_export_symbols,    # macho_export_symbols (keep as array!)
                    current_time,                         # analysis_date
                ])

            column_names = [
                'sha256',
                'macho_exports_total', 'macho_export_symbols',
                'analysis_date'
            ]

            if not data:
                return None

            column_type_names = [
                'FixedString(64)',
                'UInt32', 'Array(Nullable(String))',
                'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

        return None

    def get_clickhouse_table(self) -> str:
        return "redb_macho_exports"