Varun Ganapathi

20 papers A* 3A 1B 1Journal 8Unranked 6
YearRankTypeTitle / Venue / Authors
2022 conf
LOUHI@EMNLP
Byung-Hak Kim, Zhongfen Deng, Philip S. Yu, Varun Ganapathi
2022 J jnl
CoRR
Byung-Hak Kim, Zhongfen Deng, Philip S. Yu, Varun Ganapathi
2022 J jnl
CoRR
Weiyao Wang, Byung-Hak Kim, Varun Ganapathi
2021 conf
MLHC
Byung-Hak Kim, Varun Ganapathi
2021 J jnl
CoRR
Byung-Hak Kim, Varun Ganapathi
2020 J jnl
CoRR
Byung-Hak Kim, Seshadri Sridharan, Andy Atwal, Varun Ganapathi
2019 J jnl
CoRR
Byung-Hak Kim, Varun Ganapathi
2018 J jnl
CoRR
Byung-Hak Kim, Ethan Vizitei, Varun Ganapathi
2018 B conf
EDM
Byung-Hak Kim, Ethan Vizitei, Varun Ganapathi
2018 J jnl
CoRR
Byung-Hak Kim, Ethan Vizitei, Varun Ganapathi
2014
Varun Ganapathi
2012 J jnl
CoRR
Varun Ganapathi, David Vickrey, John C. Duchi, Daphne Koller
2012 conf
ECCV (6)
Varun Ganapathi, Christian Plagemann, Daphne Koller, Sebastian Thrun
2011 A* conf
ICRA
Ellen Klingbeil, Deepak Rao, Blake Carpenter, Varun Ganapathi, Andrew Y. Ng, Oussama Khatib
2010 A* conf
CVPR
Varun Ganapathi, Christian Plagemann, Daphne Koller, Sebastian Thrun
2010 A* conf
ICRA
Christian Plagemann, Varun Ganapathi, Daphne Koller, Sebastian Thrun
2008 A conf
UAI
Varun Ganapathi, David Vickrey, John C. Duchi, Daphne Koller
2006 conf
NIPS
Su-In Lee, Varun Ganapathi, Daphne Koller
2005 conf
NIPS
Pieter Abbeel, Varun Ganapathi, Andrew Y. Ng
2004 conf
ISER
Andrew Y. Ng, Adam Coates, Mark Diel, Varun Ganapathi, Jamie Schulte, Ben Tse, Eric Berger, Eric Liang
redb/extractors/macho_extractors/macho_dylibs.py
← Index redb/extractors/macho_extractors/macho_dylibs.py python
import hashlib
import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.macho_extractor import MachOExtractor
from redb.models.dataclasses import MachODylib


class MachODylibExtractor(MachOExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        macho=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            macho,
        )
        self.elastic_index = self.index_prefix + "-macho_dylibs"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.MACHO_DYLIB.value

    def _extract_dylibs(self):
        """Extract dynamic library information from all architectures in the MachO binary."""
        self.log.debug(inspect.currentframe().f_code.co_name)
        dylibs = []

        if not self.macho:
            return dylibs

        try:
            # Get architectures using new API (already parsed in base class)
            architectures = self.macho.get_architectures()
            if not architectures:
                return dylibs

            # Process each architecture
            for arch_name in architectures:
                # Get dylib commands using new API with architecture parameter
                dylib_commands = self.macho.get_dylib_commands(arch=arch_name)
                if not dylib_commands:
                    continue

                # Extract dylib commands for this architecture
                for dylib_cmd in dylib_commands:
                    try:
                        dylib_name = dylib_cmd.get('dylib_name', 'Unknown')
                        if isinstance(dylib_name, bytes):
                            dylib_name = dylib_name.decode('utf-8', errors='replace')

                        # Create dylib dataclass with architecture info
                        macho_dylib = MachODylib(
                            dylib_name=dylib_name,
                            dylib_timestamp=dylib_cmd.get('dylib_timestamp', 0),
                            dylib_current_version=dylib_cmd.get('dylib_current_version', 0),
                            dylib_compat_version=dylib_cmd.get('dylib_compat_version', 0),
                        )
                        # Add architecture info to the dylib
                        macho_dylib.architecture = arch_name
                        dylibs.append(macho_dylib)

                    except Exception as e:
                        self.log.warning(
                            f'Unable to process dylib "{dylib_cmd.get("dylib_name", "Unknown")}" for architecture {arch_name} in {self.hash.sha256}: {e}'
                        )
                        continue

            return dylibs

        except Exception as e:
            self.log.error(f"Error extracting MachO dylibs: {e}")
            return dylibs

    def extract(self):
        self.log.debug(inspect.currentframe().f_code.co_name)
        try:
            dylibs = self._extract_dylibs()
            return dylibs
        except Exception as e:
            self.log.error(f"Error extracting MachO dylibs: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ElasticsearchExporter":
            return self.extract()
        elif exporter_type == "ClickHouseExporter":
            if not self.macho:
                return None

            # Get architectures (macho is already parsed in base class)
            try:
                architectures = self.macho.get_architectures()
                is_fat = len(architectures) > 1
            except Exception as e:
                self.log.error(f"Could not get architectures: {e}")
                return None

            data = []
            current_time = datetime.now(timezone.utc)

            # Loop through each architecture (1 for single, multiple for FAT)
            for arch_name in architectures:
                # Get architecture-specific sha256
                try:
                    arch_general_info = self.macho.get_general_info(arch=arch_name)
                    arch_header_raw = self.macho.get_macho_header(arch=arch_name)
                    arch_sha256 = arch_general_info.get('SHA256', self.sha256)
                    arch_cputype_raw = arch_header_raw.get('cputype', 0) if arch_header_raw else 0
                except Exception as e:
                    self.log.warning(f"Could not get arch-specific data for {arch_name}: {e}")
                    arch_sha256 = self.sha256
                    arch_cputype_raw = 0

                # Get dylib commands for this architecture
                dylib_commands = self.macho.get_dylib_commands(arch=arch_name)
                if not dylib_commands:
                    continue

                # Process each dylib for this architecture
                for dylib_cmd in dylib_commands:
                    try:
                        dylib_name = dylib_cmd.get('dylib_name', 'Unknown')
                        if isinstance(dylib_name, bytes):
                            dylib_name = dylib_name.decode('utf-8', errors='replace')

                        data.append([
                            arch_sha256,                          # sha256 (architecture-specific)
                            dylib_name,                           # dylib_name
                            dylib_cmd.get('dylib_timestamp', 0), # dylib_timestamp
                            dylib_cmd.get('dylib_current_version', 0), # dylib_current_version
                            dylib_cmd.get('dylib_compat_version', 0),  # dylib_compat_version
                            current_time,                         # analysis_date
                        ])
                    except Exception as e:
                        self.log.warning(
                            f'Unable to process dylib "{dylib_cmd.get("dylib_name", "Unknown")}" for architecture {arch_name}: {e}'
                        )
                        continue

            column_names = [
                'sha256',
                'dylib_name', 'dylib_timestamp', 'dylib_current_version',
                'dylib_compat_version', 'analysis_date'
            ]

            if not data:
                return None

            column_type_names = [
                'FixedString(64)',
                'String', 'UInt32', 'UInt32',
                'UInt32', 'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

        return None

    def get_clickhouse_table(self) -> str:
        return "redb_macho_dylibs"