Carlos Caldas

14 papers B 1Journal 11Unranked 2
YearRankTypeTitle / Venue / Authors
2021 J jnl
Comput. Methods Programs Biomed.
Maureen van Eijnatten, Leonardo Rundo, Kees Joost Batenburg, Felix Lucka, Emma Beddowes, Carlos Caldas, Ferdia A. Gallagher, Evis Sala, Carola-Bibiane Schönlieb, Ramona Woitek
2020 J jnl
CoRR
Maureen van Eijnatten, Leonardo Rundo, Kees Joost Batenburg, Felix Lucka, Emma Beddowes, Carlos Caldas, Ferdia A. Gallagher, Evis Sala, Carola-Bibiane Schönlieb, Ramona Woitek
2019 J jnl
Nat.
Oscar M. Rueda, Stephen John Sammut, José A. Seoane, Suet-Feung Chin, Jennifer L. Caswell-Jin, Maurizio Callari, Rajbir Nath Batra, Bernard Pereira, Alejandra Bruna, H. Raza Ali, Elena Provenzano, Bin Liu, Michelle Parisien, Cheryl Gillett, Steven McKinney, Andrew R. Green, Leigh Murphy, Arnie Purushotham, Ian O. Ellis, Paul D. P. Pharoah, Cristina Rueda, Samuel Aparicio, Carlos Caldas, Christina Curtis
2015 J jnl
PLoS Comput. Biol.
Christopher R. S. Banerji, Simone Severini, Carlos Caldas, Andrew E. Teschendorff
2013 J jnl
BMC Bioinform.
Lorna Morris, Andrew Tsui, Charles Crichton, Steve Harris, Peter Maccallum, William J. Howat, Jim Davies, James D. Brenton, Carlos Caldas
2013 J jnl
PLoS Comput. Biol.
Erhan Bilal, Janusz Dutkowski, Justin Guinney, In Sock Jang, Benjamin A. Logsdon, Gaurav Pandey, Benjamin A. Sauerwine, Yishai Shimoni, Hans Kristian Moen Vollan, Brigham H. Mecham, Oscar M. Rueda, Jorg Tost, Christina Curtis, Mariano J. Alvarez, Vessela N. Kristensen, Samuel Aparicio, Anne-Lise Børresen-Dale, Carlos Caldas, Andrea Califano, Stephen H. Friend, Trey Ideker, Eric E. Schadt, Gustavo A. Stolovitzky, Adam A. Margolin
2012 J jnl
IEEE ACM Trans. Comput. Biol. Bioinform.
Yinyin Yuan, Christina Curtis, Carlos Caldas, Florian Markowetz
2012 J jnl
Bioinform.
Ruijie Liu, Ana-Teresa Maia, Roslin Russell, Carlos Caldas, Bruce A. Ponder, Matthew E. Ritchie
2010 conf
BIBM
Yinyin Yuan, Christina Curtis, Carlos Caldas, Florian Markowetz
2008 conf
eScience
Tianyi Zang, Radu Calinescu, Steve Harris, Andrew Tsui, Charles Crichton, Marta Z. Kwiatkowska, Jeremy Gibbons, Jim Davies, James D. Brenton, Carlos Caldas
2008 B conf
CCGRID
Tianyi Zang, Radu Calinescu, Steve Harris, Andrew Tsui, Marta Z. Kwiatkowska, Jeremy Gibbons, Jim Davies, Peter Maccallum, Carlos Caldas
2007 J jnl
PLoS Comput. Biol.
Andrew E. Teschendorff, Michel Journée, Pierre-Antoine Absil, Rodolphe Sepulchre, Carlos Caldas
2006 J jnl
Bioinform.
Andrew E. Teschendorff, Ali Naderi, Nuno L. Barbosa-Morais, Carlos Caldas
2005 J jnl
Bioinform.
Andrew E. Teschendorff, Yanzhong Wang, Nuno L. Barbosa-Morais, James D. Brenton, Carlos Caldas
redb/extractors/macho_extractors/macho_similarity_hashes.py
← Index redb/extractors/macho_extractors/macho_similarity_hashes.py python
import inspect
from datetime import datetime, timezone
from typing import Any

from redb.extractors.enum import Tag
from redb.extractors.macho_extractor import MachOExtractor


class MachOSimilarityHashExtractor(MachOExtractor):
    """Extract Mach-O similarity hashes using machofile API.

    Similarity hashes are MD5 fingerprints of sorted, deduplicated binary components:
    - dylib_hash: MD5 of dynamic library names
    - import_hash: MD5 of imported function names
    - export_hash: MD5 of exported symbol names
    - entitlement_hash: MD5 of entitlement names and array values
    - symhash: MD5 of external undefined symbols

    For FAT binaries:
    - Inserts one row per architecture slice with per-slice hashes
    - Inserts one row for the FAT container with combined hashes

    For single-arch binaries:
    - Inserts one row with that architecture's hashes

    Note: parent_sha256 and architecture relationships are tracked in redb_basic_properties,
    not duplicated here. Use JOIN with redb_basic_properties when needed.
    """

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        macho=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            macho,
        )
        self.elastic_index = self.index_prefix + "-macho_hashes"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.MACHO_HASHES.value

    def _extract_similarity_hashes(self, arch_name=None):
        """Extract similarity hashes for a specific architecture."""
        self.log.debug(inspect.currentframe().f_code.co_name)

        if not self.macho:
            return None

        try:
            similarity_hashes = self.macho.get_similarity_hashes(arch=arch_name)
            return similarity_hashes if similarity_hashes else None
        except Exception as e:
            self.log.error(f"Error extracting similarity hashes for arch {arch_name}: {e}")
            return None

    def extract(self):
        self.log.debug(inspect.currentframe().f_code.co_name)
        try:
            if not self.macho:
                return None

            architectures = self.macho.get_architectures()
            if not architectures:
                return None

            if len(architectures) > 1:
                # FAT binary - return combined hashes + per-arch hashes
                results = []

                # First add combined hashes for the FAT container
                all_hashes = self.macho.get_similarity_hashes()
                combined_hashes = all_hashes.get('combined', {}) if all_hashes else {}
                if combined_hashes:
                    combined_hashes['arch_identifier'] = 'fat'
                    results.append(combined_hashes)

                # Then add per-arch hashes
                for arch_name in architectures:
                    hashes = self._extract_similarity_hashes(arch_name)
                    if hashes:
                        hashes['arch_identifier'] = arch_name
                        results.append(hashes)
                return results
            else:
                # Single architecture - return single result
                return self._extract_similarity_hashes(architectures[0])
        except Exception as e:
            self.log.error(f"Error extracting similarity hashes: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ElasticsearchExporter":
            return self.extract()
        elif exporter_type == "ClickHouseExporter":
            if not self.macho:
                return None

            try:
                architectures = self.macho.get_architectures()
                is_fat = len(architectures) > 1
            except Exception as e:
                self.log.error(f"Could not get architectures: {e}")
                return None

            data = []
            current_time = datetime.now(timezone.utc)

            # For FAT binaries, first insert a row for the container with combined hashes
            if is_fat:
                all_hashes = self.macho.get_similarity_hashes()  # Without arch returns all including 'combined'
                combined_hashes = all_hashes.get('combined', {}) if all_hashes else {}
                if combined_hashes:
                    data.append([
                        self.sha256,                                    # sha256 (FAT container)
                        combined_hashes.get('dylib_hash'),              # dylib_hash
                        combined_hashes.get('import_hash'),             # import_hash
                        combined_hashes.get('export_hash'),             # export_hash
                        combined_hashes.get('entitlement_hash'),        # entitlement_hash
                        combined_hashes.get('symhash'),                 # symhash
                        current_time,                                   # analysis_date
                    ])

            # Insert rows for each architecture slice
            for arch_name in architectures:
                # Get architecture-specific sha256
                try:
                    arch_general_info = self.macho.get_general_info(arch=arch_name)
                    arch_sha256 = arch_general_info.get('SHA256', self.sha256)
                except Exception as e:
                    self.log.warning(f"Could not get arch-specific sha256 for {arch_name}: {e}")
                    arch_sha256 = self.sha256

                # Get similarity hashes for this architecture
                similarity_hashes = self._extract_similarity_hashes(arch_name)
                if not similarity_hashes:
                    continue

                data.append([
                    arch_sha256,                                    # sha256 (arch-specific)
                    similarity_hashes.get('dylib_hash'),            # dylib_hash
                    similarity_hashes.get('import_hash'),           # import_hash
                    similarity_hashes.get('export_hash'),           # export_hash
                    similarity_hashes.get('entitlement_hash'),      # entitlement_hash
                    similarity_hashes.get('symhash'),               # symhash
                    current_time,                                   # analysis_date
                ])

            if not data:
                return None

            column_names = [
                'sha256',
                'macho_dylib_hash', 'macho_import_hash', 'macho_export_hash',
                'macho_entitlement_hash', 'macho_symhash',
                'analysis_date'
            ]

            column_type_names = [
                'FixedString(64)',
                'Nullable(FixedString(32))', 'Nullable(FixedString(32))', 'Nullable(FixedString(32))',
                'Nullable(FixedString(32))', 'Nullable(FixedString(32))',
                'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

        return None

    def get_clickhouse_table(self) -> str:
        return "redb_hashes"