Cem Aksoy

11 papers B 1Journal 3Unranked 7
YearRankTypeTitle / Venue / Authors
2017 J jnl
J. Web Eng.
Ananya Dass, Cem Aksoy, Aggeliki Dimitriou, Dimitri Theodoratos
2016 conf
WAIM (2)
Cem Aksoy, Ananya Dass, Dimitri Theodoratos, Xiaoying Wu
2016 conf
WISE (1)
Ananya Dass, Cem Aksoy, Aggeliki Dimitriou, Dimitri Theodoratos, Xiaoying Wu
2015 conf
WISE (2)
Ananya Dass, Aggeliki Dimitriou, Cem Aksoy, Dimitri Theodoratos
2015 B conf
ICWE
Ananya Dass, Cem Aksoy, Aggeliki Dimitriou, Dimitri Theodoratos
2015 J jnl
VLDB J.
Cem Aksoy, Aggeliki Dimitriou, Dimitri Theodoratos
2014 conf
WAIM
Cem Aksoy, Ananya Dass, Dimitri Theodoratos, Xiaoying Wu
2014 conf
WISE (1)
Ananya Dass, Cem Aksoy, Aggeliki Dimitriou, Dimitri Theodoratos
2012 J jnl
J. Assoc. Inf. Sci. Technol.
Cem Aksoy, Fazli Can, Seyit Kocberber
2009 conf
ISCIS
Cem Aksoy, Ahmet Bugdayci, Tunay Gur, Ibrahim Uysal, Fazli Can
2007 conf
TRECVID
Sercan Aksoy, Pinar Duygulu, Cem Aksoy, E. Aydin, D. Günaydin, K. Hadimh, L. Koç, Y. Olgun, Cihan Orhan, G. Yakin
redb/extractors/decompiler/bninja/analysis/strings.py
← Index redb/extractors/decompiler/bninja/analysis/strings.py python
from collections import Counter
import math

class StringAnalysis:
    def __init__(self, bv, functions):
        self.bv = bv
        self.functions = functions

    def entropy(self, s: str) -> float:
        """Compute Shannon entropy of a string."""
        if not s:
            return 0.0
        freq = Counter(s)
        length = len(s)
        return -sum((count / length) * math.log2(count / length) for count in freq.values())

    def analyze(self):
        """
        Extract unique strings from the binary.

        Deduplicates by (string, encoding) within the same binary, keeping the
        first occurrence (lowest offset). Cross-binary deduplication and
        aggregation is handled by ClickHouse materialized views.
        """
        strings = {}

        # Sort strings by their starting address
        sorted_entries = sorted(self.bv.strings, key=lambda e: e.start)

        for entry in sorted_entries:
            # Key is the string and its encoding
            key = (entry.value, entry.type.name)

            # Skip if this string (value + encoding) was already added.
            # Because entries are sorted by address, the first one is always kept.
            if key in strings:
                continue

            # Store only the first occurrence with schema-matching field names
            # entry.length is the raw byte length, len(entry.value) is decoded string length
            string_entry = {
                "string": entry.value,
                "string_raw": entry.raw,
                "string_encoding": entry.type.name,
                "string_offset": entry.start,
                "string_length": len(entry.value),
                "string_raw_length": entry.length,
                "string_entropy": self.entropy(entry.value),
            }

            strings[key] = string_entry

        # Return as list for export compatibility
        return list(strings.values())