Karel Palecek

18 papers A 1C 1Journal 1Unranked 15
YearRankTypeTitle / Venue / Authors
2023 conf
TSP
Josef Chaloupka, Karel Palecek
2021 conf
TSP
Karel Palecek, Josef Chaloupka
2020 conf
CogInfoCom
Josef Chaloupka, Karel Palecek, Petr Cerva, Jindrich Zdánský
2020 conf
TSP
Karel Palecek, Josef Chaloupka
2019 conf
TSP
Karel Palecek
2018 J jnl
J. Multimodal User Interfaces
Karel Palecek
2017 conf
EUSIPCO
Francesco Nesta, Saeed Mosayyebpour, Zbynek Koldovský, Karel Palecek
2017 conf
TSD
Karel Palecek
2017 conf
SPECOM
Karel Palecek
2016 conf
TSP
Karel Palecek, Josef Chaloupka
2016 conf
EUSIPCO
Karel Palecek
2015 conf
TSP
Karel Palecek
2014 conf
SPECOM
Karel Palecek
2013 conf
TSP
Karel Palecek, Josef Chaloupka
2012 C conf
MMSP
Petr Cerva, Jan Silovský, Jindrich Zdánský, Ondrej Smola, Karel Blavka, Karel Palecek, Jan Nouza, Jirí Málek
2011 conf
COST 2102 Training School
Karel Palecek, David Gerónimo, Frédéric Lerasle
2011 A conf
INTERSPEECH
Petr Cerva, Karel Palecek, Jan Silovský, Jan Nouza
2010 conf
COST 2102 Conference
Karel Palecek
redb/extractors/decompiler/bninja/analysis/strings.py
← Index redb/extractors/decompiler/bninja/analysis/strings.py python
from collections import Counter
import math

class StringAnalysis:
    def __init__(self, bv, functions):
        self.bv = bv
        self.functions = functions

    def entropy(self, s: str) -> float:
        """Compute Shannon entropy of a string."""
        if not s:
            return 0.0
        freq = Counter(s)
        length = len(s)
        return -sum((count / length) * math.log2(count / length) for count in freq.values())

    def analyze(self):
        """
        Extract unique strings from the binary.

        Deduplicates by (string, encoding) within the same binary, keeping the
        first occurrence (lowest offset). Cross-binary deduplication and
        aggregation is handled by ClickHouse materialized views.
        """
        strings = {}

        # Sort strings by their starting address
        sorted_entries = sorted(self.bv.strings, key=lambda e: e.start)

        for entry in sorted_entries:
            # Key is the string and its encoding
            key = (entry.value, entry.type.name)

            # Skip if this string (value + encoding) was already added.
            # Because entries are sorted by address, the first one is always kept.
            if key in strings:
                continue

            # Store only the first occurrence with schema-matching field names
            # entry.length is the raw byte length, len(entry.value) is decoded string length
            string_entry = {
                "string": entry.value,
                "string_raw": entry.raw,
                "string_encoding": entry.type.name,
                "string_offset": entry.start,
                "string_length": len(entry.value),
                "string_raw_length": entry.length,
                "string_entropy": self.entropy(entry.value),
            }

            strings[key] = string_entry

        # Return as list for export compatibility
        return list(strings.values())