Valentina Polli

13 papers B 2Journal 6Unranked 4
YearRankTypeTitle / Venue / Authors
2013 J jnl
IEEE Trans. Commun.
Nicola Cordeschi, Valentina Polli, Enzo Baccarelli
2013 J jnl
IEEE/ACM Trans. Netw.
Enzo Baccarelli, Nicola Cordeschi, Valentina Polli
2012
Valentina Polli
2012 J jnl
Comput. Networks
Nicola Cordeschi, Valentina Polli, Enzo Baccarelli
2011 J jnl
Int. J. Commun. Networks Distributed Syst.
Enzo Baccarelli, Nicola Cordeschi, Tatiana Patriarca, Valentina Polli
2011 J jnl
IEEE Trans. Commun.
Mauro Biagi, Valentina Polli, Tatiana Patriarca
2011 B conf
GLOBECOM
Enzo Baccarelli, Nicola Cordeschi, Valentina Polli
2011 J jnl
IEEE Trans. Commun.
Mauro Biagi, Valentina Polli
2010 conf
ISWCS
Mauro Biagi, Valentina Polli, Jose Alberto Andrade Freitas
2010 B conf
WCNC
Mauro Biagi, Enzo Baccarelli, Nicola Cordeschi, Valentina Polli, Tatiana Patriarca
2009 conf
ICUMT
Enzo Baccarelli, Nicola Cordeschi, Mauro Biagi, Tatiana Patriarca, Valentina Polli
2009 conf
Wireless Days
Mauro Biagi, Enzo Baccarelli, Nicola Cordeschi, Valentina Polli, Tatiana Patriarca
2008 conf
ISWCS
Mauro Biagi, Enzo Baccarelli, Nicola Cordeschi, Cristian Pelizzoni, Valentina Polli
redb/extractors/decompiler/bninja/analysis/strings.py
← Index redb/extractors/decompiler/bninja/analysis/strings.py python
from collections import Counter
import math

class StringAnalysis:
    def __init__(self, bv, functions):
        self.bv = bv
        self.functions = functions

    def entropy(self, s: str) -> float:
        """Compute Shannon entropy of a string."""
        if not s:
            return 0.0
        freq = Counter(s)
        length = len(s)
        return -sum((count / length) * math.log2(count / length) for count in freq.values())

    def analyze(self):
        """
        Extract unique strings from the binary.

        Deduplicates by (string, encoding) within the same binary, keeping the
        first occurrence (lowest offset). Cross-binary deduplication and
        aggregation is handled by ClickHouse materialized views.
        """
        strings = {}

        # Sort strings by their starting address
        sorted_entries = sorted(self.bv.strings, key=lambda e: e.start)

        for entry in sorted_entries:
            # Key is the string and its encoding
            key = (entry.value, entry.type.name)

            # Skip if this string (value + encoding) was already added.
            # Because entries are sorted by address, the first one is always kept.
            if key in strings:
                continue

            # Store only the first occurrence with schema-matching field names
            # entry.length is the raw byte length, len(entry.value) is decoded string length
            string_entry = {
                "string": entry.value,
                "string_raw": entry.raw,
                "string_encoding": entry.type.name,
                "string_offset": entry.start,
                "string_length": len(entry.value),
                "string_raw_length": entry.length,
                "string_entropy": self.entropy(entry.value),
            }

            strings[key] = string_entry

        # Return as list for export compatibility
        return list(strings.values())