Ramunas Girdziusas

12 papers A* 1B 2Journal 1Unranked 8
YearRankTypeTitle / Venue / Authors
2014 conf
3DOR@Eurographics
Davide Boscaini, Ramunas Girdziusas, Michael M. Bronstein
2012 conf
LION
Janis Janusevskis, Rodolphe Le Riche, David Ginsbourger, Ramunas Girdziusas
2007 conf
ACCV (1)
Ramunas Girdziusas, Jorma Laaksonen
2007 A* conf
ICCV
Ramunas Girdziusas, Jorma Laaksonen
2005 B conf
IJCNN
Ramunas Girdziusas, Jorma Laaksonen
2005 conf
ICANN (2)
Ramunas Girdziusas, Jorma Laaksonen
2005 conf
SCIA
Ramunas Girdziusas, Jorma Laaksonen
2005 conf
ECML
Ramunas Girdziusas, Jorma Laaksonen
2005 conf
ICAPR (1)
Ramunas Girdziusas, Jorma Laaksonen
2004 B conf
ICONIP
Ramunas Girdziusas, Jorma Laaksonen
2003 J jnl
Int. J. Document Anal. Recognit.
Matti Aksela, Ramunas Girdziusas, Jorma Laaksonen, Erkki Oja, Jari Kangas
2002 conf
IWFHR
Matti Aksela, Ramunas Girdziusas, Jorma Laaksonen, Erkki Oja, Jari Kangas
redb/extractors/decompiler/bninja/analysis/strings.py
← Index redb/extractors/decompiler/bninja/analysis/strings.py python
from collections import Counter
import math

class StringAnalysis:
    def __init__(self, bv, functions):
        self.bv = bv
        self.functions = functions

    def entropy(self, s: str) -> float:
        """Compute Shannon entropy of a string."""
        if not s:
            return 0.0
        freq = Counter(s)
        length = len(s)
        return -sum((count / length) * math.log2(count / length) for count in freq.values())

    def analyze(self):
        """
        Extract unique strings from the binary.

        Deduplicates by (string, encoding) within the same binary, keeping the
        first occurrence (lowest offset). Cross-binary deduplication and
        aggregation is handled by ClickHouse materialized views.
        """
        strings = {}

        # Sort strings by their starting address
        sorted_entries = sorted(self.bv.strings, key=lambda e: e.start)

        for entry in sorted_entries:
            # Key is the string and its encoding
            key = (entry.value, entry.type.name)

            # Skip if this string (value + encoding) was already added.
            # Because entries are sorted by address, the first one is always kept.
            if key in strings:
                continue

            # Store only the first occurrence with schema-matching field names
            # entry.length is the raw byte length, len(entry.value) is decoded string length
            string_entry = {
                "string": entry.value,
                "string_raw": entry.raw,
                "string_encoding": entry.type.name,
                "string_offset": entry.start,
                "string_length": len(entry.value),
                "string_raw_length": entry.length,
                "string_entropy": self.entropy(entry.value),
            }

            strings[key] = string_entry

        # Return as list for export compatibility
        return list(strings.values())