Carter Zenke

12 papers Unranked 12
YearRankTypeTitle / Venue / Authors
2024 conf
SIGCSE (2)
Doug Lloyd, Carter Zenke, David J. Malan
2024 conf
SIGCSE (2)
Rongxin Liu, Charlie Liu, Carter Zenke, David J. Malan
2024 conf
SIGCSE (2)
David J. Malan, Rongxin Liu, Carter Zenke, Doug Lloyd
2024 conf
SIGCSE (1)
Rongxin Liu, Carter Zenke, Charlie Liu, Andrew Holmes, Patrick Thornton, David J. Malan
2024 conf
SIGCSE (2)
Rongxin Liu, Carter Zenke, Charlie Liu, Andrew Holmes, Patrick Thornton, David J. Malan
2024 conf
SIGCSE (2)
Rongxin Liu, Carter Zenke, Doug Lloyd, David J. Malan
2024 conf
SIGCSE (2)
Yuliia Zhukovets, Carter Zenke, David J. Malan
2023 conf
SIGCSE (2)
David J. Malan, Doug Lloyd, Carter Zenke
2023 conf
SIGCSE (2)
Carter Zenke, David J. Malan
2023 conf
SIGCSE (2)
Ryan Hecht, Rongxin Liu, Carter Zenke, David J. Malan
2023 conf
SIGCSE (2)
David J. Malan, Jonathan Carter, Rongxin Liu, Carter Zenke
2022 conf
SIGCSE (2)
David J. Malan, Doug Lloyd, Carter Zenke
redb/extractors/decompiler/bninja/analysis/strings.py
← Index redb/extractors/decompiler/bninja/analysis/strings.py python
from collections import Counter
import math

class StringAnalysis:
    def __init__(self, bv, functions):
        self.bv = bv
        self.functions = functions

    def entropy(self, s: str) -> float:
        """Compute Shannon entropy of a string."""
        if not s:
            return 0.0
        freq = Counter(s)
        length = len(s)
        return -sum((count / length) * math.log2(count / length) for count in freq.values())

    def analyze(self):
        """
        Extract unique strings from the binary.

        Deduplicates by (string, encoding) within the same binary, keeping the
        first occurrence (lowest offset). Cross-binary deduplication and
        aggregation is handled by ClickHouse materialized views.
        """
        strings = {}

        # Sort strings by their starting address
        sorted_entries = sorted(self.bv.strings, key=lambda e: e.start)

        for entry in sorted_entries:
            # Key is the string and its encoding
            key = (entry.value, entry.type.name)

            # Skip if this string (value + encoding) was already added.
            # Because entries are sorted by address, the first one is always kept.
            if key in strings:
                continue

            # Store only the first occurrence with schema-matching field names
            # entry.length is the raw byte length, len(entry.value) is decoded string length
            string_entry = {
                "string": entry.value,
                "string_raw": entry.raw,
                "string_encoding": entry.type.name,
                "string_offset": entry.start,
                "string_length": len(entry.value),
                "string_raw_length": entry.length,
                "string_entropy": self.entropy(entry.value),
            }

            strings[key] = string_entry

        # Return as list for export compatibility
        return list(strings.values())