Maha Cherif

12 papers B 1C 3Misc 2Journal 4Unranked 2
YearRankTypeTitle / Venue / Authors
2024 J jnl
IET Commun.
Ahlem Arfaoui, Maha Cherif, Ridha Bouallegue
2023 C conf
ISCC
Ahlem Arfaoui, Maha Cherif, Ridha Bouallegue
2023 J jnl
IET Commun.
Maha Cherif, Ahlem Arfaoui, Ridha Bouallegue
2023 C conf
ISCC
Ahlem Arfaoui, Maha Cherif, Ridha Bouallegue
2023 J jnl
IEEE Syst. J.
Maha Cherif, Ahlem Arfaoui, Rafik Zayani, Ridha Bouallegue
2023 conf
ComNet
Moez Hizem, Imen Abidi, Maha Cherif, Ridha Bouallègue
2022 J jnl
IEEE Access
Noura Derria Lahbib, Maha Cherif, Moez Hizem, Ridha Bouallegue
2022 conf
ComNet
Moez Hizem, Imen Abidi, Maha Cherif, Ridha Bouallegue
2022 B conf
IWCMC
Ahlem Arfaoui, Maha Cherif, Ridha Bouallegue
2021 C conf
ISCC
Imen Abidi, Maha Cherif, Moez Hizem, Iness Ahriz, Ridha Bouallegue
2020 Misc conf
SoftCOM
Noura Derria Lahbib, Maha Cherif, Moez Hizem, Ridha Bouallegue
2020 Misc conf
SoftCOM
Maha Cherif, Ridha Bouallègue
redb/extractors/decompiler/bninja/analysis/strings.py
← Index redb/extractors/decompiler/bninja/analysis/strings.py python
from collections import Counter
import math

class StringAnalysis:
    def __init__(self, bv, functions):
        self.bv = bv
        self.functions = functions

    def entropy(self, s: str) -> float:
        """Compute Shannon entropy of a string."""
        if not s:
            return 0.0
        freq = Counter(s)
        length = len(s)
        return -sum((count / length) * math.log2(count / length) for count in freq.values())

    def analyze(self):
        """
        Extract unique strings from the binary.

        Deduplicates by (string, encoding) within the same binary, keeping the
        first occurrence (lowest offset). Cross-binary deduplication and
        aggregation is handled by ClickHouse materialized views.
        """
        strings = {}

        # Sort strings by their starting address
        sorted_entries = sorted(self.bv.strings, key=lambda e: e.start)

        for entry in sorted_entries:
            # Key is the string and its encoding
            key = (entry.value, entry.type.name)

            # Skip if this string (value + encoding) was already added.
            # Because entries are sorted by address, the first one is always kept.
            if key in strings:
                continue

            # Store only the first occurrence with schema-matching field names
            # entry.length is the raw byte length, len(entry.value) is decoded string length
            string_entry = {
                "string": entry.value,
                "string_raw": entry.raw,
                "string_encoding": entry.type.name,
                "string_offset": entry.start,
                "string_length": len(entry.value),
                "string_raw_length": entry.length,
                "string_entropy": self.entropy(entry.value),
            }

            strings[key] = string_entry

        # Return as list for export compatibility
        return list(strings.values())