Kamal Ali

13 papers A* 5A 1B 1Misc 2Journal 3Unranked 1
YearRankTypeTitle / Venue / Authors
2023 J jnl
CoRR
Augustine Ukpebor, James Addy, Kamal Ali, Ali Abu-El Humos
2012 A* conf
KDD
Kent Shi, Kamal Ali
2011 J jnl
AI Mag.
David J. Stracuzzi, Alan Fern, Kamal Ali, Robin Hess, Jervis Pinto, Nan Li, Tolga Könik, Daniel G. Shapiro
2010 Misc conf
FLAIRS
Tolga Könik, Kamal Ali, Daniel G. Shapiro, Nan Li, David J. Stracuzzi
2009 conf
AIIDE
Nan Li, David J. Stracuzzi, Gary Cleveland, Tolga Könik, Daniel G. Shapiro, Matthew Molineaux, David W. Aha, Kamal Ali
2009 B conf
ILP
Kamal Ali, Kevin Leung, Tolga Könik, Dongkyu Choi, Daniel G. Shapiro
2007 A* conf
WWW
Kamal Ali, Mark Scarr
2005 A conf
ECIR
Kamal Ali, Chi-Chao Chang, Yun-Fang Juan
2004 A* conf
KDD
Kamal Ali, Wijnand van Stam
2003 A* conf
KDD
Kamal Ali, Steven P. Ketchpel
2000 Misc conf
CATA
Kamal Ali, Isam Taha, Michael S. Jacobson
1997 A* conf
KDD
Kamal Ali, Stefanos Manganaris, Ramakrishnan Srikant
1989 J jnl
Knowl. Based Syst.
Kamal Ali, Chris Horsfall, Raymond Lister
redb/extractors/decompiler/bninja/analysis/strings.py
← Index redb/extractors/decompiler/bninja/analysis/strings.py python
from collections import Counter
import math

class StringAnalysis:
    def __init__(self, bv, functions):
        self.bv = bv
        self.functions = functions

    def entropy(self, s: str) -> float:
        """Compute Shannon entropy of a string."""
        if not s:
            return 0.0
        freq = Counter(s)
        length = len(s)
        return -sum((count / length) * math.log2(count / length) for count in freq.values())

    def analyze(self):
        """
        Extract unique strings from the binary.

        Deduplicates by (string, encoding) within the same binary, keeping the
        first occurrence (lowest offset). Cross-binary deduplication and
        aggregation is handled by ClickHouse materialized views.
        """
        strings = {}

        # Sort strings by their starting address
        sorted_entries = sorted(self.bv.strings, key=lambda e: e.start)

        for entry in sorted_entries:
            # Key is the string and its encoding
            key = (entry.value, entry.type.name)

            # Skip if this string (value + encoding) was already added.
            # Because entries are sorted by address, the first one is always kept.
            if key in strings:
                continue

            # Store only the first occurrence with schema-matching field names
            # entry.length is the raw byte length, len(entry.value) is decoded string length
            string_entry = {
                "string": entry.value,
                "string_raw": entry.raw,
                "string_encoding": entry.type.name,
                "string_offset": entry.start,
                "string_length": len(entry.value),
                "string_raw_length": entry.length,
                "string_entropy": self.entropy(entry.value),
            }

            strings[key] = string_entry

        # Return as list for export compatibility
        return list(strings.values())