Nan Liu

11 papers A 1Misc 1Journal 4Unranked 5
YearRankTypeTitle / Venue / Authors
2013 Misc conf
ICIG
Zhenfeng Zhu, Nan Liu, Yao Zhao
2012 conf
PCM
Yanhui Xiao, Zhenfeng Zhu, Nan Liu, Yao Zhao
2011 J jnl
IEEE Trans. Multim.
Nan Liu, Yao Zhao, Zhenfeng Zhu, Hanqing Lu
2011 conf
ICIMCS
Houde Yang, Nan Liu, Yao Zhao, Zhenfeng Zhu
2010 J jnl
J. Signal Process. Syst.
Shikui Wei, Yao Zhao, Zhenfeng Zhu, Nan Liu
2010 conf
PCM (1)
Nan Liu, Yao Zhao, Zhenfeng Zhu
2010 J jnl
IEICE Trans. Inf. Syst.
Nan Liu, Yao Zhao, Zhenfeng Zhu, Rongrong Ni
2010 A conf
ICME
Nan Liu, Yao Zhao, Zhenfeng Zhu, Hanqing Lu
2010 J jnl
IEEE Trans. Knowl. Data Eng.
Shikui Wei, Yao Zhao, Zhenfeng Zhu, Nan Liu
2007 conf
TRECVID
Shikui Wei, Yao Zhao, Zhenfeng Zhu, Nan Liu, Yufeng Zhao, Fang Wang, Xie Lin
2006 conf
TRECVID
Shikui Wei, Yao Zhao, Zhenfeng Zhu, Nan Liu, Yufeng Zhao, Liang Zhang, Fang Wang
redb/extractors/decompiler/bninja/analysis/strings.py
← Index redb/extractors/decompiler/bninja/analysis/strings.py python
from collections import Counter
import math

class StringAnalysis:
    def __init__(self, bv, functions):
        self.bv = bv
        self.functions = functions

    def entropy(self, s: str) -> float:
        """Compute Shannon entropy of a string."""
        if not s:
            return 0.0
        freq = Counter(s)
        length = len(s)
        return -sum((count / length) * math.log2(count / length) for count in freq.values())

    def analyze(self):
        """
        Extract unique strings from the binary.

        Deduplicates by (string, encoding) within the same binary, keeping the
        first occurrence (lowest offset). Cross-binary deduplication and
        aggregation is handled by ClickHouse materialized views.
        """
        strings = {}

        # Sort strings by their starting address
        sorted_entries = sorted(self.bv.strings, key=lambda e: e.start)

        for entry in sorted_entries:
            # Key is the string and its encoding
            key = (entry.value, entry.type.name)

            # Skip if this string (value + encoding) was already added.
            # Because entries are sorted by address, the first one is always kept.
            if key in strings:
                continue

            # Store only the first occurrence with schema-matching field names
            # entry.length is the raw byte length, len(entry.value) is decoded string length
            string_entry = {
                "string": entry.value,
                "string_raw": entry.raw,
                "string_encoding": entry.type.name,
                "string_offset": entry.start,
                "string_length": len(entry.value),
                "string_raw_length": entry.length,
                "string_entropy": self.entropy(entry.value),
            }

            strings[key] = string_entry

        # Return as list for export compatibility
        return list(strings.values())