Xiaobo Zeng

11 papers B 1Journal 3Unranked 7
YearRankTypeTitle / Venue / Authors
2024 conf
OFC
Meng Cai, Xiaomin Liu, Mengfan Fu, Xiaobo Zeng, Yichen Liu, Yihao Zhang, Lilin Yi, Weisheng Hu, Qunbi Zhuge
2023 conf
OFC
Xiaobo Zeng, Yixiao Zhu, Yicheng Xu, Mengfan Fu, Hexun Jiang, Lilin Yi, Weisheng Hu, Qunbi Zhuge
2022 conf
OECC/PSC
Hexun Jiang, Mengfan Fu, Xiaobo Zeng, Huazhi Lun, Lei Liu, Lilin Yi, Weisheng Hu, Qunbi Zhuge
2022 conf
OFC
Qunbi Zhuge, Yicheng Xu, Yunyun Fan, Xiaobo Zeng, Mengfan Fu, Lilin Yi, Weisheng Hu, Xiang Liu
2021 J jnl
IEEE Access
Min Zhu, Jiahua Gu, Xiaobo Zeng, Chunping Yan, Pingping Gu
2020 conf
OFC
Mengfan Fu, Qiaoya Liu, Xiaobo Zeng, Yiwen Wu, Lilin Yi, Weisheng Hu, Qunbi Zhuge
2020 conf
OFC
Mengfan Fu, Qiaoya Liu, Xiaobo Zeng, Yiwen Wu, Lilin Yi, Weisheng Hu, Qunbi Zhuge
2018 J jnl
Photonic Netw. Commun.
Xiaobo Zeng, Min Zhu, Lu Wang, Xiaohan Sun
2017 B conf
GLOBECOM
Min Zhu, Pan Gao, Jiao Zhang, Xiaobo Zeng, Shengyu Zhang
2017 J jnl
JOCN
Min Zhu, Xiaobo Zeng, Yang Lin, Xiaohan Sun
2017 conf
ICTON
Xiaobo Zeng, Min Zhu, Guixin Li
redb/extractors/decompiler/apk/smali_normalization.py
← Index redb/extractors/decompiler/apk/smali_normalization.py python
"""Semantic normalization of Dalvik/smali instructions.

Analogous to Binary Ninja's LLIL normalization: strips register allocation
noise and instruction encoding variants while preserving semantic operations.

Three normalization levels (most aggressive to most detailed):
  - 'category':    semantic category only (MOV, ALU, CALL, ...)
  - 'opcode':      base opcode, width-invariant (add, sub, invoke, ...)
  - 'opcode_api':  opcode category + API method/field references for
                   invoke/field/alloc instructions (default for MinHash)

References:
  - Smali+ 12-category reduction (Canfora et al.)
  - MOSDroid opcode family grouping
  - DroidSIFT/DroidSim API-sensitive similarity
"""

import re
from typing import List

# ---------------------------------------------------------------------------
# Dalvik opcode -> semantic category mapping
# ---------------------------------------------------------------------------
# Prefix-matched against instruction opcodes. Order matters for overlapping
# prefixes (longer/more-specific prefixes should come first in iteration,
# but since we use startswith and break on first match, we order by
# specificity within the list).

OPCODE_CATEGORIES = {
    # Arithmetic/logic
    "add": "ALU", "sub": "ALU", "mul": "ALU", "div": "ALU",
    "rem": "ALU", "and": "ALU", "or": "ALU", "xor": "ALU",
    "shl": "ALU", "shr": "ALU", "ushr": "ALU", "neg": "ALU",
    "not": "ALU",
    # Data movement
    "move": "MOV", "const": "CONST",
    # Memory access (field/array)
    "iget": "LOAD", "sget": "LOAD", "aget": "LOAD",
    "iput": "STORE", "sput": "STORE", "aput": "STORE",
    # Invocations
    "invoke": "CALL",
    # Control flow
    "if": "BRANCH", "goto": "JMP",
    "switch": "SWITCH",
    "return": "RET",
    # Object/type
    "new": "ALLOC", "check": "TYPE", "instance": "TYPE",
    # Array
    "fill": "ARR", "array": "ARR",
    # Comparison
    "cmpl": "CMP", "cmpg": "CMP", "cmp": "CMP",
    # Exception / synchronization
    "throw": "EXC", "monitor": "SYNC",
    # Conversion (int-to-long, float-to-int, etc.)
    "int-to": "CONV", "long-to": "CONV", "float-to": "CONV",
    "double-to": "CONV",
}

# Pre-compiled regexes for operand extraction
_METHOD_REF_RE = re.compile(r"(L[\w/$]+;->[\w<>]+\(.*?\)[\w/$;\[]*)")
_FIELD_REF_RE = re.compile(r"(L[\w/$]+;->[\w]+:[\w/$;\[]+)")
_CLASS_REF_RE = re.compile(r"(L[\w/$]+;)")
_CONST_STRING_RE = re.compile(r'^const-string(?:/jumbo)?\s')


def categorize_opcode(opcode: str) -> str:
    """Map a Dalvik opcode to its semantic category.

    Prefix-matched: 'add-int/2addr' matches 'add' -> 'ALU'.
    Returns 'OTHER' for unrecognized opcodes.
    """
    for prefix, cat in OPCODE_CATEGORIES.items():
        if opcode.startswith(prefix):
            return cat
    return "OTHER"


# Mapping from semantic categories to the ACFG feature vector indices
# used by Binary Ninja's build_block_features (cfg_features.py).
# This enables cross-platform ACFG feature comparison.
CATEGORY_TO_ACFG_INDEX = {
    "ALU": 0,       # CAT_ARITHMETIC
    "CONV": 0,      # arithmetic-adjacent
    "CMP": 4,       # CAT_COMPARISON
    "MOV": 2,       # CAT_TRANSFER
    "CONST": 2,     # transfer-adjacent (loading constants)
    "LOAD": 5,      # CAT_MEMORY
    "STORE": 5,     # CAT_MEMORY
    "CALL": 3,      # CAT_CALL
    "BRANCH": 1,    # CAT_LOGIC (conditional logic)
    "JMP": 1,       # CAT_LOGIC
    "SWITCH": 1,    # CAT_LOGIC
    "RET": 2,       # CAT_TRANSFER
    "ALLOC": 5,     # CAT_MEMORY (heap allocation)
    "TYPE": 6,      # CAT_OTHER
    "ARR": 5,       # CAT_MEMORY
    "EXC": 6,       # CAT_OTHER
    "SYNC": 6,      # CAT_OTHER
    "OTHER": 6,     # CAT_OTHER
}


def normalize_instruction(line: str, level: str = "opcode_api") -> str:
    """Normalize a single smali instruction line.

    Args:
        line: A single smali instruction (whitespace-stripped).
        level: Normalization level:
            'category'   - most aggressive: just semantic category
            'opcode'     - base opcode only, width/addressing-mode invariant
            'opcode_api' - category + API references for invoke/field/alloc
                          (default, best for MinHash similarity)

    Returns:
        Normalized instruction string, or empty string for non-instructions.
    """
    stripped = line.strip()
    if not stripped:
        return ""

    parts = stripped.split(None, 1)
    opcode = parts[0]
    operands = parts[1] if len(parts) > 1 else ""

    if level == "category":
        return categorize_opcode(opcode)

    if level == "opcode":
        # Strip type/width suffixes for invariance:
        # add-int, add-long, add-float -> 'add'
        # add-int/2addr -> 'add'
        base = re.split(r"[-/]", opcode)[0]
        return base

    if level == "opcode_api":
        # const-string: preserve string content (encrypted strings are a
        # key malware indicator)
        if _CONST_STRING_RE.match(stripped):
            # Extract the string literal
            str_match = re.search(r'"(.*)"', operands)
            if str_match:
                return f"CONST_STR \"{str_match.group(1)}\""
            return "CONST_STR"

        # invoke-*: preserve method reference
        if opcode.startswith("invoke"):
            ref = _METHOD_REF_RE.search(operands)
            if ref:
                return f"CALL {ref.group(1)}"
            return "CALL"

        # Field access: preserve field reference
        if opcode.startswith(("iget", "iput", "sget", "sput")):
            ref = _FIELD_REF_RE.search(operands)
            if ref:
                cat = "LOAD" if "get" in opcode else "STORE"
                return f"{cat} {ref.group(1)}"
            # Fallback: try space-separated format from androguard
            # e.g. "iget v0, p0, Lcom/Foo;->field Ljava/lang/String;"
            space_ref = re.search(
                r"(L[\w/$]+;->[\w]+)\s+([\w/$;\[]+)", operands
            )
            if space_ref:
                cat = "LOAD" if "get" in opcode else "STORE"
                return f"{cat} {space_ref.group(1)}:{space_ref.group(2)}"
            cat = "LOAD" if "get" in opcode else "STORE"
            return cat

        # new-instance: preserve allocated type
        if opcode.startswith("new-instance") or opcode == "new-array":
            ref = _CLASS_REF_RE.search(operands)
            if ref:
                return f"ALLOC {ref.group(1)}"
            return "ALLOC"

        # Everything else: just the category
        return categorize_opcode(opcode)

    # Unknown level: return raw opcode
    return opcode


def normalize_method_body(
    body: str, level: str = "opcode_api"
) -> List[str]:
    """Normalize all instructions in a smali method body.

    Filters out directives (.), labels (:), comments (#), and blank lines.
    Returns a list of normalized instruction strings.

    Args:
        body: Raw smali method body text.
        level: Normalization level (see normalize_instruction).

    Returns:
        List of normalized instruction strings (no empty strings).
    """
    normalized = []
    for line in body.split("\n"):
        stripped = line.strip()
        # Skip non-instructions
        if not stripped:
            continue
        if stripped.startswith((".",":", "#")):
            continue
        result = normalize_instruction(stripped, level)
        if result:
            normalized.append(result)
    return normalized