Ramkishore Bhattacharyya

12 papers Misc 1Journal 5Unranked 6
YearRankTypeTitle / Venue / Authors
2023 J jnl
IEEE ACM Trans. Comput. Biol. Bioinform.
Ramkishore Bhattacharyya
2021 J jnl
Bioinform.
Balaram Bhattacharyya, Uddalak Mitra, Ramkishore Bhattacharyya
2013 J jnl
Inf. Sci.
Ramkishore Bhattacharyya
2013 conf
PReMI
Ramkishore Bhattacharyya
2011 J jnl
Appl. Soft Comput.
Ramkishore Bhattacharyya
2011 J jnl
Pattern Recognit. Lett.
Ramkishore Bhattacharyya
2009 conf
ICAPR
Ramkishore Bhattacharyya
2008 conf
ICETET
Ramkishore Bhattacharyya, Balaram Bhattacharyya
2008 conf
BIRD
Ramkishore Bhattacharyya, Balaram Bhattacharyya
2007 conf
IICAI
Ramkishore Bhattacharyya, Balaram Bhattacharyya
2007 conf
PReMI
Ramkishore Bhattacharyya, Balaram Bhattacharyya
2006 Misc conf
COMAD
Ramkishore Bhattacharyya
redb/extractors/decompiler/apk/smali_normalization.py
← Index redb/extractors/decompiler/apk/smali_normalization.py python
"""Semantic normalization of Dalvik/smali instructions.

Analogous to Binary Ninja's LLIL normalization: strips register allocation
noise and instruction encoding variants while preserving semantic operations.

Three normalization levels (most aggressive to most detailed):
  - 'category':    semantic category only (MOV, ALU, CALL, ...)
  - 'opcode':      base opcode, width-invariant (add, sub, invoke, ...)
  - 'opcode_api':  opcode category + API method/field references for
                   invoke/field/alloc instructions (default for MinHash)

References:
  - Smali+ 12-category reduction (Canfora et al.)
  - MOSDroid opcode family grouping
  - DroidSIFT/DroidSim API-sensitive similarity
"""

import re
from typing import List

# ---------------------------------------------------------------------------
# Dalvik opcode -> semantic category mapping
# ---------------------------------------------------------------------------
# Prefix-matched against instruction opcodes. Order matters for overlapping
# prefixes (longer/more-specific prefixes should come first in iteration,
# but since we use startswith and break on first match, we order by
# specificity within the list).

OPCODE_CATEGORIES = {
    # Arithmetic/logic
    "add": "ALU", "sub": "ALU", "mul": "ALU", "div": "ALU",
    "rem": "ALU", "and": "ALU", "or": "ALU", "xor": "ALU",
    "shl": "ALU", "shr": "ALU", "ushr": "ALU", "neg": "ALU",
    "not": "ALU",
    # Data movement
    "move": "MOV", "const": "CONST",
    # Memory access (field/array)
    "iget": "LOAD", "sget": "LOAD", "aget": "LOAD",
    "iput": "STORE", "sput": "STORE", "aput": "STORE",
    # Invocations
    "invoke": "CALL",
    # Control flow
    "if": "BRANCH", "goto": "JMP",
    "switch": "SWITCH",
    "return": "RET",
    # Object/type
    "new": "ALLOC", "check": "TYPE", "instance": "TYPE",
    # Array
    "fill": "ARR", "array": "ARR",
    # Comparison
    "cmpl": "CMP", "cmpg": "CMP", "cmp": "CMP",
    # Exception / synchronization
    "throw": "EXC", "monitor": "SYNC",
    # Conversion (int-to-long, float-to-int, etc.)
    "int-to": "CONV", "long-to": "CONV", "float-to": "CONV",
    "double-to": "CONV",
}

# Pre-compiled regexes for operand extraction
_METHOD_REF_RE = re.compile(r"(L[\w/$]+;->[\w<>]+\(.*?\)[\w/$;\[]*)")
_FIELD_REF_RE = re.compile(r"(L[\w/$]+;->[\w]+:[\w/$;\[]+)")
_CLASS_REF_RE = re.compile(r"(L[\w/$]+;)")
_CONST_STRING_RE = re.compile(r'^const-string(?:/jumbo)?\s')


def categorize_opcode(opcode: str) -> str:
    """Map a Dalvik opcode to its semantic category.

    Prefix-matched: 'add-int/2addr' matches 'add' -> 'ALU'.
    Returns 'OTHER' for unrecognized opcodes.
    """
    for prefix, cat in OPCODE_CATEGORIES.items():
        if opcode.startswith(prefix):
            return cat
    return "OTHER"


# Mapping from semantic categories to the ACFG feature vector indices
# used by Binary Ninja's build_block_features (cfg_features.py).
# This enables cross-platform ACFG feature comparison.
CATEGORY_TO_ACFG_INDEX = {
    "ALU": 0,       # CAT_ARITHMETIC
    "CONV": 0,      # arithmetic-adjacent
    "CMP": 4,       # CAT_COMPARISON
    "MOV": 2,       # CAT_TRANSFER
    "CONST": 2,     # transfer-adjacent (loading constants)
    "LOAD": 5,      # CAT_MEMORY
    "STORE": 5,     # CAT_MEMORY
    "CALL": 3,      # CAT_CALL
    "BRANCH": 1,    # CAT_LOGIC (conditional logic)
    "JMP": 1,       # CAT_LOGIC
    "SWITCH": 1,    # CAT_LOGIC
    "RET": 2,       # CAT_TRANSFER
    "ALLOC": 5,     # CAT_MEMORY (heap allocation)
    "TYPE": 6,      # CAT_OTHER
    "ARR": 5,       # CAT_MEMORY
    "EXC": 6,       # CAT_OTHER
    "SYNC": 6,      # CAT_OTHER
    "OTHER": 6,     # CAT_OTHER
}


def normalize_instruction(line: str, level: str = "opcode_api") -> str:
    """Normalize a single smali instruction line.

    Args:
        line: A single smali instruction (whitespace-stripped).
        level: Normalization level:
            'category'   - most aggressive: just semantic category
            'opcode'     - base opcode only, width/addressing-mode invariant
            'opcode_api' - category + API references for invoke/field/alloc
                          (default, best for MinHash similarity)

    Returns:
        Normalized instruction string, or empty string for non-instructions.
    """
    stripped = line.strip()
    if not stripped:
        return ""

    parts = stripped.split(None, 1)
    opcode = parts[0]
    operands = parts[1] if len(parts) > 1 else ""

    if level == "category":
        return categorize_opcode(opcode)

    if level == "opcode":
        # Strip type/width suffixes for invariance:
        # add-int, add-long, add-float -> 'add'
        # add-int/2addr -> 'add'
        base = re.split(r"[-/]", opcode)[0]
        return base

    if level == "opcode_api":
        # const-string: preserve string content (encrypted strings are a
        # key malware indicator)
        if _CONST_STRING_RE.match(stripped):
            # Extract the string literal
            str_match = re.search(r'"(.*)"', operands)
            if str_match:
                return f"CONST_STR \"{str_match.group(1)}\""
            return "CONST_STR"

        # invoke-*: preserve method reference
        if opcode.startswith("invoke"):
            ref = _METHOD_REF_RE.search(operands)
            if ref:
                return f"CALL {ref.group(1)}"
            return "CALL"

        # Field access: preserve field reference
        if opcode.startswith(("iget", "iput", "sget", "sput")):
            ref = _FIELD_REF_RE.search(operands)
            if ref:
                cat = "LOAD" if "get" in opcode else "STORE"
                return f"{cat} {ref.group(1)}"
            # Fallback: try space-separated format from androguard
            # e.g. "iget v0, p0, Lcom/Foo;->field Ljava/lang/String;"
            space_ref = re.search(
                r"(L[\w/$]+;->[\w]+)\s+([\w/$;\[]+)", operands
            )
            if space_ref:
                cat = "LOAD" if "get" in opcode else "STORE"
                return f"{cat} {space_ref.group(1)}:{space_ref.group(2)}"
            cat = "LOAD" if "get" in opcode else "STORE"
            return cat

        # new-instance: preserve allocated type
        if opcode.startswith("new-instance") or opcode == "new-array":
            ref = _CLASS_REF_RE.search(operands)
            if ref:
                return f"ALLOC {ref.group(1)}"
            return "ALLOC"

        # Everything else: just the category
        return categorize_opcode(opcode)

    # Unknown level: return raw opcode
    return opcode


def normalize_method_body(
    body: str, level: str = "opcode_api"
) -> List[str]:
    """Normalize all instructions in a smali method body.

    Filters out directives (.), labels (:), comments (#), and blank lines.
    Returns a list of normalized instruction strings.

    Args:
        body: Raw smali method body text.
        level: Normalization level (see normalize_instruction).

    Returns:
        List of normalized instruction strings (no empty strings).
    """
    normalized = []
    for line in body.split("\n"):
        stripped = line.strip()
        # Skip non-instructions
        if not stripped:
            continue
        if stripped.startswith((".",":", "#")):
            continue
        result = normalize_instruction(stripped, level)
        if result:
            normalized.append(result)
    return normalized