Ka C. Cheok

36 papers B 2Misc 2Journal 11Unranked 21
YearRankTypeTitle / Venue / Authors
2024 conf
EIT
Tien-Chuong Lim, Ka C. Cheok, Subramaniam Ganesan
2023 conf
SICE
Natsumi Matsumoto, Kazuyuki Kobayashi, Tomoyuki Ohkubo, Kaiqiao Tian, Nashwan J. Sebi, Ka C. Cheok, Changqing Cai
2022 conf
SICE
Kimihiro Mori, Tomoyuki Ohkubo, Kazuyuki Kobayashi, Kajiro Watanabe, Kaiqiao Tian, Nashwan J. Sebi, Ka C. Cheok
2022 conf
SysCon
Kaiqiao Tian, Micho Radovnikovich, Ka C. Cheok
2022 conf
SICE
Koki Kubota, Kazuyuki Kobayashi, Tomoyuki Ohkubo, Kajiro Watanabe, Nashwan J. Sebi, Kaiqiao Tian, Ka C. Cheok
2022 conf
SICE
Shunki Shibuya, Kazuyuki Kobayashi, Tomoyuki Ohkubo, Kajiro Watanabe, Kaiqiao Tian, Nashwan J. Sebi, Ka C. Cheok
2021 conf
IntelliSys (3)
Nashwan J. Sebi, Kazuyuki Kobayashi, Ka C. Cheok
2021 conf
SICE
Riki Uchida, Kazuyuki Kobayashi, Tomoyuki Ohkubo, Kajiro Watanabe, Nashwan J. Sebi, Ka C. Cheok
2021 conf
SICE
Riku Yamamoto, Tomoyuki Ohkubo, Kazuyuki Kobayashi, Kajiro Watanabe, Nashwan J. Sebi, Ka C. Cheok
2021 conf
SICE
Koki Kuroki, Kazuyuki Kobayashi, Kajiro Watanabe, Tomoyuki Ohkubo, Nashwan J. Sebi, Ka C. Cheok
2021 conf
SICE
Yuto Miura, Kazuyuki Kobayashi, Tomoyuki Ohkubo, Kajiro Watanabe, Nashwan J. Sebi, Ka C. Cheok
2021 conf
SysCon
Ana Farhat, Ka C. Cheok
2021 conf
EIT
Tien-Chuong Lim, Ka C. Cheok, Subramaniam Ganesan
2020 conf
SII
Kaiqiao Tian, Kazuyuki Kobayashi, Ka C. Cheok
2019 Misc conf
CATA
Ana Farhat, Kyle Hagen, Ka C. Cheok, Balaji Boominathan
2019 Misc conf
CATA
Ka C. Cheok, Kiran Iyengar, Sami Oweis
2017 conf
SysCon
Ana Farhat, Ka C. Cheok
2014 J jnl
Int. J. Handheld Comput. Res.
Sami Oweis, Subramaniam Ganesan, Ka C. Cheok
2014 conf
EIT
Sami Oweis, Subramaniam Ganesan, Ka C. Cheok
2014 conf
TePRA
Micho Radovnikovich, Ka C. Cheok
2013 conf
MWSCAS
Abdulhakim A. Ezzabi, Ka C. Cheok, Fatma A. Alazabi
2010 B conf
PIMRC
Ka C. Cheok, Micho Radovnikovich, P. K. Vempaty, Gregory R. Hudas, James L. Overholt, Paul Fleck
2008 conf
EIT
Riyadh Kenaya, Ka C. Cheok
2007 J jnl
Int. J. Approx. Reason.
Gregory R. Hudas, Ka C. Cheok, James L. Overholt
2004 J jnl
J. Field Robotics
Ka C. Cheok, Gert-Edzko Smid, Gerald R. Lane, William G. Agnew, Asif Khan
2004 J jnl
J. Field Robotics
Jerry Lane, Ka C. Cheok, Bill Agnew
2002 conf
IS
Gert-Edzko Smid, Ka C. Cheok, Grant Gerhart, Gregory R. Hudas
2002 conf
Mobile Robots
Gregory R. Hudas, Ka C. Cheok, James L. Overholt, Gert-Edzko Smid
2001 B conf
SMC
James L. Overholt, Ka C. Cheok
1998 J jnl
IEEE Trans. Ind. Electron.
Kazuyuki Kobayashi, Ka C. Cheok, Kajiro Watanabe, Fumio Munekata
1997 J jnl
J. Field Robotics
Ka C. Cheok, Gert-Edzko Smid, Kazuyuki Kobayashi, James L. Overholt, Paul Lescoe
1996 J jnl
Autom.
Naim A. Kheir, Karl Johan Åström, David M. Auslander, Ka C. Cheok, Gene F. Franklin, M. Masten, M. Rabins
1995 J jnl
Intell. Autom. Soft Comput.
Kazuyuki Kobayashi, Ka C. Cheok, Kajiro Watanabe
1993 J jnl
J. Field Robotics
Ka C. Cheok, James L. Overholt, Ronald R. Beck
1993 J jnl
Simul.
Ningjian Huang, Ka C. Cheok, Thomas G. Horner, Timothy Settle
1992 J jnl
Simul.
Ka C. Cheok, Ningjian Huang
redb/extractors/decompiler/apk/smali_normalization.py
← Index redb/extractors/decompiler/apk/smali_normalization.py python
"""Semantic normalization of Dalvik/smali instructions.

Analogous to Binary Ninja's LLIL normalization: strips register allocation
noise and instruction encoding variants while preserving semantic operations.

Three normalization levels (most aggressive to most detailed):
  - 'category':    semantic category only (MOV, ALU, CALL, ...)
  - 'opcode':      base opcode, width-invariant (add, sub, invoke, ...)
  - 'opcode_api':  opcode category + API method/field references for
                   invoke/field/alloc instructions (default for MinHash)

References:
  - Smali+ 12-category reduction (Canfora et al.)
  - MOSDroid opcode family grouping
  - DroidSIFT/DroidSim API-sensitive similarity
"""

import re
from typing import List

# ---------------------------------------------------------------------------
# Dalvik opcode -> semantic category mapping
# ---------------------------------------------------------------------------
# Prefix-matched against instruction opcodes. Order matters for overlapping
# prefixes (longer/more-specific prefixes should come first in iteration,
# but since we use startswith and break on first match, we order by
# specificity within the list).

OPCODE_CATEGORIES = {
    # Arithmetic/logic
    "add": "ALU", "sub": "ALU", "mul": "ALU", "div": "ALU",
    "rem": "ALU", "and": "ALU", "or": "ALU", "xor": "ALU",
    "shl": "ALU", "shr": "ALU", "ushr": "ALU", "neg": "ALU",
    "not": "ALU",
    # Data movement
    "move": "MOV", "const": "CONST",
    # Memory access (field/array)
    "iget": "LOAD", "sget": "LOAD", "aget": "LOAD",
    "iput": "STORE", "sput": "STORE", "aput": "STORE",
    # Invocations
    "invoke": "CALL",
    # Control flow
    "if": "BRANCH", "goto": "JMP",
    "switch": "SWITCH",
    "return": "RET",
    # Object/type
    "new": "ALLOC", "check": "TYPE", "instance": "TYPE",
    # Array
    "fill": "ARR", "array": "ARR",
    # Comparison
    "cmpl": "CMP", "cmpg": "CMP", "cmp": "CMP",
    # Exception / synchronization
    "throw": "EXC", "monitor": "SYNC",
    # Conversion (int-to-long, float-to-int, etc.)
    "int-to": "CONV", "long-to": "CONV", "float-to": "CONV",
    "double-to": "CONV",
}

# Pre-compiled regexes for operand extraction
_METHOD_REF_RE = re.compile(r"(L[\w/$]+;->[\w<>]+\(.*?\)[\w/$;\[]*)")
_FIELD_REF_RE = re.compile(r"(L[\w/$]+;->[\w]+:[\w/$;\[]+)")
_CLASS_REF_RE = re.compile(r"(L[\w/$]+;)")
_CONST_STRING_RE = re.compile(r'^const-string(?:/jumbo)?\s')


def categorize_opcode(opcode: str) -> str:
    """Map a Dalvik opcode to its semantic category.

    Prefix-matched: 'add-int/2addr' matches 'add' -> 'ALU'.
    Returns 'OTHER' for unrecognized opcodes.
    """
    for prefix, cat in OPCODE_CATEGORIES.items():
        if opcode.startswith(prefix):
            return cat
    return "OTHER"


# Mapping from semantic categories to the ACFG feature vector indices
# used by Binary Ninja's build_block_features (cfg_features.py).
# This enables cross-platform ACFG feature comparison.
CATEGORY_TO_ACFG_INDEX = {
    "ALU": 0,       # CAT_ARITHMETIC
    "CONV": 0,      # arithmetic-adjacent
    "CMP": 4,       # CAT_COMPARISON
    "MOV": 2,       # CAT_TRANSFER
    "CONST": 2,     # transfer-adjacent (loading constants)
    "LOAD": 5,      # CAT_MEMORY
    "STORE": 5,     # CAT_MEMORY
    "CALL": 3,      # CAT_CALL
    "BRANCH": 1,    # CAT_LOGIC (conditional logic)
    "JMP": 1,       # CAT_LOGIC
    "SWITCH": 1,    # CAT_LOGIC
    "RET": 2,       # CAT_TRANSFER
    "ALLOC": 5,     # CAT_MEMORY (heap allocation)
    "TYPE": 6,      # CAT_OTHER
    "ARR": 5,       # CAT_MEMORY
    "EXC": 6,       # CAT_OTHER
    "SYNC": 6,      # CAT_OTHER
    "OTHER": 6,     # CAT_OTHER
}


def normalize_instruction(line: str, level: str = "opcode_api") -> str:
    """Normalize a single smali instruction line.

    Args:
        line: A single smali instruction (whitespace-stripped).
        level: Normalization level:
            'category'   - most aggressive: just semantic category
            'opcode'     - base opcode only, width/addressing-mode invariant
            'opcode_api' - category + API references for invoke/field/alloc
                          (default, best for MinHash similarity)

    Returns:
        Normalized instruction string, or empty string for non-instructions.
    """
    stripped = line.strip()
    if not stripped:
        return ""

    parts = stripped.split(None, 1)
    opcode = parts[0]
    operands = parts[1] if len(parts) > 1 else ""

    if level == "category":
        return categorize_opcode(opcode)

    if level == "opcode":
        # Strip type/width suffixes for invariance:
        # add-int, add-long, add-float -> 'add'
        # add-int/2addr -> 'add'
        base = re.split(r"[-/]", opcode)[0]
        return base

    if level == "opcode_api":
        # const-string: preserve string content (encrypted strings are a
        # key malware indicator)
        if _CONST_STRING_RE.match(stripped):
            # Extract the string literal
            str_match = re.search(r'"(.*)"', operands)
            if str_match:
                return f"CONST_STR \"{str_match.group(1)}\""
            return "CONST_STR"

        # invoke-*: preserve method reference
        if opcode.startswith("invoke"):
            ref = _METHOD_REF_RE.search(operands)
            if ref:
                return f"CALL {ref.group(1)}"
            return "CALL"

        # Field access: preserve field reference
        if opcode.startswith(("iget", "iput", "sget", "sput")):
            ref = _FIELD_REF_RE.search(operands)
            if ref:
                cat = "LOAD" if "get" in opcode else "STORE"
                return f"{cat} {ref.group(1)}"
            # Fallback: try space-separated format from androguard
            # e.g. "iget v0, p0, Lcom/Foo;->field Ljava/lang/String;"
            space_ref = re.search(
                r"(L[\w/$]+;->[\w]+)\s+([\w/$;\[]+)", operands
            )
            if space_ref:
                cat = "LOAD" if "get" in opcode else "STORE"
                return f"{cat} {space_ref.group(1)}:{space_ref.group(2)}"
            cat = "LOAD" if "get" in opcode else "STORE"
            return cat

        # new-instance: preserve allocated type
        if opcode.startswith("new-instance") or opcode == "new-array":
            ref = _CLASS_REF_RE.search(operands)
            if ref:
                return f"ALLOC {ref.group(1)}"
            return "ALLOC"

        # Everything else: just the category
        return categorize_opcode(opcode)

    # Unknown level: return raw opcode
    return opcode


def normalize_method_body(
    body: str, level: str = "opcode_api"
) -> List[str]:
    """Normalize all instructions in a smali method body.

    Filters out directives (.), labels (:), comments (#), and blank lines.
    Returns a list of normalized instruction strings.

    Args:
        body: Raw smali method body text.
        level: Normalization level (see normalize_instruction).

    Returns:
        List of normalized instruction strings (no empty strings).
    """
    normalized = []
    for line in body.split("\n"):
        stripped = line.strip()
        # Skip non-instructions
        if not stripped:
            continue
        if stripped.startswith((".",":", "#")):
            continue
        result = normalize_instruction(stripped, level)
        if result:
            normalized.append(result)
    return normalized