Machteld J. Boonstra

18 papers Journal 3Unranked 15
YearRankTypeTitle / Venue / Authors
2025 conf
PRIME@MICCAI
Enrique Almar-Munoz, Marawan Elbatel, Cristian Izquierdo, Illia Stepin, Machteld J. Boonstra, Xiaomeng Li, Christian Kremser, Matthias Schwab, Steffen E. Petersen, Markus Haltmeier, Agnes Mayr, Karim Lekadir
2025 J jnl
CoRR
Ali Anil Sinaci, Senan Postaci, Dogukan Cavdaroglu, Machteld J. Boonstra, Okan Mercan, Kerem Yilmaz, Gokce Banu Laleci Erturkmen, Folkert W. Asselbergs, Karim Lekadir
2025 J jnl
Comput. Biol. Medicine
Manon Kloosterman, Iris van der Schaaf, Machteld J. Boonstra, Thom F. Oostendorp, Veronique Meijborg, Ruben Coronel, Peter Loh, Peter M. van Dam
2025 J jnl
CoRR
Vittorio Torri, Machteld J. Boonstra, Marielle C. van de Veerdonk, Deborah N. Kalkman, Alicia Uijl, Francesca Ieva, Ameen Abu-Hanna, Folkert W. Asselbergs, Iacer Calixto
2024 conf
SIPAIM
Xènia Puig-Bosch, Machteld J. Boonstra, Miriam Cabrita, Joan Perramon, Sara Munive, Andrea Guala, Vladimír Kincl, Saskia Haitjema, Carina Dantas, Folkert W. Asselbergs, Karim Lekadir
2023 conf
CinC
Manon Kloosterman, Machteld J. Boonstra, Iris van der Schaaf, Peter Loh, Peter M. van Dam
2022 conf
CinC
Narimane Gassa, Machteld J. Boonstra, Beata Ondrusova, Jana Svehlíková, Dana H. Brooks, Akil Narayan, Ali S. Rababah, Peter M. van Dam, Rob S. MacLeod, Jess D. Tate, Nejib Zemzemi
2022 conf
CinC
Manon Kloosterman, Machteld J. Boonstra, Folkert W. Asselbergs, Peter Loh, Thom F. Oostendorp, Peter M. van Dam
2022 conf
CinC
Jeanne van der Waal, Veronique Meijborg, Machteld J. Boonstra, Thom F. Oostendorp, Ruben Coronel
2022 conf
CinC
Jess D. Tate, Nejib Zemzemi, Shireen Y. Elhabian, Beáta Ondrusová, Machteld J. Boonstra, Peter M. van Dam, Akil Narayan, Dana H. Brooks, Rob S. MacLeod
2022 conf
CinC
Beata Ondrusova, Machteld J. Boonstra, Jana Svehlíková, Dana H. Brooks, Peter M. van Dam, Ali S. Rababah, Akil Narayan, Rob S. MacLeod, Nejib Zemzemi, Jess D. Tate
2021 conf
CinC
Machteld J. Boonstra, Dana H. Brooks, Peter Loh, Peter M. van Dam
2021 conf
FIMH
Jess D. Tate, Wilson W. Good, Nejib Zemzemi, Machteld J. Boonstra, Peter M. van Dam, Dana H. Brooks, Akil Narayan, Rob S. MacLeod
2021 conf
CinC
Manon Kloosterman, Machteld J. Boonstra, Feddo P. Kirkels, Cornelis H. Slump, Peter Loh, Peter M. van Dam
2020 conf
CinC
Machteld J. Boonstra, Rob W. Roudijk, Peter Loh, Peter M. van Dam
2020 conf
CinC
Janna Ruisch, Machteld J. Boonstra, Rob W. Roudijk, Peter M. van Dam, Cornelis H. Slump, Peter Loh
2020 conf
CinC
Peter M. van Dam, Machteld J. Boonstra, Rob Roudijk, Marijke P. M. Linschoten, Emanuela T. Locati, Giuseppe Ciconte, Valeria Borrelli, Vincenzo Santinelli, G. Vicedomini, M. M. Monasky, Emanuele Micaglio, Luigi Giannelli, Valerio Mecarocci, Zarko Calovic, Carlo Pappone, Peter Loh
2019 conf
CinC
Machteld J. Boonstra, Rob W. Roudijk, Peter Loh, Peter M. van Dam
redb/extractors/decompiler/apk/smali_normalization.py
← Index redb/extractors/decompiler/apk/smali_normalization.py python
"""Semantic normalization of Dalvik/smali instructions.

Analogous to Binary Ninja's LLIL normalization: strips register allocation
noise and instruction encoding variants while preserving semantic operations.

Three normalization levels (most aggressive to most detailed):
  - 'category':    semantic category only (MOV, ALU, CALL, ...)
  - 'opcode':      base opcode, width-invariant (add, sub, invoke, ...)
  - 'opcode_api':  opcode category + API method/field references for
                   invoke/field/alloc instructions (default for MinHash)

References:
  - Smali+ 12-category reduction (Canfora et al.)
  - MOSDroid opcode family grouping
  - DroidSIFT/DroidSim API-sensitive similarity
"""

import re
from typing import List

# ---------------------------------------------------------------------------
# Dalvik opcode -> semantic category mapping
# ---------------------------------------------------------------------------
# Prefix-matched against instruction opcodes. Order matters for overlapping
# prefixes (longer/more-specific prefixes should come first in iteration,
# but since we use startswith and break on first match, we order by
# specificity within the list).

OPCODE_CATEGORIES = {
    # Arithmetic/logic
    "add": "ALU", "sub": "ALU", "mul": "ALU", "div": "ALU",
    "rem": "ALU", "and": "ALU", "or": "ALU", "xor": "ALU",
    "shl": "ALU", "shr": "ALU", "ushr": "ALU", "neg": "ALU",
    "not": "ALU",
    # Data movement
    "move": "MOV", "const": "CONST",
    # Memory access (field/array)
    "iget": "LOAD", "sget": "LOAD", "aget": "LOAD",
    "iput": "STORE", "sput": "STORE", "aput": "STORE",
    # Invocations
    "invoke": "CALL",
    # Control flow
    "if": "BRANCH", "goto": "JMP",
    "switch": "SWITCH",
    "return": "RET",
    # Object/type
    "new": "ALLOC", "check": "TYPE", "instance": "TYPE",
    # Array
    "fill": "ARR", "array": "ARR",
    # Comparison
    "cmpl": "CMP", "cmpg": "CMP", "cmp": "CMP",
    # Exception / synchronization
    "throw": "EXC", "monitor": "SYNC",
    # Conversion (int-to-long, float-to-int, etc.)
    "int-to": "CONV", "long-to": "CONV", "float-to": "CONV",
    "double-to": "CONV",
}

# Pre-compiled regexes for operand extraction
_METHOD_REF_RE = re.compile(r"(L[\w/$]+;->[\w<>]+\(.*?\)[\w/$;\[]*)")
_FIELD_REF_RE = re.compile(r"(L[\w/$]+;->[\w]+:[\w/$;\[]+)")
_CLASS_REF_RE = re.compile(r"(L[\w/$]+;)")
_CONST_STRING_RE = re.compile(r'^const-string(?:/jumbo)?\s')


def categorize_opcode(opcode: str) -> str:
    """Map a Dalvik opcode to its semantic category.

    Prefix-matched: 'add-int/2addr' matches 'add' -> 'ALU'.
    Returns 'OTHER' for unrecognized opcodes.
    """
    for prefix, cat in OPCODE_CATEGORIES.items():
        if opcode.startswith(prefix):
            return cat
    return "OTHER"


# Mapping from semantic categories to the ACFG feature vector indices
# used by Binary Ninja's build_block_features (cfg_features.py).
# This enables cross-platform ACFG feature comparison.
CATEGORY_TO_ACFG_INDEX = {
    "ALU": 0,       # CAT_ARITHMETIC
    "CONV": 0,      # arithmetic-adjacent
    "CMP": 4,       # CAT_COMPARISON
    "MOV": 2,       # CAT_TRANSFER
    "CONST": 2,     # transfer-adjacent (loading constants)
    "LOAD": 5,      # CAT_MEMORY
    "STORE": 5,     # CAT_MEMORY
    "CALL": 3,      # CAT_CALL
    "BRANCH": 1,    # CAT_LOGIC (conditional logic)
    "JMP": 1,       # CAT_LOGIC
    "SWITCH": 1,    # CAT_LOGIC
    "RET": 2,       # CAT_TRANSFER
    "ALLOC": 5,     # CAT_MEMORY (heap allocation)
    "TYPE": 6,      # CAT_OTHER
    "ARR": 5,       # CAT_MEMORY
    "EXC": 6,       # CAT_OTHER
    "SYNC": 6,      # CAT_OTHER
    "OTHER": 6,     # CAT_OTHER
}


def normalize_instruction(line: str, level: str = "opcode_api") -> str:
    """Normalize a single smali instruction line.

    Args:
        line: A single smali instruction (whitespace-stripped).
        level: Normalization level:
            'category'   - most aggressive: just semantic category
            'opcode'     - base opcode only, width/addressing-mode invariant
            'opcode_api' - category + API references for invoke/field/alloc
                          (default, best for MinHash similarity)

    Returns:
        Normalized instruction string, or empty string for non-instructions.
    """
    stripped = line.strip()
    if not stripped:
        return ""

    parts = stripped.split(None, 1)
    opcode = parts[0]
    operands = parts[1] if len(parts) > 1 else ""

    if level == "category":
        return categorize_opcode(opcode)

    if level == "opcode":
        # Strip type/width suffixes for invariance:
        # add-int, add-long, add-float -> 'add'
        # add-int/2addr -> 'add'
        base = re.split(r"[-/]", opcode)[0]
        return base

    if level == "opcode_api":
        # const-string: preserve string content (encrypted strings are a
        # key malware indicator)
        if _CONST_STRING_RE.match(stripped):
            # Extract the string literal
            str_match = re.search(r'"(.*)"', operands)
            if str_match:
                return f"CONST_STR \"{str_match.group(1)}\""
            return "CONST_STR"

        # invoke-*: preserve method reference
        if opcode.startswith("invoke"):
            ref = _METHOD_REF_RE.search(operands)
            if ref:
                return f"CALL {ref.group(1)}"
            return "CALL"

        # Field access: preserve field reference
        if opcode.startswith(("iget", "iput", "sget", "sput")):
            ref = _FIELD_REF_RE.search(operands)
            if ref:
                cat = "LOAD" if "get" in opcode else "STORE"
                return f"{cat} {ref.group(1)}"
            # Fallback: try space-separated format from androguard
            # e.g. "iget v0, p0, Lcom/Foo;->field Ljava/lang/String;"
            space_ref = re.search(
                r"(L[\w/$]+;->[\w]+)\s+([\w/$;\[]+)", operands
            )
            if space_ref:
                cat = "LOAD" if "get" in opcode else "STORE"
                return f"{cat} {space_ref.group(1)}:{space_ref.group(2)}"
            cat = "LOAD" if "get" in opcode else "STORE"
            return cat

        # new-instance: preserve allocated type
        if opcode.startswith("new-instance") or opcode == "new-array":
            ref = _CLASS_REF_RE.search(operands)
            if ref:
                return f"ALLOC {ref.group(1)}"
            return "ALLOC"

        # Everything else: just the category
        return categorize_opcode(opcode)

    # Unknown level: return raw opcode
    return opcode


def normalize_method_body(
    body: str, level: str = "opcode_api"
) -> List[str]:
    """Normalize all instructions in a smali method body.

    Filters out directives (.), labels (:), comments (#), and blank lines.
    Returns a list of normalized instruction strings.

    Args:
        body: Raw smali method body text.
        level: Normalization level (see normalize_instruction).

    Returns:
        List of normalized instruction strings (no empty strings).
    """
    normalized = []
    for line in body.split("\n"):
        stripped = line.strip()
        # Skip non-instructions
        if not stripped:
            continue
        if stripped.startswith((".",":", "#")):
            continue
        result = normalize_instruction(stripped, level)
        if result:
            normalized.append(result)
    return normalized