Chaitanya Kulkarni

19 papers Misc 1Journal 12Unranked 6
YearRankTypeTitle / Venue / Authors
2026 J jnl
CoRR
Yirou Ge, Yixi Li, Alec Chiu, Shivani Shekhar, Zijie Pan, Avinash Thangali, Yun-Shiuan Chuang, Chaitanya Kulkarni, Uma Kona, Linsey Pang, Prakhar Mehrotra
2026 conf
ISDFS
Chandrashekhar Medicherla, Chaitanya Kulkarni, Viswanathan Ranganathan, Milan Parikh, Vinay Soni
2026 J jnl
SoftwareX
Bharath Irigireddy, Varaprasad Bandaru, Sachin Velmurugan, Chaitanya Kulkarni
2026 J jnl
CoRR
Yun-Shiuan Chuang, Chaitanya Kulkarni, Alec Chiu, Avinash Thangali, Zijie Pan, Shivani Shekhar, Yirou Ge, Yixi Li, Uma Kona, Linsey Pang, Prakhar Mehrotra
2025 J jnl
Hum. Factors
Shiyu Deng, Chaitanya Kulkarni, Jinwoo Oh, Sarah Henrickson Parker, Nathan Lau
2025 J jnl
Int. J. Data Sci. Anal.
Chaitanya Kulkarni, Makarand Y. Naniwadekar, Yuldasheva Minavar Mirzaxmatovna, Shashikant V. Athawale, Mohit Kumar Bhadla, Haewon Byeon
2025 J jnl
CoRR
Ali Sahami, Sudhanshu Garg, Andrew Wang, Chaitanya Kulkarni, Farhad Farahani, Sean Yun-Shiuan Chuang, Jian Wan, Srinivasan Manoharan, Uma Kona, Nitin Sharma, Linsey Pang, Prakhar Mehrotra, Jessica Clark, Mark Moyou
2024 J jnl
Appl. Soft Comput.
Mayur Wankhade, Chaitanya Kulkarni, Annavarapu Chandra Sekhara Rao
2024 J jnl
BMC Medical Informatics Decis. Mak.
Chaitanya Kulkarni, Aadam Quraishi, Mohan Raparthi, Mohammad Shabaz, Muhammad Attique Khan, Raj A. Varma, Ismail Keshta, Mukesh Soni, Haewon Byeon
2023 conf
WWW (Companion Volume)
Yunzhong He, Cong Zhang, Ruoyan Kong, Chaitanya Kulkarni, Qing Liu, Ashish Gandhe, Amit Nithianandan, Arul Prakash
2023 J jnl
CoRR
Yunzhong He, Cong Zhang, Ruoyan Kong, Chaitanya Kulkarni, Qing Liu, Ashish Gandhe, Amit Nithianandan, Arul Prakash
2022 J jnl
Artif. Intell. Rev.
Mayur Wankhade, Annavarapu Chandra Sekhara Rao, Chaitanya Kulkarni
2022 conf
EMBC
Luoluo Liu, Dennis Swearingen, Eran Simhon, Chaitanya Kulkarni, David Noren, Ronny Mans
2021 conf
ACL/IJCNLP (1)
Chaitanya Kulkarni, Jany Chan, Eric Fosler-Lussier, Raghu Machiraju
2021 J jnl
CoRR
Luoluo Liu, Eran Simhon, Chaitanya Kulkarni, Ronny Mans
2018 conf
NAACL-HLT (2)
Chaitanya Kulkarni, Wei Xu, Alan Ritter, Raghu Machiraju
2018 J jnl
CoRR
Chaitanya Kulkarni, Wei Xu, Alan Ritter, Raghu Machiraju
2018 conf
Bench
Arjun Pandya, Chaitanya Kulkarni, Kunal Mali, Jianwu Wang
2018 Misc conf
PSB
Arunima Srivastava, Chaitanya Kulkarni, Parag Mallick, Kun Huang, Raghu Machiraju
redb/extractors/decompiler/bninja/analysis/cfg-old.py
← Index redb/extractors/decompiler/bninja/analysis/cfg-old.py python
from collections import deque
from enum import Enum

from binaryninja.enums import (
    BranchType,
    InstructionTextTokenType,
)

# Support both package and standalone imports
try:
    from ..utils.hashes import calculate_md5, calculate_sha256
except ImportError:
    # Fallback to absolute imports (for multiprocessing spawned processes)
    from redb.extractors.decompiler.bninja.utils.hashes import calculate_md5, calculate_sha256


class CFGAnalysis:
    def __init__(self, function):
        self.function = function

    def determine_block_type(self, block) -> str:
        """Determine the type of a basic block."""
        # Check if it's a thunk function (usually just a jump or call)
        if len(block.disassembly_text) <= 2 and any(
            "jmp" in line.tokens[0].text.lower() for line in block.disassembly_text
        ):
            return "THUNK"

        # Check if it contains only data (no valid instructions)
        if all(not line.tokens for line in block.disassembly_text):
            return "DATA"

        # Default to code
        return "CODE"

    def extract_cyclomatic_complexity(self):
        """
        Cyclomatic complexity (McCabe’s metric) measures the number of linearly independent paths
        through a function’s control flow graph (CFG).
        The standard formula is:

            M = E - N + 2

        where:
            - E = number of edges in the CFG
            - N = number of nodes (basic blocks)
            - 2 accounts for the entry and exit nodes of a single connected graph
        """
        if self.function is None:
            return 0

        # number of basic blocks
        num_blocks = len(self.function.basic_blocks)
        # number of edges in the graph
        num_edges = sum(
            len(basic_block.outgoing_edges)
            for basic_block in self.function.basic_blocks
        )
        return num_edges - num_blocks + 2

    def extract_function_cfg(self):
        """Extract information about a function CFG and return it as a dictionary."""

        function = self.function
        function_data = {
            "function_address": self.function.start,
            "blocks": [],
            "measures": {
                "cyclomatic_complexity": self.extract_cyclomatic_complexity(),
            },
        }

        if self.function is None:
            return function_data

        # Get the map of the depth associated to every block
        depths = self.get_map_depth()

        # Get the map of the positions associated to every block
        id_maps = self.get_block_id_map()

        # Extract block data with graph structure information
        for block in function.basic_blocks:
            # dominators per every block translated
            dominators = sorted(self.extract_dominators(block, id_maps))

            # post dominators
            post_dominators = sorted(self.extract_post_dominators(block, id_maps))

            # Build block instructions string
            block_instructions = "\n".join(str(line) for line in block.disassembly_text)

            # Determine block type
            block_type = self.determine_block_type(block)

            # Extract successors directly from basic block
            successor_blocks = [edge.target.start for edge in block.outgoing_edges]
            # We ensure a canonical order and we sort the edges
            successor_blocks.sort()

            # Extract predecessors directly from basic block
            predecessor_blocks = [edge.source.start for edge in block.incoming_edges]
            # We ensure a canonical order and we sort the edges
            predecessor_blocks.sort()

            # Determine branch type from outgoing edges
            branch_type = self.determine_branch_type(block)

            instructions_count = len(block.disassembly_text)

            # Create block record
            block_json = {
                "function_address": self.function.start,
                "block_start_address": block.start,
                "block_end_address": block.end,
                "block_size": block.end - block.start,
                "instructions_count": instructions_count,
                "block_instructions_hash": calculate_sha256(block_instructions),
                "predecessor_blocks": predecessor_blocks,
                "successor_blocks": successor_blocks,
                "depth": depths[block.start],
                "position": id_maps[block.start],
                "branch_type": branch_type,
                "block_type": block_type,
                "flags": self.extract_block_flags(block),
                "dominators": dominators,
                "post_dominators": post_dominators,
            }
            function_data["blocks"].append(block_json)

        return function_data

    def extract_dominators(self, bb, id_maps):
        """Extract the dominators normalized"""
        dom_idx = [id_maps[d.start] for d in bb.dominators]
        return dom_idx

    def extract_post_dominators(self, bb, id_maps):
        """Extract the post-dominators normalized"""
        post_dom_idx = [id_maps[d.start] for d in bb.post_dominators]
        return post_dom_idx

    def determine_branch_type(self, block):
        """
        Determine the type of branch at the end of a basic block.
        This combines edge type information with instruction analysis.
        """
        # If no outgoing edges, it might be a return or terminal block
        if not block.outgoing_edges:
            # Check if the last instruction is a return
            for line in reversed(list(block.disassembly_text)):
                if line.tokens and any(
                    token.text.lower() in ["ret", "retn"] for token in line.tokens
                ):
                    return "RETURN"
            return "UNKNOWN"

        # Collect branch types from all outgoing edges
        branch_types = []
        for edge in block.outgoing_edges:
            edge_type = edge.type
            # Map edge type to our branch type enum
            if isinstance(edge_type, str):
                if edge_type == "IndirectCall":
                    branch_types.append("CALL")
                else:
                    branch_types.append("UNKNOWN")
            else:
                # Use our mapping for integer/enum values
                type_mapping = {
                    BranchType.UnconditionalBranch: "DIRECT",
                    BranchType.FalseBranch: "CONDITIONAL",
                    BranchType.TrueBranch: "CONDITIONAL",
                    BranchType.CallDestination: "CALL",
                    BranchType.FunctionReturn: "RETURN",
                    BranchType.SystemCall: "CALL",
                    BranchType.IndirectBranch: "INDIRECT",
                    BranchType.ExceptionBranch: "UNKNOWN",
                    BranchType.UnresolvedBranch: "UNKNOWN",
                    BranchType.UserDefinedBranch: "UNKNOWN",
                }
                branch_types.append(type_mapping.get(edge_type, "UNKNOWN"))

        # Determine overall branch type (prioritize CALL > RETURN > CONDITIONAL > DIRECT)
        if "CALL" in branch_types:
            return "CALL"
        elif "RETURN" in branch_types:
            return "RETURN"
        elif "CONDITIONAL" in branch_types:
            return "CONDITIONAL"
        elif "DIRECT" in branch_types:
            return "DIRECT"
        elif len(block.outgoing_edges) == 1:
            return "FALLTHROUGH"

        # If edge analysis was inconclusive, fall back to instruction analysis
        last_instr = None
        for line in reversed(list(block.disassembly_text)):
            if line.tokens:
                last_instr = line
                break

        if last_instr:
            mnemonic = None
            for token in last_instr.tokens:
                if token.type == InstructionTextTokenType.InstructionToken:
                    mnemonic = token.text.lower()
                    break

            if mnemonic:
                if mnemonic == "call":
                    return "CALL"
                elif mnemonic == "jmp":
                    return "DIRECT"
                elif mnemonic.startswith("j") and mnemonic != "jmp":
                    return "CONDITIONAL"
                elif mnemonic in ["ret", "retn"]:
                    return "RETURN"

        return "UNKNOWN"

    def get_map_depth(self):
        """
        Run a BFS on the basic blocks of the function to assign a depth to every block
        """

        depths = {}
        entry = self.function.get_basic_block_at(self.function.start)

        ### Simple BFS
        q = deque()
        q.append(entry)
        depths[entry.start] = 0

        while q:
            b = q.popleft()
            b_depth = depths[b.start]
            for edge in b.outgoing_edges:
                tgt = edge.target

                if tgt is None:
                    continue

                if tgt.start not in depths:
                    depths[tgt.start] = b_depth + 1
                    q.append(tgt)

        return depths

    def get_block_id_map(self):
        """
        Assign a unique, sequential ID to each basic block of the function using a BFS starting from the entry block.
        """

        id_map = {}
        entry = self.function.get_basic_block_at(self.function.start)

        q = deque()
        q.append(entry)

        current_id = 0
        id_map[entry.start] = current_id

        while q:
            b = q.popleft()
            for edge in b.outgoing_edges:
                tgt = edge.target

                if tgt is None:
                    continue

                if tgt.start not in id_map:
                    current_id += 1
                    id_map[tgt.start] = current_id
                    q.append(tgt)

        return id_map

    def extract_block_flags(self, block):
        """
        Get the flags for every basic block. Currently, we implemented these heuristics:
            - if a basic block is the entry node for a function
            - if a basic block is the exit block for a function
            - if a basic block is part of a natural loop
        """
        flags = []

        if block.start == self.function.start:
            flags.append(BlockFlags.EntryBlock.value)

        if any(edge.type == BranchType.FunctionReturn for edge in block.outgoing_edges):
            flags.append(BlockFlags.ExitBlock.value)

        # if this block is in its dominance frontier, then it's part of a natural loop
        if block in block.dominance_frontier:
            flags.append(BlockFlags.LoopBlock.value)

        return flags


class BlockFlags(Enum):
    # generally, the basic block identifying the entry point of the function
    EntryBlock = "EntryBlock"
    # any basic blocks that makes the control flow exiting from the current function
    ExitBlock = "ExitBlock"
    # any block is in a natural loop if it is in its own dominance frontier
    LoopBlock = "LoopBlock"


class BlockType(Enum):
    THUNK = "THUNK"
    DATA = "DATA"
    PADDING = "PADDING"
    CODE = "CODE"