Oleksandra Levchenko

12 papers A 1C 2Journal 4Unranked 5
YearRankTypeTitle / Venue / Authors
2021 J jnl
Knowl. Inf. Syst.
Oleksandra Levchenko, Boyan Kolev, Djamel Edine Yagoubi, Reza Akbarinia, Florent Masseglia, Themis Palpanas, Dennis E. Shasha, Patrick Valduriez
2021 J jnl
Distributed Parallel Databases
Pavlos Kranas, Boyan Kolev, Oleksandra Levchenko, Esther Pacitti, Patrick Valduriez, Ricardo Jiménez-Peris, Marta Patiño-Martínez
2019 conf
ECML/PKDD (3)
Oleksandra Levchenko, Boyan Kolev, Djamel Edine Yagoubi, Dennis E. Shasha, Themis Palpanas, Patrick Valduriez, Reza Akbarinia, Florent Masseglia
2019 C conf
CLOSER
Boyan Kolev, Reza Akbarinia, Ricardo Jiménez-Peris, Oleksandra Levchenko, Florent Masseglia, Marta Patiño, Patrick Valduriez
2019 C conf
DATA
Boyan Kolev, Reza Akbarinia, Ricardo Jiménez-Peris, Oleksandra Levchenko, Florent Masseglia, Marta Patiño, Patrick Valduriez
2018 J jnl
Data Min. Knowl. Discov.
Djamel Edine Yagoubi, Reza Akbarinia, Boyan Kolev, Oleksandra Levchenko, Florent Masseglia, Patrick Valduriez, Dennis E. Shasha
2018 conf
IEEE BigData
Boyan Kolev, Oleksandra Levchenko, Esther Pacitti, Patrick Valduriez, Ricardo Vilaça, Rui C. Gonçalves, Ricardo Jiménez-Peris, Pavlos Kranas
2018 A conf
CIKM
Oleksandra Levchenko, Djamel Edine Yagoubi, Reza Akbarinia, Florent Masseglia, Boyan Kolev, Dennis E. Shasha
2016 conf
IEEE BigData
Boyan Kolev, Raquel Pau, Oleksandra Levchenko, Patrick Valduriez, Ricardo Jiménez-Peris, José Pereira
2016 conf
CLOSER (1)
Boyan Kolev, Carlyna Bondiombouy, Oleksandra Levchenko, Patrick Valduriez, Ricardo Jiménez-Peris, Raquel Pau, José Pereira
2016 J jnl
Trans. Large Scale Data Knowl. Centered Syst.
Carlyna Bondiombouy, Boyan Kolev, Oleksandra Levchenko, Patrick Valduriez
2015 conf
DEXA (1)
Carlyna Bondiombouy, Boyan Kolev, Oleksandra Levchenko, Patrick Valduriez
tests/unit/test_decompile_medium_level.py
← Index tests/unit/test_decompile_medium_level.py python
# tests/unit/test_decompile_medium_level.py
"""Unit tests (mocked BN) for bninja/analysis/medium_level.py
   and bninja/analysis/medium_level_normalization.py."""
# tests/unit/test_decompile_medium_level.py
import sys
from unittest.mock import MagicMock, patch

# Installa gli stubs BN
from tests.unit.conftest_binja_stubs import install_binja_stubs
install_binja_stubs()

# ── Definisci MockMLILInstruction PRIMA di importare il modulo ──
class MockMLILInstruction:
    def __init__(self, operation, address=0, operands=None):
        self.operation = operation
        self.address = address
        self.operands = operands or []

# ── Patcha il modulo BN in modo che isinstance() funzioni ──
sys.modules["binaryninja"].MediumLevelILInstruction = MockMLILInstruction
sys.modules["binaryninja"].SSAVariable = type("SSAVariable", (), {})
sys.modules["binaryninja"].Variable = type("Variable", (), {})
sys.modules["binaryninja"].ILIntrinsic = type("ILIntrinsic", (), {})

# Ora importa il modulo — vede già i tipi corretti
from redb.extractors.decompiler.bninja.analysis.medium_level_normalization import (
    MediumLevelNormalization,
)	

class MockMLILFunction:
    def __init__(self, instructions):
        self._instructions = instructions

    @property
    def instructions(self):
        return iter(self._instructions)

    @property
    def basic_blocks(self):
        # one block containing all instructions, good enough for MinHasher
        block = MagicMock()
        block.__iter__ = lambda self_: iter([])  # not used by MediumLevelAnalysis
        return [block]


class MockFunction:
    def __init__(self, name="func", start=0x1000, mlil=None):
        self.name = name
        self.start = start
        self.mlil = mlil



class TestMediumLevelNormalization:
    def setup_method(self):
        from redb.extractors.decompiler.bninja.analysis.medium_level_normalization import (
            MediumLevelNormalization,
        )
        self.norm = MediumLevelNormalization()

    def test_normalize_skeleton_single_instruction(self):
        il = MockMLILInstruction(operation=42, operands=[])
        result = self.norm.normalize_instruction_all_levels(il)
        assert result == [42]

    def test_normalize_skeleton_nested(self):
        inner = MockMLILInstruction(operation=7, operands=[])
        outer = MockMLILInstruction(operation=1, operands=[inner])
        result = self.norm.normalize_instruction_all_levels(outer)
        assert result == [1, 7]

    def test_normalize_skeleton_with_list_operand(self):
        inner_a = MockMLILInstruction(operation=10, operands=[])
        inner_b = MockMLILInstruction(operation=11, operands=[])
        outer = MockMLILInstruction(operation=2, operands=[[inner_a, inner_b]])
        result = self.norm.normalize_instruction_all_levels(outer)
        assert result == [2, 10, 11]

    def test_normalize_skeleton_none(self):
        result = self.norm.normalize_instruction_all_levels(None)
        # collect on None should leave ops empty
        assert result == []

    def test_normalize_typed_appends_leaf_types(self):
        # operand is a plain int -> "CONST"
        il = MockMLILInstruction(operation=3, operands=[42])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [3, "CONST"]

    def test_normalize_typed_bool_before_int(self):
        # bool must be detected before int (since bool is an int subclass)
        il = MockMLILInstruction(operation=4, operands=[True])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [4, "BOOL"]

    def test_normalize_typed_float(self):
        il = MockMLILInstruction(operation=5, operands=[1.5])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [5, "FLOAT_CONST"]

    def test_normalize_typed_str(self):
        il = MockMLILInstruction(operation=6, operands=["hello"])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [6, "STR"]

    def test_normalize_typed_unknown_falls_back_to_typename(self):
        class Weird:
            pass
        il = MockMLILInstruction(operation=8, operands=[Weird()])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [8, "WEIRD"]

    def test_normalize_typed_nested_mlil(self):
        inner = MockMLILInstruction(operation=99, operands=[7])
        outer = MockMLILInstruction(operation=1, operands=[inner])
        result = self.norm.normalize_instr_with_operands(outer)
        assert result == [1, 99, "CONST"]

    def test_normalize_typed_list_mixed(self):
        inner = MockMLILInstruction(operation=50, operands=[])
        il = MockMLILInstruction(operation=2, operands=[[inner, 99]])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [2, 50, "CONST"]


class TestMediumLevelAnalysis:
    def _make_analysis(self, instructions=None, mlil=True, start=0x1000):
        from redb.extractors.decompiler.bninja.analysis.medium_level import (
            MediumLevelAnalysis,
        )
        mlil_func = MockMLILFunction(instructions or []) if mlil else None
        func = MockFunction(name="testfunc", start=start, mlil=mlil_func)
        bv = MagicMock()
        return MediumLevelAnalysis(func, bv, MagicMock())

    def test_collect_returns_empty_when_no_mlil(self):
        a = self._make_analysis(mlil=False)
        sk, sk_addr, ty, ty_addr = a._collect_mlil_skeleton_and_typed()
        assert sk == [] and sk_addr == [] and ty == [] and ty_addr == []

    def test_collect_skeleton_and_typed_basic(self):
        instrs = [
            MockMLILInstruction(operation=1, address=0x1000, operands=[]),
            MockMLILInstruction(operation=2, address=0x1004, operands=[42]),
        ]
        a = self._make_analysis(instructions=instrs, start=0x1000)
        sk, sk_addr, ty, ty_addr = a._collect_mlil_skeleton_and_typed()

        assert sk == [[1], [2]]
        assert ty == [[1], [2, "CONST"]]
        assert sk_addr == [(0, [1]), (4, [2])]
        assert ty_addr == [(0, [1]), (4, [2, "CONST"])]

    def test_collect_negative_offset_clamped_to_zero(self):
        instrs = [
            MockMLILInstruction(operation=1, address=0x900, operands=[]),
        ]
        a = self._make_analysis(instructions=instrs, start=0x1000)
        _, sk_addr, _, ty_addr = a._collect_mlil_skeleton_and_typed()
        assert sk_addr[0][0] == 0
        assert ty_addr[0][0] == 0

    def test_log_error_records_entry(self):
        a = self._make_analysis()
        a.log_error("boom", "fname", 0x1234, ValueError("x"), "loc")
        assert len(a.errors) == 1
        err = a.errors[0]
        assert err["function_name"] == "fname"
        assert err["function_address"] == "4660"  # hex 0x1234
        assert err["error_location"] == "loc"
        assert err["error_message"] == "boom"
        assert err["error_type"] == "ValueError"
        assert "timestamp" in err

    @patch(
        "redb.extractors.decompiler.bninja.analysis.medium_level.MinHasher"
    )
    def test_analyze_returns_expected_keys(self, mock_minhasher):
        mock_minhasher.return_value.calculateMinHash.return_value = [1, 2, 3]

        instrs = [
            MockMLILInstruction(operation=1, address=0x1000, operands=[]),
            MockMLILInstruction(operation=2, address=0x1004, operands=[42]),
            MockMLILInstruction(operation=3, address=0x1008, operands=[]),
        ]
        a = self._make_analysis(instructions=instrs, start=0x1000)
        result, errors = a.analyze()

        expected_keys = {
            "function_address",
            "body_mlil_skeleton_vector",
            "sha256_mlil_skeleton",
            "tlsh_mlil_skeleton",
            "minhash_mlil_skeleton",
            "body_mlil_typed_vector",
            "sha256_mlil_typed",
            "tlsh_mlil_typed",
            "minhash_mlil_typed",
        }
        assert set(result.keys()) == expected_keys
        assert result["function_address"] == 0x1000
        assert result["minhash_mlil_skeleton"] == [1, 2, 3]
        assert result["minhash_mlil_typed"] == [1, 2, 3]
        assert errors == []

    @patch(
        "redb.extractors.decompiler.bninja.analysis.medium_level.MinHasher"
    )
    def test_analyze_empty_mlil(self, mock_minhasher):
        mock_minhasher.return_value.calculateMinHash.return_value = []
        a = self._make_analysis(mlil=False)
        result, errors = a.analyze()
        assert result["body_mlil_skeleton_vector"] == []
        assert result["body_mlil_typed_vector"] == []
        assert errors == []

    @patch(
        "redb.extractors.decompiler.bninja.analysis.medium_level.MinHasher"
    )
    def test_analyze_sha256_differs_skeleton_vs_typed(self, mock_minhasher):
        mock_minhasher.return_value.calculateMinHash.return_value = []

        instrs = [
            MockMLILInstruction(operation=1, address=0x1000, operands=[42]),
            MockMLILInstruction(operation=2, address=0x1004, operands=["foo"]),
            MockMLILInstruction(operation=3, address=0x1008, operands=[True]),
        ]
        a = self._make_analysis(instructions=instrs)
        result, _ = a.analyze()
        # skeleton ignores operand leaves, typed includes them -> different hashes
        assert result["sha256_mlil_skeleton"] != result["sha256_mlil_typed"]


class TestMinHasherMLILKinds:
    def _make_func(self, instrs):
        # MinHasher iterates basic_blocks then over each block
        block = MagicMock()
        block.__iter__ = lambda self_: iter(instrs)
        f = MagicMock()
        f.basic_blocks = [block]
        return f

    def test_mlil_skeleton_uses_medium_normalizer(self):
        from redb.extractors.decompiler.bninja.similarity.minhasher import (
            MinHasher, TokenKind,
        )
        instrs = [
            MockMLILInstruction(operation=i, operands=[]) for i in range(5)
        ]
        func = self._make_func(instrs)
        hasher = MinHasher(seed=42, il_function=func, kind=TokenKind.MLIL)
        result = hasher.calculateMinHash()
        # 5 instructions -> 3 trigrams -> non-empty signature
        assert result != []

    def test_typed_mlil_differs_from_skeleton(self):
        from redb.extractors.decompiler.bninja.similarity.minhasher import (
            MinHasher, TokenKind,
        )
        instrs = [
            MockMLILInstruction(operation=1, operands=[42]),
            MockMLILInstruction(operation=2, operands=["s"]),
            MockMLILInstruction(operation=3, operands=[True]),
            MockMLILInstruction(operation=4, operands=[1.5]),
        ]
        func = self._make_func(instrs)
        skel = MinHasher(seed=42, il_function=func, kind=TokenKind.MLIL).calculateMinHash()
        typed = MinHasher(seed=42, il_function=func, kind=TokenKind.TYPED_MLIL).calculateMinHash()
        # Same seed, same instructions, but typed has extra leaf tokens
        # -> hashes should generally differ
        assert skel != typed

    def test_mlil_too_few_instructions(self):
        from redb.extractors.decompiler.bninja.similarity.minhasher import (
            MinHasher, TokenKind,
        )
        instrs = [MockMLILInstruction(operation=1, operands=[])] * 2
        func = self._make_func(instrs)
        hasher = MinHasher(seed=42, il_function=func, kind=TokenKind.MLIL)
        assert hasher.calculateMinHash() == []

    def test_unsupported_kind_raises(self):
        from redb.extractors.decompiler.bninja.similarity.minhasher import MinHasher
        func = self._make_func([])
        hasher = MinHasher(seed=42, il_function=func, kind="bogus")
        with pytest.raises(ValueError):
            hasher.calculateMinHash()