Ona De Gibert Bonet

12 papers B 3Journal 4Unranked 5
YearRankTypeTitle / Venue / Authors
2025 conf
ACL (1)
Laurie Burchell, Ona De Gibert Bonet, Nikolay Arefyev, Mikko Aulamo, Marta Bañón, Pinzhen Chen, Mariia Fedorova, Liane Guillou, Barry Haddow, Jan Hajic, Jindrich Helcl, Erik Henriksson, Mateusz Klimaszewski, Ville Komulainen, Andrey Kutuzov, Joona Kytöniemi, Veronika Laippala, Petter Mæhlum, Bhavitvya Malik, Farrokh Mehryary, Vladislav Mikhailov, Nikita Moghe, Amanda Myntti, Dayyán O'Brien, Stephan Oepen, Proyag Pal, Jousia Piha, Sampo Pyysalo, Gema Ramírez-Sánchez, David Samuel, Pavel Stepachev, Jörg Tiedemann, Dusan Varis, Tereza Vojtechová, Jaume Zaragoza-Bernabeu
2024 conf
EAMT (2)
Nikolay Arefyev, Mikko Aulamo, Pinzhen Chen, Ona De Gibert Bonet, Barry Haddow, Jindrich Helcl, Bhavitvya Malik, Gema Ramírez-Sánchez, Pavel Stepachev, Jörg Tiedemann, Dusan Varis, Jaume Zaragoza-Bernabeu
2024 conf
EACL (Demonstrations)
Timothee Mickus, Stig-Arne Grönroos, Joseph Attieh, Michele Boggia, Ona De Gibert Bonet, Shaoxiong Ji, Niki Andreas Lopi, Alessandro Raganato, Raúl Vázquez, Jörg Tiedemann
2022 B conf
LREC
Jordi Armengol-Estapé, Ona De Gibert Bonet, Maite Melero
2022 B conf
LREC
Ona De Gibert Bonet, Aitor García-Pablos, Montse Cuadros, Maite Melero
2022 B conf
LREC
Ona De Gibert Bonet, Iakes Goenaga, Jordi Armengol-Estapé, Olatz Perez-de-Viñaspre, Carla Parra Escartín, Marina Sanchez, Marcis Pinnis, Gorka Labaka, Maite Melero
2021 conf
ACL/IJCNLP (Findings)
Jordi Armengol-Estapé, Casimiro Pio Carrino, Carlos Rodríguez Penagos, Ona De Gibert Bonet, Carme Armentano-Oller, Aitor Gonzalez-Agirre, Maite Melero, Marta Villegas
2021 J jnl
CoRR
Jordi Armengol-Estapé, Casimiro Pio Carrino, Carlos Rodríguez Penagos, Ona De Gibert Bonet, Carme Armentano-Oller, Aitor Gonzalez-Agirre, Maite Melero, Marta Villegas
2021 J jnl
CoRR
Jordi Armengol-Estapé, Ona De Gibert Bonet, Maite Melero
2021 J jnl
CoRR
Casimiro Pio Carrino, Jordi Armengol-Estapé, Ona De Gibert Bonet, Asier Gutiérrez-Fandiño, Aitor Gonzalez-Agirre, Martin Krallinger, Marta Villegas
2021 J jnl
CoRR
Carlos Rodríguez Penagos, Carme Armentano-Oller, Marta Villegas, Maite Melero, Aitor Gonzalez, Ona De Gibert Bonet, Casimiro Pio Carrino
2021 conf
WMT@EMNLP
Ksenia Kharitonova, Ona De Gibert Bonet, Jordi Armengol-Estapé, Mar Rodriguez i Alvarez, Maite Melero
tests/unit/test_decompile_medium_level.py
← Index tests/unit/test_decompile_medium_level.py python
# tests/unit/test_decompile_medium_level.py
"""Unit tests (mocked BN) for bninja/analysis/medium_level.py
   and bninja/analysis/medium_level_normalization.py."""
# tests/unit/test_decompile_medium_level.py
import sys
from unittest.mock import MagicMock, patch

# Installa gli stubs BN
from tests.unit.conftest_binja_stubs import install_binja_stubs
install_binja_stubs()

# ── Definisci MockMLILInstruction PRIMA di importare il modulo ──
class MockMLILInstruction:
    def __init__(self, operation, address=0, operands=None):
        self.operation = operation
        self.address = address
        self.operands = operands or []

# ── Patcha il modulo BN in modo che isinstance() funzioni ──
sys.modules["binaryninja"].MediumLevelILInstruction = MockMLILInstruction
sys.modules["binaryninja"].SSAVariable = type("SSAVariable", (), {})
sys.modules["binaryninja"].Variable = type("Variable", (), {})
sys.modules["binaryninja"].ILIntrinsic = type("ILIntrinsic", (), {})

# Ora importa il modulo — vede già i tipi corretti
from redb.extractors.decompiler.bninja.analysis.medium_level_normalization import (
    MediumLevelNormalization,
)	

class MockMLILFunction:
    def __init__(self, instructions):
        self._instructions = instructions

    @property
    def instructions(self):
        return iter(self._instructions)

    @property
    def basic_blocks(self):
        # one block containing all instructions, good enough for MinHasher
        block = MagicMock()
        block.__iter__ = lambda self_: iter([])  # not used by MediumLevelAnalysis
        return [block]


class MockFunction:
    def __init__(self, name="func", start=0x1000, mlil=None):
        self.name = name
        self.start = start
        self.mlil = mlil



class TestMediumLevelNormalization:
    def setup_method(self):
        from redb.extractors.decompiler.bninja.analysis.medium_level_normalization import (
            MediumLevelNormalization,
        )
        self.norm = MediumLevelNormalization()

    def test_normalize_skeleton_single_instruction(self):
        il = MockMLILInstruction(operation=42, operands=[])
        result = self.norm.normalize_instruction_all_levels(il)
        assert result == [42]

    def test_normalize_skeleton_nested(self):
        inner = MockMLILInstruction(operation=7, operands=[])
        outer = MockMLILInstruction(operation=1, operands=[inner])
        result = self.norm.normalize_instruction_all_levels(outer)
        assert result == [1, 7]

    def test_normalize_skeleton_with_list_operand(self):
        inner_a = MockMLILInstruction(operation=10, operands=[])
        inner_b = MockMLILInstruction(operation=11, operands=[])
        outer = MockMLILInstruction(operation=2, operands=[[inner_a, inner_b]])
        result = self.norm.normalize_instruction_all_levels(outer)
        assert result == [2, 10, 11]

    def test_normalize_skeleton_none(self):
        result = self.norm.normalize_instruction_all_levels(None)
        # collect on None should leave ops empty
        assert result == []

    def test_normalize_typed_appends_leaf_types(self):
        # operand is a plain int -> "CONST"
        il = MockMLILInstruction(operation=3, operands=[42])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [3, "CONST"]

    def test_normalize_typed_bool_before_int(self):
        # bool must be detected before int (since bool is an int subclass)
        il = MockMLILInstruction(operation=4, operands=[True])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [4, "BOOL"]

    def test_normalize_typed_float(self):
        il = MockMLILInstruction(operation=5, operands=[1.5])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [5, "FLOAT_CONST"]

    def test_normalize_typed_str(self):
        il = MockMLILInstruction(operation=6, operands=["hello"])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [6, "STR"]

    def test_normalize_typed_unknown_falls_back_to_typename(self):
        class Weird:
            pass
        il = MockMLILInstruction(operation=8, operands=[Weird()])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [8, "WEIRD"]

    def test_normalize_typed_nested_mlil(self):
        inner = MockMLILInstruction(operation=99, operands=[7])
        outer = MockMLILInstruction(operation=1, operands=[inner])
        result = self.norm.normalize_instr_with_operands(outer)
        assert result == [1, 99, "CONST"]

    def test_normalize_typed_list_mixed(self):
        inner = MockMLILInstruction(operation=50, operands=[])
        il = MockMLILInstruction(operation=2, operands=[[inner, 99]])
        result = self.norm.normalize_instr_with_operands(il)
        assert result == [2, 50, "CONST"]


class TestMediumLevelAnalysis:
    def _make_analysis(self, instructions=None, mlil=True, start=0x1000):
        from redb.extractors.decompiler.bninja.analysis.medium_level import (
            MediumLevelAnalysis,
        )
        mlil_func = MockMLILFunction(instructions or []) if mlil else None
        func = MockFunction(name="testfunc", start=start, mlil=mlil_func)
        bv = MagicMock()
        return MediumLevelAnalysis(func, bv, MagicMock())

    def test_collect_returns_empty_when_no_mlil(self):
        a = self._make_analysis(mlil=False)
        sk, sk_addr, ty, ty_addr = a._collect_mlil_skeleton_and_typed()
        assert sk == [] and sk_addr == [] and ty == [] and ty_addr == []

    def test_collect_skeleton_and_typed_basic(self):
        instrs = [
            MockMLILInstruction(operation=1, address=0x1000, operands=[]),
            MockMLILInstruction(operation=2, address=0x1004, operands=[42]),
        ]
        a = self._make_analysis(instructions=instrs, start=0x1000)
        sk, sk_addr, ty, ty_addr = a._collect_mlil_skeleton_and_typed()

        assert sk == [[1], [2]]
        assert ty == [[1], [2, "CONST"]]
        assert sk_addr == [(0, [1]), (4, [2])]
        assert ty_addr == [(0, [1]), (4, [2, "CONST"])]

    def test_collect_negative_offset_clamped_to_zero(self):
        instrs = [
            MockMLILInstruction(operation=1, address=0x900, operands=[]),
        ]
        a = self._make_analysis(instructions=instrs, start=0x1000)
        _, sk_addr, _, ty_addr = a._collect_mlil_skeleton_and_typed()
        assert sk_addr[0][0] == 0
        assert ty_addr[0][0] == 0

    def test_log_error_records_entry(self):
        a = self._make_analysis()
        a.log_error("boom", "fname", 0x1234, ValueError("x"), "loc")
        assert len(a.errors) == 1
        err = a.errors[0]
        assert err["function_name"] == "fname"
        assert err["function_address"] == "4660"  # hex 0x1234
        assert err["error_location"] == "loc"
        assert err["error_message"] == "boom"
        assert err["error_type"] == "ValueError"
        assert "timestamp" in err

    @patch(
        "redb.extractors.decompiler.bninja.analysis.medium_level.MinHasher"
    )
    def test_analyze_returns_expected_keys(self, mock_minhasher):
        mock_minhasher.return_value.calculateMinHash.return_value = [1, 2, 3]

        instrs = [
            MockMLILInstruction(operation=1, address=0x1000, operands=[]),
            MockMLILInstruction(operation=2, address=0x1004, operands=[42]),
            MockMLILInstruction(operation=3, address=0x1008, operands=[]),
        ]
        a = self._make_analysis(instructions=instrs, start=0x1000)
        result, errors = a.analyze()

        expected_keys = {
            "function_address",
            "body_mlil_skeleton_vector",
            "sha256_mlil_skeleton",
            "tlsh_mlil_skeleton",
            "minhash_mlil_skeleton",
            "body_mlil_typed_vector",
            "sha256_mlil_typed",
            "tlsh_mlil_typed",
            "minhash_mlil_typed",
        }
        assert set(result.keys()) == expected_keys
        assert result["function_address"] == 0x1000
        assert result["minhash_mlil_skeleton"] == [1, 2, 3]
        assert result["minhash_mlil_typed"] == [1, 2, 3]
        assert errors == []

    @patch(
        "redb.extractors.decompiler.bninja.analysis.medium_level.MinHasher"
    )
    def test_analyze_empty_mlil(self, mock_minhasher):
        mock_minhasher.return_value.calculateMinHash.return_value = []
        a = self._make_analysis(mlil=False)
        result, errors = a.analyze()
        assert result["body_mlil_skeleton_vector"] == []
        assert result["body_mlil_typed_vector"] == []
        assert errors == []

    @patch(
        "redb.extractors.decompiler.bninja.analysis.medium_level.MinHasher"
    )
    def test_analyze_sha256_differs_skeleton_vs_typed(self, mock_minhasher):
        mock_minhasher.return_value.calculateMinHash.return_value = []

        instrs = [
            MockMLILInstruction(operation=1, address=0x1000, operands=[42]),
            MockMLILInstruction(operation=2, address=0x1004, operands=["foo"]),
            MockMLILInstruction(operation=3, address=0x1008, operands=[True]),
        ]
        a = self._make_analysis(instructions=instrs)
        result, _ = a.analyze()
        # skeleton ignores operand leaves, typed includes them -> different hashes
        assert result["sha256_mlil_skeleton"] != result["sha256_mlil_typed"]


class TestMinHasherMLILKinds:
    def _make_func(self, instrs):
        # MinHasher iterates basic_blocks then over each block
        block = MagicMock()
        block.__iter__ = lambda self_: iter(instrs)
        f = MagicMock()
        f.basic_blocks = [block]
        return f

    def test_mlil_skeleton_uses_medium_normalizer(self):
        from redb.extractors.decompiler.bninja.similarity.minhasher import (
            MinHasher, TokenKind,
        )
        instrs = [
            MockMLILInstruction(operation=i, operands=[]) for i in range(5)
        ]
        func = self._make_func(instrs)
        hasher = MinHasher(seed=42, il_function=func, kind=TokenKind.MLIL)
        result = hasher.calculateMinHash()
        # 5 instructions -> 3 trigrams -> non-empty signature
        assert result != []

    def test_typed_mlil_differs_from_skeleton(self):
        from redb.extractors.decompiler.bninja.similarity.minhasher import (
            MinHasher, TokenKind,
        )
        instrs = [
            MockMLILInstruction(operation=1, operands=[42]),
            MockMLILInstruction(operation=2, operands=["s"]),
            MockMLILInstruction(operation=3, operands=[True]),
            MockMLILInstruction(operation=4, operands=[1.5]),
        ]
        func = self._make_func(instrs)
        skel = MinHasher(seed=42, il_function=func, kind=TokenKind.MLIL).calculateMinHash()
        typed = MinHasher(seed=42, il_function=func, kind=TokenKind.TYPED_MLIL).calculateMinHash()
        # Same seed, same instructions, but typed has extra leaf tokens
        # -> hashes should generally differ
        assert skel != typed

    def test_mlil_too_few_instructions(self):
        from redb.extractors.decompiler.bninja.similarity.minhasher import (
            MinHasher, TokenKind,
        )
        instrs = [MockMLILInstruction(operation=1, operands=[])] * 2
        func = self._make_func(instrs)
        hasher = MinHasher(seed=42, il_function=func, kind=TokenKind.MLIL)
        assert hasher.calculateMinHash() == []

    def test_unsupported_kind_raises(self):
        from redb.extractors.decompiler.bninja.similarity.minhasher import MinHasher
        func = self._make_func([])
        hasher = MinHasher(seed=42, il_function=func, kind="bogus")
        with pytest.raises(ValueError):
            hasher.calculateMinHash()