Maite Taboada

36 papers A* 1B 3Misc 1Journal 19Unranked 12
YearRankTypeTitle / Venue / Authors
2026 J jnl
AI Soc.
Evan Donahue, Matthew Canute, Adjua Akinwumi, Philippa R. Adams, Maite Taboada, Wendy Hui Kyong Chun
2023 J jnl
First Monday
Varada Kolhatkar, Nithum Thain, Jeffrey Sorensen, Lucas Dixon, Maite Taboada
2023 conf
EMNLP (Findings)
Matt Canute, Mali Jin, hannah holtzclaw, Alberto Lusoli, Philippa R. Adams, Mugdha Pandya, Maite Taboada, Diana Maynard, Wendy Hui Kyong Chun
2023 J jnl
CoRR
Matt Canute, Mali Jin, hannah holtzclaw, Alberto Lusoli, Philippa R. Adams, Mugdha Pandya, Maite Taboada, Diana Maynard, Wendy Hui Kyong Chun
2023 conf
Canadian AI
Valentin-Gabriel Soumah, Prashanth Rao, Philipp Eibl, Maite Taboada
2023 J jnl
CoRR
Valentin-Gabriel Soumah, Prashanth Rao, Philipp Eibl, Maite Taboada
2021 J jnl
Frontiers Artif. Intell.
Katharina Ehret, Maite Taboada
2021 J jnl
Frontiers Artif. Intell.
Prashanth Rao, Maite Taboada
2021 J jnl
Nat. Lang. Eng.
Salud María Jiménez-Zafra, Noa P. Cruz Díaz, Maite Taboada, María Teresa Martín-Valdivia
2020 J jnl
CoRR
Varada Kolhatkar, Nithum Thain, Jeffrey Sorensen, Lucas Dixon, Maite Taboada
2019 J jnl
Big Data Soc.
Fatemeh Torabi Asr, Maite Taboada
2018 J jnl
Comput. Linguistics
Farah Benamara, Diana Inkpen, Maite Taboada
2018 J jnl
Lang. Resour. Evaluation
Debopam Das, Maite Taboada
2018 conf
FEVER@EMNLP
Fatemeh Torabi Asr, Maite Taboada
2017 conf
ALW@ACL
Varada Kolhatkar, Maite Taboada
2017 J jnl
Comput. Linguistics
Farah Benamara, Maite Taboada, Yvette Yannick Mathieu
2017 J jnl
Knowl. Based Syst.
Joseph J. Thompson, Betty H. M. Leung, Mark R. Blair, Maite Taboada
2017 conf
NLPmJ@EMNLP
Varada Kolhatkar, Maite Taboada
2016 J jnl
J. Assoc. Inf. Sci. Technol.
Noa P. Cruz Díaz, Maite Taboada, Ruslan Mitkov
2016 B conf
ICWE
Rowan Hoogervorst, Erik Essink, Wouter Jansen, Max van den Helder, Kim Schouten, Flavius Frasincar, Maite Taboada
2016 J jnl
CoRR
Cliff Goddard, Maite Taboada, Radoslava Trnavac
2015 J jnl
Lang. Resour. Evaluation
Mikel Iruskieta, Iria da Cunha, Maite Taboada
2015 conf
*SEM@NAACL-HLT
Farah Benamara, Maite Taboada
2013 J jnl
Dialogue Discourse
Maite Taboada, Debopam Das
2013 conf
SIGDIAL Conference
Paula Christina Figueira Cardoso, Maite Taboada, Thiago Alexandre Salgueiro Pardo
2013 conf
STIL
Paula Christina Figueira Cardoso, Maite Taboada, Thiago A. S. Pardo
2012 B conf
LREC
Natalia Konstantinova, Sheila C. M. de Sousa, Noa P. Cruz Díaz, Manuel J. Maña López, Maite Taboada, Ruslan Mitkov
2011 J jnl
Comput. Linguistics
Maite Taboada, Julian Brooke, Milan Tofiloski, Kimberly D. Voll, Manfred Stede
2009 conf
ACL/IJCNLP (2)
Milan Tofiloski, Julian Brooke, Maite Taboada
2009 Misc conf
RANLP
Julian Brooke, Milan Tofiloski, Maite Taboada
2009 conf
SIGDIAL Conference
Maite Taboada, Julian Brooke, Manfred Stede
2007 conf
Australian Conference on Artificial Intelligence
Kimberly D. Voll, Maite Taboada
2006 B conf
LREC
Maite Taboada, Caroline Anthony, Kimberly D. Voll
2003 J jnl
Comput. Humanit.
Maite Taboada
1997 A* conf
ACL
Maite Taboada
1996 conf
ICSLP
Alon Lavie, Lori S. Levin, Yan Qu, Alex Waibel, Donna Gates, Marsal Gavaldà, Laura Mayfield, Maite Taboada
redb/extractors/pe_extractors/pe_sections.py
← Index redb/extractors/pe_extractors/pe_sections.py python
import base64
import hashlib
import inspect
from redb.extractors.enum import Tag
from redb.extractors.pe_extractor import PEExtractor
from redb.models.dataclasses import PESection
from datetime import datetime, timezone
from typing import Any


class PESectionExtractor(PEExtractor):

    def __init__(
        self,
        filepath,
        log,
        exporters=None,
        index_prefix=None,
        elastic_index=None,
        known_benign=False,
        known_malicious=False,
        pe=None,
    ):
        super().__init__(
            filepath,
            log,
            exporters,
            index_prefix,
            elastic_index,
            known_benign,
            known_malicious,
            pe,
        )
        self.elastic_index = self.index_prefix + "-pe_sections"
        self.log.debug(inspect.currentframe().f_code.co_name)

    def tag(self):
        return Tag.PE_SECTION.value

    def _extract_sections(self):
        self.log.debug(inspect.currentframe().f_code.co_name)
        sections = []
        for section in self.pe.sections:
            try:
                name = self.process_binary_string(section.Name)
            except Exception as e:
                name = "UnableToDecode"
                self.log.warning(
                    f'Unable to store section Name "{section.Name}" for {self.hash.sha256}'
                    f" exception {e}"
                )
            sec_sha256 = section.get_hash_sha256()
            sec_md5 = section.get_hash_md5()
            # sec_entropy = "%.2f" % section.get_entropy()
            sec_entropy = section.get_entropy()
            pe_section = PESection(
                _id=hashlib.sha256(
                    name.encode()
                ).hexdigest(),  # usecase 8e035beb02a411f8a9e92d4cf184ad34f52bbd0a81a50c222cdd4706e4e45104, all section have same sha256
                section_name=name,
                section_name_b64=base64.b64encode(
                    section.Name.rstrip(b'\x00')
                ).decode(),  # base64.b64decode(b64) to decode
                section_v_addr=section.VirtualAddress,
                section_v_addr_hex=hex(section.VirtualAddress),
                section_v_size=section.Misc_VirtualSize,
                section_size=section.SizeOfRawData,
                section_pointer_to_raw_data=hex(section.PointerToRawData),
                section_md5=sec_md5,
                section_sha256=sec_sha256,
                section_entropy=sec_entropy,
            )
            sections.append(pe_section)
        return sections

    def extract(self):
        self.log.debug(inspect.currentframe().f_code.co_name)
        try:
            sections = self._extract_sections()
            # self.export_to_elastic(sections)  # Let the exporters handle this
            return sections
        except Exception as e:
            self.log.error(f"Error extracting PE sections: {e}")
            return None

    def prepare_export_data(self, exporter_type: str) -> Any:
        if exporter_type == "ElasticsearchExporter":
            return self.extract()
        elif exporter_type == "ClickHouseExporter":
            sections = self.extract()
            if sections is None:
                return None
            
            data = []
            current_time = datetime.now(timezone.utc)
            
            for section in sections:
                data.append([
                    self.sha256,                          # sha256
                    self.md5,                             # md5
                    self.sha1,                            # sha1
                    section.section_name,                 # section_name
                    section.section_name_b64,             # section_name_b64
                    section.section_entropy,              # section_entropy
                    section.section_sha256,               # section_sha256
                    section.section_md5,                  # section_md5
                    section.section_size,                 # section_size
                    section.section_v_addr,               # section_v_addr
                    section.section_v_size,               # section_v_size
                    int(section.section_pointer_to_raw_data, 16),  # section_pointer_to_raw_data - convert from hex
                    current_time                          # analysis_date
                ])
            
            column_names = [
                'sha256', 'md5', 'sha1', 'section_name', 'section_name_b64',
                'section_entropy', 'section_sha256', 'section_md5', 'section_size',
                'section_v_addr', 'section_v_size', 'section_pointer_to_raw_data',
                'analysis_date'
            ]
            
            if not data:
                return None

            column_type_names = [
                'FixedString(64)', 'FixedString(32)', 'FixedString(40)',
                'LowCardinality(String)', 'LowCardinality(String)',
                'Float64', 'FixedString(64)', 'FixedString(32)', 'UInt64',
                'UInt64', 'UInt64', 'UInt64',
                'DateTime64(3, \'UTC\')'
            ]

            return (data, column_names, column_type_names)

    def get_clickhouse_table(self) -> str:
        return "redb_pe_sections"