Nan Li

39 papers A 6B 3C 3Misc 1Journal 19Unranked 7
YearRankTypeTitle / Venue / Authors
2025 conf
ITiCSE (1)
William Billingsley, Ljiljana Brankovic, Nan Li, David J. Paul, Amin Sakzad, Matthew P. Skerritt, Judithe Sheard
2025 J jnl
Comput. J.
Nan Li, Yingjiu Li, Yangguang Tian
2024 conf
SIGCSE (1)
Amin Sakzad, David J. Paul, Judithe Sheard, Ljiljana Brankovic, Matthew P. Skerritt, Nan Li, Sepehr Minagar, Simon, William Billingsley
2024 A conf
ICDCS
Cody Lewis, Vijay Varadharajan, Nasimul Noman, Udaya Kiran Tupakula, Nan Li
2024 J jnl
IEEE Trans. Inf. Forensics Secur.
Jinguang Han, Willy Susilo, Nan Li, Xinyi Huang
2024 J jnl
Comput. J.
Yangguang Tian, Yingjiu Li, Robert H. Deng, Guomin Yang, Nan Li
2024 J jnl
Comput. J.
Nan Li, Yingjiu Li, Mark Manulis, Yangguang Tian, Guomin Yang
2023 B conf
CANS
Nan Li, Yingjiu Li, Atsuko Miyaji, Yangguang Tian, Tsz Hon Yuen
2023 J jnl
Comput. J.
Yangguang Tian, Atsuko Miyaji, Koki Matsubara, Hui Cui, Nan Li
2023 J jnl
IEEE Internet Things J.
Cody Lewis, Nan Li, Vijay Varadharajan
2023 J jnl
J. Imaging
Yang-Wai Chow, Willy Susilo, Yannan Li, Nan Li, Chau Nguyen
2022 J jnl
Sensors
Fariza Sabrina, Nan Li, Shaleeza Sohail
2021 J jnl
IACR Cryptol. ePrint Arch.
Nan Li, Yingjiu Li, Atsuko Miyaji, Yangguang Tian, Tsz Hon Yuen
2021 J jnl
J. Netw. Comput. Appl.
Yang-Wai Chow, Willy Susilo, Jianfeng Wang, Richard Buckland, Joonsang Baek, Jongkil Kim, Nan Li
2020 conf
TPS-ISA
Nan Li, Dongxi Liu, Surya Nepal, Guangyu Pei
2020 J jnl
Comput. J.
Yangguang Tian, Yingjiu Li, Robert H. Deng, Nan Li, Guomin Yang, Zheng Yang
2020 J jnl
J. Comput. Secur.
Yangguang Tian, Yingjiu Li, Robert H. Deng, Nan Li, Pengfei Wu, Anyi Liu
2020 J jnl
Theor. Comput. Sci.
Yangguang Tian, Yingjiu Li, Binanda Sengupta, Nan Li, Chunhua Su
2020 A conf
ACSAC
Yangguang Tian, Nan Li, Yingjiu Li, Pawel Szalachowski, Jianying Zhou
2019 J jnl
IACR Cryptol. ePrint Arch.
Jongkil Kim, Willy Susilo, Fuchun Guo, Joonsang Baek, Nan Li
2019 B conf
ACNS
Jongkil Kim, Willy Susilo, Fuchun Guo, Joonsang Baek, Nan Li
2019 B conf
CANS
Yangguang Tian, Yingjiu Li, Binanda Sengupta, Nan Li, Yong Yu
2019 A conf
ICDCS
Nan Li, Vijay Varadharajan, Surya Nepal
2019 C conf
WOWMOM
Jiannan Wei, Nan Li
2019 conf
ML4CS
Yang-Wai Chow, Willy Susilo, Jianfeng Wang, Richard Buckland, Joonsang Baek, Jongkil Kim, Nan Li
2018 J jnl
IEEE Access
Jiannan Wei, Xiaojie Wang, Nan Li, Guomin Yang, Yi Mu
2018 C conf
ISPEC
Yangguang Tian, Yingjiu Li, Yinghui Zhang, Nan Li, Guomin Yang, Yong Yu
2018 conf
SecureComm (1)
Yangguang Tian, Yingjiu Li, Rongmao Chen, Nan Li, Ximeng Liu, Bing Chang, Xingjie Yu
2017 J jnl
IACR Cryptol. ePrint Arch.
Dongxi Liu, Nan Li, Jongkil Kim, Surya Nepal
2017 A conf
ICDCS
Nan Li, Fuchun Guo, Yi Mu, Willy Susilo, Surya Nepal
2017 J jnl
IEEE Trans. Sustain. Comput.
Nan Li, Dongxi Liu, Surya Nepal
2015 A conf
AsiaCCS
Nan Li, Yi Mu, Willy Susilo, Vijay Varadharajan
2015 J jnl
Comput. Stand. Interfaces
Nan Li, Yi Mu, Willy Susilo, Vijay Varadharajan
2015 J jnl
Secur. Commun. Networks
Nan Li, Yi Mu, Willy Susilo, Fuchun Guo, Vijay Varadharajan
2014 conf
RFIDSec
Nan Li, Yi Mu, Willy Susilo, Fuchun Guo, Vijay Varadharajan
2013 conf
RFIDSec Asia
Nan Li, Yi Mu, Willy Susilo, Fuchun Guo, Vijay Varadharajan
2013 C conf
ISPEC
Nan Li, Yi Mu, Willy Susilo, Vijay Varadharajan
2011 Misc conf
Inscrypt
Nan Li, Yi Mu, Willy Susilo
2011 A conf
AsiaCCS
Nan Li, Yi Mu, Willy Susilo, Fuchun Guo
redb/extractors/ioc_extractor/ioc_extractor.py
← Index redb/extractors/ioc_extractor/ioc_extractor.py python
"""
IOC Extractor - Extractor class for extracting IOCs from decompilation results.

This extractor works with in-memory data from DecompileBinja, following the
standard Extractor pattern to support both ClickHouse and PrintExporter (dry-run).

Usage:
    # After DecompileBinja completes:
    ioc_extractor = IOCExtractorFromResults(
        analysis_results=decompiler.analysis_results,
        sha256=sha256,
        log=logger,
        exporters=exporters,
        index_prefix=index_prefix
    )
    ioc_extractor.export_data()
"""

import inspect
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, List, Dict, Optional

from redb.extractors.enum import Tag
from redb.extractors.database_exporters import DatabaseExporter

# Import the IOCScraper and related classes from standalone module
from redb.extractors.ioc_extractor.standalone_ioc_extractor import (
    IOCScraper,
    IOCType,
    SourceType,
    ExtractedIOC,
)
from typing import Set


class IOCExtractorFromResults:
    """
    Extracts IOCs from in-memory decompilation results.

    This follows a simplified Extractor pattern but doesn't inherit from Extractor
    since it doesn't read from a binary file - instead it takes already-processed
    analysis results from DecompileBinja.
    """

    def __init__(
        self,
        analysis_results: Dict[str, Any],
        sha256: str,
        log: Any,
        exporters: Optional[List[DatabaseExporter]] = None,
        index_prefix: Optional[str] = None,
        tld_file: Optional[Path] = None,
        suppress_types: Optional[Set[IOCType]] = None,
        js_context: bool = False,
    ):
        """
        Initialize IOC Extractor with analysis results.

        Args:
            analysis_results: Dict containing 'strings' and 'decompiled' lists from DecompileBinja
            sha256: Sample SHA256 hash
            log: Logger instance
            exporters: List of database exporters (ClickHouse, Print, etc.)
            index_prefix: Index prefix for database
            tld_file: Optional path to TLD list file
            js_context: When True, the underlying IOCScraper rejects FQDN
                candidates that match JS object-access syntax (see
                JS_FP_TLDS / JS_FP_SLDS). Set this for the JS pipeline only;
                APK suppresses FQDN entirely via suppress_types and binary
                callers leave it disabled.
        """
        self.log = log
        self.log.debug(f"Creating {self.__class__.__name__}")
        self.analysis_results = analysis_results
        self.sha256 = sha256
        self.exporters = exporters or []
        self.index_prefix = index_prefix
        self.scraper = IOCScraper(
            tld_file, suppress_types=suppress_types, js_context=js_context,
        )
        self.extracted_iocs: List[ExtractedIOC] = []

    def extract(self) -> List[ExtractedIOC]:
        """
        Extract IOCs from strings and decompiled functions in analysis_results.

        Returns:
            List of ExtractedIOC objects
        """
        self.log.debug(inspect.currentframe().f_code.co_name)
        self.extracted_iocs = []

        # Extract from strings
        strings_count = self._extract_from_strings()

        # Extract from decompiled functions
        functions_count = self._extract_from_decompiled()

        # Extract from text-based artefact surfaces (JS, PowerShell, etc.)
        text_count = self._extract_from_text()

        self.log.info(
            f"Extracted {len(self.extracted_iocs)} IOCs for {self.sha256[:16]}... "
            f"(strings: {strings_count}, functions: {functions_count}, "
            f"text: {text_count})"
        )

        return self.extracted_iocs

    def _extract_from_strings(self) -> int:
        """Extract IOCs from sample's strings."""
        count = 0
        strings = self.analysis_results.get("strings", [])

        for s in strings:
            string_value = s.get("string", "")
            string_offset = s.get("string_offset", 0)

            if isinstance(string_value, bytes):
                string_value = string_value.decode('utf-8', errors='replace')

            for ioc in self.scraper.scrape(string_value, SourceType.STRING, str(string_offset)):
                self.extracted_iocs.append(ioc)
                count += 1

        return count

    def _extract_from_decompiled(self) -> int:
        """Extract IOCs from sample's decompiled functions.

        Supports both Binja format (key: "decompiled", fields: "decompiled_function",
        "decompiled_function_hash", "function_type") and APK format (key:
        "decompiled_content", fields: "decompiled_method", "decompiled_method_hash",
        "method_type").
        """
        count = 0

        # Binja format
        decompiled = self.analysis_results.get("decompiled", [])
        for func in decompiled:
            func_type = func.get("function_type", "UNKNOWN")
            if func_type in ("LIBRARY", "THUNK"):
                continue

            func_content = func.get("decompiled_function", "")
            func_hash = func.get("decompiled_function_hash", "unknown")

            if isinstance(func_content, bytes):
                func_content = func_content.decode('utf-8', errors='replace')

            for ioc in self.scraper.scrape(func_content, SourceType.DECOMPILED_FUNCTION, func_hash):
                self.extracted_iocs.append(ioc)
                count += 1

        # APK format (decompiled_content with method-level fields)
        decompiled_content = self.analysis_results.get("decompiled_content", [])
        for func in decompiled_content:
            func_type = func.get("method_type", "UNKNOWN")
            if func_type in ("LIBRARY", "THUNK"):
                continue

            func_content = func.get("decompiled_method", "")
            func_hash = func.get("decompiled_method_hash", "unknown")

            if isinstance(func_content, bytes):
                func_content = func_content.decode('utf-8', errors='replace')

            for ioc in self.scraper.scrape(func_content, SourceType.DECOMPILED_FUNCTION, func_hash):
                self.extracted_iocs.append(ioc)
                count += 1

        return count

    def _extract_from_text(self) -> int:
        """Extract IOCs from text-based artefact surfaces.

        Walks `analysis_results["text_raw"]` and `analysis_results["text_normalized"]`,
        each a list of `{"content": str, "content_hash": str}` dicts. Each
        list is routed through its own SourceType (`TEXT_RAW` /
        `TEXT_NORMALIZED`) so analysts can distinguish IOCs that were already
        present in the raw source from those exposed only after normalisation
        (deobfuscation/beautification). Generic across text-based formats —
        used by JS today, intended for PowerShell, Python, email body,
        extracted PDF/Office text in the future.
        """
        count = 0

        for key, source_type in (
            ("text_raw", SourceType.TEXT_RAW),
            ("text_normalized", SourceType.TEXT_NORMALIZED),
        ):
            for entry in self.analysis_results.get(key, []):
                content = entry.get("content", "")
                content_hash = entry.get("content_hash", "unknown")

                if isinstance(content, bytes):
                    content = content.decode('utf-8', errors='replace')

                for ioc in self.scraper.scrape(content, source_type, content_hash):
                    self.extracted_iocs.append(ioc)
                    count += 1

        return count

    def prepare_export_data(self, exporter_type: str) -> Any:
        """
        Prepare data for specific export type.

        Returns tuple for ClickHouse or list of dicts for Print/Elasticsearch.
        """
        self.log.debug(inspect.currentframe().f_code.co_name)

        if not self.extracted_iocs:
            return None

        now = datetime.now(timezone.utc)

        if exporter_type == "ClickHouseExporter":
            data = [
                [
                    self.sha256,
                    ioc.ioc_type.value,
                    ioc.ioc_value,
                    ioc.source_type.value,
                    ioc.source_identifier,
                    now,
                ]
                for ioc in self.extracted_iocs
            ]

            column_names = [
                "sha256",
                "ioc_type",
                "ioc_value",
                "source_type",
                "source_identifier",
                "extracted_at",
            ]

            column_type_names = [
                "FixedString(64)",
                "Enum8('ipv4'=1, 'ipv6'=2, 'fqdn'=3, 'url'=4, 'email'=5, 'server'=6, "
                "'hash_md5'=10, 'hash_sha1'=11, 'hash_sha256'=12, 'cve'=20, 'cwe'=21, 'cpe'=22, "
                "'crypto_btc'=30, 'crypto_eth'=31, 'crypto_xrp'=32, 'crypto_bch'=33, "
                "'crypto_ada'=34, 'crypto_substrate'=35, 'path_linux'=40, 'path_windows'=41, "
                "'registry_key'=42, 'onion'=50)",
                "String",
                "Enum8('decompiled_function'=1, 'disassembled_function'=2, 'string'=3, "
                "'text_raw'=4, 'text_normalized'=5)",
                "String",
                "DateTime64(3, 'UTC')",
            ]

            return (data, column_names, column_type_names)

        else:
            # For PrintExporter and others - return list of dicts
            return [
                {
                    "sha256": self.sha256,
                    "ioc_type": ioc.ioc_type.value,
                    "ioc_value": ioc.ioc_value,
                    "source_type": ioc.source_type.value,
                    "source_identifier": ioc.source_identifier,
                    "extracted_at": now.isoformat(),
                }
                for ioc in self.extracted_iocs
            ]

    def get_clickhouse_table(self) -> str:
        """Return the ClickHouse table name for IOCs."""
        return "redb_iocs"

    def tag(self) -> str:
        """Return the tag for this extractor."""
        return Tag.IOC.value if hasattr(Tag, 'IOC') else "ioc"

    def export_data(self) -> bool:
        """
        Export extracted IOCs to all configured exporters.

        Returns:
            True if export succeeded, False if failed, None if no data
        """
        self.log.debug(inspect.currentframe().f_code.co_name)

        # First extract the IOCs
        extracted = self.extract()

        if not extracted:
            self.log.debug("No IOCs extracted, skipping export")
            return None

        success = True

        from redb.extractors.database_exporters import PrintExporter, ClickHouseExporter

        for exporter in self.exporters:
            try:
                if isinstance(exporter, PrintExporter):
                    # For PrintExporter, pass the list of dicts
                    export_data = self.prepare_export_data("PrintExporter")
                    success &= exporter.export(export_data)

                elif isinstance(exporter, ClickHouseExporter):
                    # For ClickHouse, pass tuple with table info
                    export_data = self.prepare_export_data("ClickHouseExporter")
                    if export_data:
                        success &= exporter.export(
                            export_data,
                            table=self.get_clickhouse_table(),
                            column_names=export_data[1],
                            column_type_names=export_data[2]
                        )

            except Exception as e:
                self.log.error(f"Error exporting IOCs to {exporter.__class__.__name__}: {e}")
                success = False

        return success