Kan Li

34 papers B 1C 3Journal 21Unranked 9
YearRankTypeTitle / Venue / Authors
2026 J jnl
Pattern Recognit.
Rama Bastola Neupane, Kan Li, Zhuqing Mao
2025 J jnl
Knowl. Based Syst.
Rama Bastola Neupane, Kan Li, Zhuqing Mao
2025 J jnl
Artif. Intell. Rev.
Rama Bastola Neupane, Kan Li, Tesfaye Fenta Boka
2025 J jnl
Comput. Electron. Agric.
Jiacheng Jiang, Tiemin Zhang, Kan Li
2025 J jnl
CoRR
Shenxi Liu, Kan Li, Mingyang Zhao, Yuhang Tian, Bin Li, Shoujun Zhou, Hongliang Li, Fuxia Yang
2025 J jnl
CoRR
Shenxi Liu, Kan Li, Mingyang Zhao, Yuhang Tian, Shoujun Zhou, Bin Li
2025 J jnl
Comput. Electron. Agric.
Jiacheng Jiang, Tiemin Zhang, Jikang Yang, Hongfeng Deng, Kan Li
2024 J jnl
Signal Image Video Process.
Yang Li, Xin Liu, Kan Li
2024 conf
ACL (Short Papers)
Bin Sun, Jianfeng Li, Hao Zhou, Fandong Meng, Kan Li, Jie Zhou
2024 J jnl
Comput. Biol. Medicine
Mengxian Jia, Jiaxin Lai, Kan Li, Jiyang Chen, Kelun Huang, Chaohui Ding, Ziwei Fan, Zongjie Yuan, Honglin Teng
2023 J jnl
Sensors
Shengran Su, Zhenyuan Xu, Xiang He, Chanling Yin, Miao Kong, Xuyuan Zhang, Yi Ruan, Kan Li, Qiang Lin
2023 conf
NLPCC (1)
Jiawen Liu, Kan Li
2023 J jnl
CoRR
Jiawen Liu, Kan Li
2023 J jnl
Database J. Biol. Databases Curation
Jarjapu Mahita, Brendan Ha, Anais Gambiez, Sharon L. Schendel, Haoyang Li, Kathryn M. Hastie, S. Moses Dennison, Kan Li, Natalia Kuzmina, Sivakumar Periasamy, Alexander Bukreyev, Jennifer E. Munt, Mary Osei-Twum, Caroline Atyeo, James A. Overton, Randi Vita, Hector Guzman-Orozco, Marcus Mendes, Mari Kojima, Peter J. Halfmann, Yoshihiro Kawaoka, Galit Alter, Luc Gagnon, Ralph S. Baric, Georgia D. Tomaras, Tim Germann, Daniel Bedinger, Jason A. Greenbaum, Erica Ollmann Saphire, Bjoern Peters
2023 C conf
iSPEC
Yuzhe Huang, Fan Mo, Zitong Zhang, Chenghan Li, Kan Li
2022 C conf
ICMLA
Murtadha D. Hssayeni, Arash Andalib, Rishabh Singh, Diego Pava, Kan Li, Steven Borzak, Robert Chait, Kaustubh Kale
2021 J jnl
Sensors
Dongmei Li, Chaofan Weng, Yi Ruan, Kan Li, Guoan Cai, Chenyao Song, Qiang Lin
2021 J jnl
Neurocomputing
Jing Wang, Kan Li
2020 conf
ICCS (7)
SuChang Li, Kan Li, Ilyes Kacher, Yuichiro Taira, Bungo Yanatori, Imari Sato
2020 conf
ICONIP (4)
Meng Wang, Kan Li
2020 J jnl
Sensors
Yujin Zhang, Ni Luan, Kan Li, Jiancai Leng, Wei Hu
2019 J jnl
Comput. Stat. Data Anal.
Kan Li, Sheng Luo
2019 J jnl
IEEE Access
Arshad Ahmad, Chong Feng, Kan Li, Syed Mohammad Asim, Tingting Sun
2018 J jnl
IEEE Access
Arshad Ahmad, Kan Li, Chong Feng, Syed Mohammad Asim, Abdallah Yousif, Shi Ge
2018 J jnl
npj Digit. Medicine
Stephen P. Lee, Grace Ha, Donald E. Wright, Yinji Ma, Ellora Sen-Gupta, Natalie R. Haubrich, Paul C. Branche, Weihua Li, Gilbert L. Huppert, Matthew Johnson, Hakan B. Mutlu, Kan Li, Nirav Sheth, John A. Wright, Yonggang Huang, Moussa Mansour, John A. Rogers, Roozbeh Ghaffari
2016 J jnl
KSII Trans. Internet Inf. Syst.
Gang Xiong, Penghao Sun, Yuxiang Hu, Julong Lan, Kan Li
2016 J jnl
Neurocomputing
Jialiang Chen, Xin Li, William K. Cheung, Kan Li
2011 C conf
CIS
Chunsheng Li, Kan Li
2011 conf
FSKD
Kan Li, Fenglan Yao, Ruipeng Liu
2010 B conf
ISIT
Kan Li, Aleksandar Kavcic, Raman Venkataramani, Mehmet Fatih Erden
2010 conf
ICC
Kan Li, Aleksandar Kavcic, Mehmet Fatih Erden
2010 conf
IPCV
Kan Li, Jian Cao, Kai Zhang
2009 conf
CIS (1)
Jian Cao, Kan Li, Chunxiao Gao, Qiongxin Liu
2001 conf
CDC
Dahua Xie, Kan Li, Qinghe Wu
redb/extractors/extractor.py
← Index redb/extractors/extractor.py python
import hashlib
import inspect
from abc import ABCMeta, abstractmethod
from dataclasses import asdict
from functools import cached_property
from datetime import datetime, timezone
import math
from typing import Counter, List, Optional, Dict, Any, Tuple
from redb import settings
from redb.models.dataclasses import Hash
from .database_exporters import DatabaseExporter, ElasticsearchExporter, ClickHouseExporter, PrintExporter

class Extractor(metaclass=ABCMeta):
    def __init__(
        self,
        filepath: str,
        log: Any,
        exporters: Optional[List[DatabaseExporter]] = None,
        index_prefix: Optional[str] = None,
        source: Optional[str] = None,
        elastic_index: Optional[str] = None,
        known_benign: bool = False,
        known_malicious: bool = False,
        precomputed_hashes: Optional[Dict[str, str]] = None,
    ):
        self.log = log
        self.log.debug(f"Creating {self.__class__.__name__}")
        self.filepath = filepath
        self.source = source
        self.exporters = exporters or []
        self.index_prefix = index_prefix if index_prefix else settings.ELASTIC_BINARIES_COLLECTION
        self.elastic_index = self.index_prefix + (elastic_index if elastic_index else "")
        self.known_benign = known_benign
        self.known_malicious = known_malicious

        # Use precomputed hashes if provided (e.g., from machofile, pefile)
        # Otherwise compute them from binary
        if precomputed_hashes:
            self.md5 = precomputed_hashes.get('md5') or precomputed_hashes.get('MD5')
            self.sha1 = precomputed_hashes.get('sha1') or precomputed_hashes.get('SHA1')
            self.sha256 = precomputed_hashes.get('sha256') or precomputed_hashes.get('SHA256')
        else:
            self.md5 = hashlib.md5(self.binary).hexdigest()
            self.sha1 = hashlib.sha1(self.binary).hexdigest()
            self.sha256 = hashlib.sha256(self.binary).hexdigest()
        self.hash = Hash(self.md5, self.sha1, self.sha256)

    @cached_property
    def binary(self):
        with open(self.filepath, "rb") as f:
            data = f.read()
        return data

    @property
    @abstractmethod
    def tag(self):
        pass

    @abstractmethod
    def extract(self):
        """
        this method defines the extracted data
        """

    @staticmethod
    def process_binary_string(s):
        # Remove \x00 padding
        s = s.rstrip(b"\x00")

        # Check if there are any non-printable characters
        has_non_printable = any(byte < 32 or byte > 126 for byte in s)

        if not has_non_printable:
            # If all characters are printable, decode the string
            return s.decode()
        else:
            # If there are non-printable characters, represent them as \xDD
            return "".join(
                [
                    f"\\x{byte:02x}" if byte < 32 or byte > 126 else chr(byte)
                    for byte in s
                ]
            )

    @staticmethod
    def remove_non_utf8(binary_string):
        decoded = b""
        for i in range(len(binary_string)):
            try:
                # Try to decode each byte
                char = binary_string[i : i + 1].decode("utf-8")
                decoded += char.encode("utf-8")
            except UnicodeDecodeError:
                # Skip this byte if it can't be decoded
                continue
        return decoded

    def calculate_entropy(self, data):
        """Calculate the entropy of a chunk of data.
        Based on pefile.SectionStructure.entropy_H.
        """
        # self.log.debug(inspect.currentframe().f_code.co_name)
        if not data:
            return 0.0

        if type(data) == str:
            counts = Counter(data)
            frequencies = ((i / len(data)) for i in counts.values())
            return - sum(f * math.log(f, 2) for f in frequencies)
        else:
            occurences = Counter(bytearray(data))
            entropy = 0
            for x in occurences.values():
                p_x = float(x) / len(data)
                entropy -= p_x * math.log(p_x, 2)
            return entropy

    @abstractmethod
    def prepare_export_data(self, exporter_type: str) -> Tuple[List[Any], List[str], List[str]]:
        """
        Prepare data for specific export type
        Returns:
            Tuple containing:
            - data: List of values to insert
            - column_names: List of column names
            - column_type_names: List of column types
        """
        pass

    def export_data(self):
        """Export data to all configured exporters

        Returns:
            True: Export succeeded
            False: Export failed (actual error)
            None: No data to export (not an error, e.g., no overlay, no signature)
        """
        self.log.debug(inspect.currentframe().f_code.co_name)
        success = True
        extracted_data = self.extract()

        if extracted_data is None:
            self.log.debug("extract() returned None, skipping export")
            return None  # No data to export, not a failure
            
        for exporter in self.exporters:
            if isinstance(exporter, PrintExporter):
                # For PrintExporter, we pass the extracted data directly
                success &= exporter.export(extracted_data)
            else:
                # Get the data prepared for this specific exporter type
                export_data = self.prepare_export_data(exporter.__class__.__name__)
                
                if export_data is None:
                    self.log.debug(f"prepare_export_data returned None for {exporter.__class__.__name__}")
                    return False
                
                if isinstance(exporter, ElasticsearchExporter):
                    success &= exporter.export(
                        export_data,
                        index=self.elastic_index,
                        tag=self.tag(),
                        hashes=asdict(self.hash),
                        known_benign=self.known_benign,
                        known_malicious=self.known_malicious
                    )
                elif isinstance(exporter, ClickHouseExporter):
                    # For ClickHouse, we need to pass the table name and the prepared data
                    success &= exporter.export(
                        export_data,
                        table=self.get_clickhouse_table(),
                        # Add these parameters explicitly
                        column_names=export_data[1] if isinstance(export_data, tuple) else None,
                        column_type_names=export_data[2] if isinstance(export_data, tuple) else None
                    )
                
        return success

    @abstractmethod
    def get_clickhouse_table(self) -> str:
        """Return the appropriate ClickHouse table name"""
        pass
    # def export_to_elastic(self, list_of_dataclasses, tag=None):
    #     self.log.debug(inspect.currentframe().f_code.co_name)

    #     if not isinstance(list_of_dataclasses, list):
    #         self.log.error("Called export_to_elastic wrongly")

    #     now_t = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M:%S")
    #     for dataclass_ in list_of_dataclasses:
    #         if not settings.ELASTIC_CLIENT.ping():
    #             self.log.error("[CONNECTION ERROR] ping to elastic failed")
            
    #         # Convert dataclass to dict and filter out None values
    #         document = {k: v for k, v in asdict(dataclass_).items() if v is not None}
            
    #         if tag:
    #             document["tag"] = [tag, self.tag()]
    #         else:
    #             document["tag"] = self.tag()
    #         hashes = asdict(self.hash)
    #         document |= hashes
    #         document["timestamp_utc"] = now_t
    #         # document["source"] = self.source
    #         document["known_benign"] = self.known_benign
    #         document["known_malicious"] = self.known_malicious

    #         if "_id" in document:
    #             tmp_id = document.pop("_id") + document["sha256"]
    #             _id = hashlib.sha256(tmp_id.encode()).hexdigest()
    #         else:
    #             _id = document["sha256"]

    #         # self.log.debug(f"[DEBUG] about to export {type(document)} {document}")
    #         try:
    #             doc_dump = json.dumps(document)
    #         except TypeError as e:
    #             self.log.error(
    #                 f"Failed export of document. " f"full document: {document}"
    #             )
    #             raise e

    #         # body={"doc": doc_dump,
    #         #       "doc_as_upsert": True  # Create the document if it doesn't exist
    #         # }

    #         # Check if the index exists, and create it if it doesn't
    #         # if not settings.ELASTIC_CLIENT.indices.exists(index=self.elastic_index):
    #         #     settings.ELASTIC_CLIENT.indices.create(index=self.elastic_index)
    #         # pprint(doc_dump) #DEBUG
    #         # print("[DEBUG] _id: " + _id)
    #         # print("[DEBUG] index: " + self.elastic_index)
    #         settings.ELASTIC_CLIENT.index(
    #             index=self.elastic_index, id=_id, document=doc_dump
    #         )