Man F. Yan

12 papers Journal 1Unranked 11
YearRankTypeTitle / Venue / Authors
2022 conf
OFC
Benyuan Zhu, Tommy Geisler, Peter Ingo Borel, Rasmus V. Jensen, Matthias Stegmaier, Bera Palsdottir, David W. Peckham, Robert Lingle, Man F. Yan, David J. DiGiovanni
2020 conf
ECOC
Vitaly Mikhailov, Jiawei Luo, Daryl Inniss, Man F. Yan, Yingzhi Sun, Gabriel S. Puc, Robert S. Windeler, Paul S. Westbrook, Yuriy Dulashko, David J. DiGiovanni
2019 conf
OFC
Vitaly Mikhailov, Mikhail Melkumov, Daryl Inniss, Aleksandr Khegai, Konstantin Riumkin, Sergei Firstov, Fedor Afanasiev, Man F. Yan, Yingzhi Sun, Jiawei Luo, Gabriel S. Puc, Scott D. Shenk, Robert S. Windeler, Paul S. Westbrook, Robert L. Lingle, Evgeny M. Dianov, David J. DiGiovanni
2018 conf
ECOC
Robert Lingle, Kasyapa Balemarthy, Yi Sun, Roman Shubochkin, David W. Peckham, Alan McCurdy, Benyuan Zhu, Rasmus V. Jensen, Peter Ingo Borel, Tommy Geisler, Bera Palsdottir, Poul Kristensen, Man F. Yan, David Braganza, Durgesh Vaidya, David J. DiGiovanni
2018 conf
ECOC
Benyuan Zhu, Peter Ingo Borel, Tommy Geisler, Rasmus V. Jensen, X. Jiang, David W. Peckham, Robert Lingle, Durgesh Vaidya, Man F. Yan, Patrick W. Wisk, David J. DiGiovanni
2017 conf
ECOC
Benyuan Zhu, Junwen Zhang, Jianjun Yu, Peter Ingo Borel, Tommy Geisler, Rasmus V. Jensen, David W. Peckham, Robert Lingle, Durgesh Vaidya, Man F. Yan, Patrick W. Wisk, David J. DiGiovanni
2016 conf
OFC
Benyuan Zhu, J. Zhang, J. Yu, David W. Peckham, Robert Lingle, Man F. Yan, Patrick W. Wisk, David J. DiGiovanni
2016 J jnl
JOCN
Benyuan Zhu, David W. Peckham, Alan McCurdy, Robert Lingle, Bera Palsdottir, Man F. Yan, Patrick W. Wisk, David J. DiGiovanni
2016 conf
OFC
David W. Peckham, Alan Klein, Peter Ingo Borel, Rasmus V. Jensen, Ole Levring, Kenneth Carlson, Man F. Yan, Patrick Wisk, Dennis Trevor, Robert Lingle, Alan McCurdy, Benyuan Zhu, Yi Zou, Rick Norris, Bera Palsdottir, Durgesh Vaidya
2015 conf
OFC
Benyuan Zhu, Chongjin Xie, Lynn E. Nelson, X. Jiang, David W. Peckham, Robert Lingle, Man F. Yan, Patrick W. Wisk, David J. DiGiovanni
2014 conf
OFC
Benyuan Zhu, Peter Ingo Borel, Kenneth Carlson, X. Jiang, David W. Peckham, Robert Lingle, M. Law, J. Rooney, Man F. Yan
2014 conf
ECOC
Lars Gruner-Nielsen, Tommy Geisler, John M. Fini, Man F. Yan, Patrick W. Wisk, Brian J. Mangan, Eric M. Monberg
redb/extractors/js_extractors/js_patterns.py
← Index redb/extractors/js_extractors/js_patterns.py python
"""Canonical, compiled JavaScript regex patterns shared across JS extractors.

All suspicious-API patterns and the few feature-only patterns live here so each
expression is compiled exactly once per Python process and so any pattern that
was previously duplicated across `js_features.py` and `js_suspicious_apis.py`
now resolves to a single shared compiled object.

JavaScript is case-sensitive at runtime, but every suspicious-API pattern matches
either a literal-case identifier (`\\beval\\s*\\(`, `String\\.fromCharCode`, etc.)
or a string-quoted token (`"powershell"`). Compiling them with `re.IGNORECASE`
matches the historical behaviour of `JSSuspiciousAPIsExtractor` and is safe for
the patterns that historically came from `JSFeaturesExtractor` — those literals
are spelled in real-world JS exactly as written.

`scan_source()` is the entry point used by extractors: it walks the source once
per pattern using the pre-compiled regexes and returns a flat
`{name: {"count": N, "lines": [unique_line_numbers_sorted]}}` dict. Both
`JSFeaturesExtractor` and `JSSuspiciousAPIsExtractor` consume the same dict so
the per-pattern × per-line loops they used to run independently collapse to a
single shared scan.
"""

import bisect
import re
from typing import Dict, Iterable, List, Mapping

_FLAGS = re.IGNORECASE

# Canonical compiled patterns, keyed by their human-readable name. The name is
# also the value emitted into `redb_js_suspicious_apis.api_name`.
PATTERNS = {
    # ---- code execution ----
    "eval": re.compile(r"\beval\s*\(", _FLAGS),
    "Function constructor": re.compile(r"\bnew\s+Function\s*\(", _FLAGS),
    "execScript": re.compile(r"\bexecScript\s*\(", _FLAGS),
    "document.write": re.compile(r"\bdocument\.write(?:ln)?\s*\(", _FLAGS),
    "innerHTML assignment": re.compile(r"\.innerHTML\s*=", _FLAGS),
    "outerHTML assignment": re.compile(r"\.outerHTML\s*=", _FLAGS),
    "insertAdjacentHTML": re.compile(r"\.insertAdjacentHTML\s*\(", _FLAGS),
    # ---- network ----
    "XMLHttpRequest": re.compile(r"\bnew\s+XMLHttpRequest\b", _FLAGS),
    "fetch": re.compile(r"\bfetch\s*\(", _FLAGS),
    "WebSocket": re.compile(r"\bnew\s+WebSocket\s*\(", _FLAGS),
    "navigator.sendBeacon": re.compile(r"\bnavigator\.sendBeacon\s*\(", _FLAGS),
    "ActiveXObject XMLHTTP": re.compile(
        r"ActiveXObject\s*\(\s*[\"\'](?:MSXML2\.XMLHTTP|Microsoft\.XMLHTTP)", _FLAGS
    ),
    "require network module": re.compile(
        r"require\s*\(\s*[\"\'](?:http|https|net|dgram)[\"\']", _FLAGS
    ),
    "axios": re.compile(r"\baxios\b", _FLAGS),
    # ---- filesystem ----
    "require fs": re.compile(r"require\s*\(\s*[\"\']fs[\"\']", _FLAGS),
    "require path": re.compile(r"require\s*\(\s*[\"\']path[\"\']", _FLAGS),
    "FileSystemObject": re.compile(r"Scripting\.FileSystemObject", _FLAGS),
    "ADODB.Stream": re.compile(r"ADODB\.Stream", _FLAGS),
    "Shell.Application": re.compile(r"Shell\.Application", _FLAGS),
    "WScript.CreateObject": re.compile(r"WScript\.CreateObject", _FLAGS),
    # ---- process ----
    "require child_process": re.compile(r"require\s*\(\s*[\"\']child_process[\"\']", _FLAGS),
    "child_process exec": re.compile(r"child_process\.(?:exec|spawn|execFile|fork)\s*\(", _FLAGS),
    "WScript.Shell": re.compile(r"WScript\.Shell", _FLAGS),
    "WScript.Shell.Run": re.compile(r"\.Run\s*\(", _FLAGS),
    "WScript.Shell.Exec": re.compile(r"\.Exec\s*\(", _FLAGS),
    "ShellExecute": re.compile(r"\bShellExecute\b", _FLAGS),
    "PowerShell reference": re.compile(r"[\"\']powershell[\"\']", _FLAGS),
    "cmd.exe reference": re.compile(r"[\"\']cmd\.exe[\"\']", _FLAGS),
    "require os": re.compile(r"require\s*\(\s*[\"\']os[\"\']", _FLAGS),
    # ---- registry ----
    "RegRead": re.compile(r"\.RegRead\s*\(", _FLAGS),
    "RegWrite": re.compile(r"\.RegWrite\s*\(", _FLAGS),
    "RegDelete": re.compile(r"\.RegDelete\s*\(", _FLAGS),
    "StdRegProv": re.compile(r"StdRegProv", _FLAGS),
    # ---- crypto / encoding ----
    "atob": re.compile(r"\batob\s*\(", _FLAGS),
    "btoa": re.compile(r"\bbtoa\s*\(", _FLAGS),
    "String.fromCharCode": re.compile(r"String\.fromCharCode\s*\(", _FLAGS),
    "unescape": re.compile(r"\bunescape\s*\(", _FLAGS),
    "decodeURIComponent": re.compile(r"\bdecodeURIComponent\s*\(", _FLAGS),
    "Buffer.from": re.compile(r"Buffer\.from\s*\(", _FLAGS),
    "crypto module": re.compile(r"crypto\.create(?:Cipher|Decipher|Hash|Hmac)", _FLAGS),
    # ---- DOM manipulation ----
    "document.forms": re.compile(r"document\.forms", _FLAGS),
    "document.cookie": re.compile(r"document\.cookie", _FLAGS),
    "querySelector sensitive input": re.compile(
        r"document\.querySelector\s*\([^)]*(?:password|credit|card|cvv|ssn)", _FLAGS
    ),
    "submit event listener": re.compile(r"addEventListener\s*\(\s*[\"\']submit", _FLAGS),
    "createElement script/iframe": re.compile(
        r"\.createElement\s*\(\s*[\"\'](?:script|iframe)", _FLAGS
    ),
    "dynamic script src": re.compile(r"\.src\s*=\s*[\"\'](?:https?://|//)", _FLAGS),
}

# Pattern name -> category (one of code_execution / network / filesystem /
# process / registry / crypto_encoding / dom_manipulation).
CATEGORIES = {
    "eval": "code_execution",
    "Function constructor": "code_execution",
    "execScript": "code_execution",
    "document.write": "code_execution",
    "innerHTML assignment": "code_execution",
    "outerHTML assignment": "code_execution",
    "insertAdjacentHTML": "code_execution",
    "XMLHttpRequest": "network",
    "fetch": "network",
    "WebSocket": "network",
    "navigator.sendBeacon": "network",
    "ActiveXObject XMLHTTP": "network",
    "require network module": "network",
    "axios": "network",
    "require fs": "filesystem",
    "require path": "filesystem",
    "FileSystemObject": "filesystem",
    "ADODB.Stream": "filesystem",
    "Shell.Application": "filesystem",
    "WScript.CreateObject": "filesystem",
    "require child_process": "process",
    "child_process exec": "process",
    "WScript.Shell": "process",
    "WScript.Shell.Run": "process",
    "WScript.Shell.Exec": "process",
    "ShellExecute": "process",
    "PowerShell reference": "process",
    "cmd.exe reference": "process",
    "require os": "process",
    "RegRead": "registry",
    "RegWrite": "registry",
    "RegDelete": "registry",
    "StdRegProv": "registry",
    "atob": "crypto_encoding",
    "btoa": "crypto_encoding",
    "String.fromCharCode": "crypto_encoding",
    "unescape": "crypto_encoding",
    "decodeURIComponent": "crypto_encoding",
    "Buffer.from": "crypto_encoding",
    "crypto module": "crypto_encoding",
    "document.forms": "dom_manipulation",
    "document.cookie": "dom_manipulation",
    "querySelector sensitive input": "dom_manipulation",
    "submit event listener": "dom_manipulation",
    "createElement script/iframe": "dom_manipulation",
    "dynamic script src": "dom_manipulation",
}

# Patterns consumed only by JSFeaturesExtractor (no category, never surfaced as
# a suspicious-API row). Kept here so every JS regex is compiled in one place.
FEATURE_PATTERNS = {
    "hex_escape": re.compile(r"\\x[0-9a-fA-F]{2}"),
    "unicode_escape": re.compile(r"\\u[0-9a-fA-F]{4}"),
    "base64_string": re.compile(r"[A-Za-z0-9+/]{40,}={0,2}"),
    # decodeURI matches BOTH decodeURI and decodeURIComponent. The latter is also
    # a suspicious-API pattern in PATTERNS; this broader form is what the
    # `decodeuri_count` feature column has historically counted.
    "decodeURI": re.compile(r"\b(?:decodeURI|decodeURIComponent)\s*\(", _FLAGS),
    "settimeout_setinterval": re.compile(r"\b(?:setTimeout|setInterval)\s*\(", _FLAGS),
    "function_decl": re.compile(r"\bfunction\s+\w+\s*\(|\bfunction\s*\("),
    "var_decl": re.compile(r"\b(?:var|let|const)\s+"),
    "string_concat": re.compile(r"[\"\'][\s]*\+[\s]*[\"\']"),
    "comment": re.compile(r"//.*?$|/\*[\s\S]*?\*/", re.MULTILINE),
    "long_string": re.compile(r"[\"\']([^\"\']{256,})[\"\']"),
    "array_function_call": re.compile(r"\[(?:0x[0-9a-f]+|[\d]+)\]\s*\(", _FLAGS),
}

# Patterns consumed only by JSStringsExtractor for encoded-string discovery.
# Scoped to *hidden* strings only — patterns whose decoded form is not visible
# to a substring search over the raw text. Plain long literals are not
# extracted here because they're already preserved in code_text_content and
# scraped by the IOC pipeline over text_raw / text_normalized.
#
# Distinct from FEATURE_PATTERNS even where the names rhyme:
#   FEATURE_PATTERNS["hex_escape"] / ["unicode_escape"]   -> single escape
#   STRING_PATTERNS["hex_escape_seq"] / ["unicode_escape_seq"] -> 4+ / 3+ in a row
#   FEATURE_PATTERNS["base64_string"]                     -> bare base64 token
#   STRING_PATTERNS["base64_quoted"]                      -> base64 inside JS quotes
# These do not share match objects with the suspicious-API or feature scans, so
# they are not folded into JSContext.scan; the strings extractor walks them
# itself (one finditer per pattern, with shared line-offset bisect in #4b).
STRING_PATTERNS = {
    "hex_escape_seq": re.compile(r"(?:\\x[0-9a-fA-F]{2}){4,}"),
    "unicode_escape_seq": re.compile(r"(?:\\u[0-9a-fA-F]{4}){3,}"),
    "charcode_call": re.compile(r"String\.fromCharCode\s*\(\s*([\d,\s]+)\s*\)"),
    "base64_quoted": re.compile(r"[\"\']([A-Za-z0-9+/]{40,}={0,2})[\"\']"),
    "concat_chain": re.compile(r"(?:[\"\'][^\"\']+[\"\']\s*\+\s*){3,}[\"\'][^\"\']+[\"\']"),
}


def line_offsets(source: str) -> List[int]:
    """Sorted list of byte offsets for every newline in `source`, plus a final
    sentinel of len(source). Used to translate match offsets into 1-indexed
    line numbers via bisect.
    """
    offsets = [-1]  # so that bisect_right of offset 0 returns line 1
    push = offsets.append
    idx = source.find("\n")
    while idx != -1:
        push(idx)
        idx = source.find("\n", idx + 1)
    return offsets


def _scan_one(
    pattern: "re.Pattern[str]", source: str, offsets: List[int]
) -> Dict[str, object]:
    """Run a single compiled pattern over `source` and return count + unique lines."""
    count = 0
    seen_lines: "set[int]" = set()
    for m in pattern.finditer(source):
        count += 1
        seen_lines.add(bisect.bisect_right(offsets, m.start()))
    if not count:
        return None  # type: ignore[return-value]
    return {"count": count, "lines": sorted(seen_lines)}


def scan_source(
    source: str,
    patterns: Iterable[Mapping[str, "re.Pattern[str]"]] = (PATTERNS, FEATURE_PATTERNS),
) -> Dict[str, Dict[str, object]]:
    """Scan `source` against every compiled pattern in `patterns`.

    Returns a dict keyed by pattern name. Each entry has:
        "count": total number of matches in the source
        "lines": sorted list of unique 1-indexed line numbers where the pattern
                 matched (deduplicated — multiple matches on the same line
                 collapse to one entry, preserving the historical
                 line-set semantics of JSSuspiciousAPIsExtractor)
    Patterns with zero matches are absent from the dict; callers should default
    to {"count": 0, "lines": []}.
    """
    if not source:
        return {}
    offsets = line_offsets(source)
    results: Dict[str, Dict[str, object]] = {}
    for table in patterns:
        for name, pat in table.items():
            entry = _scan_one(pat, source, offsets)
            if entry is not None:
                results[name] = entry
    return results