Ines Reinecke

21 papers Journal 5Unranked 16
YearRankTypeTitle / Venue / Authors
2025 J jnl
J. Medical Syst.
Franziska Bathelt, Stephan Lorenz, Jens Weidner, Martin Sedlmayr, Ines Reinecke
2025 conf
GMDS
Anne Pelz, Philipp Heinrich, Gabriele Mueller, Anne Seim, Peter Penndorf, Martin Bialke, Martin Sedlmayr, Ines Reinecke, Markus Wolfien, Katja Hoffmann
2024 conf
MIE
Katja Hoffmann, Richard Gebler, Sophia Grummt, Yuan Peng, Ines Reinecke, Markus Wolfien, Martin Sedlmayr
2024 conf
GMDS
Hung Manh Nguyen, Luise Donat, Jens Helbig, Martin Sedlmayr, Miriam Goldammer, Ines Reinecke
2024 J jnl
BMC Medical Informatics Decis. Mak.
Elisa Henke, Michele Zoch, Yuan Peng, Ines Reinecke, Martin Sedlmayr, Franziska Bathelt
2024 J jnl
Comput. Biol. Medicine
Christian Gulden, Philipp Macho, Ines Reinecke, Cosima Strantz, Hans-Ulrich Prokosch, Romina Blasini
2023 J jnl
Int. J. Medical Informatics
Yuan Peng, Elisa Henke, Ines Reinecke, Michéle Zoch, Martin Sedlmayr, Franziska Bathelt
2023 conf
MIE
Stephan Lorenz, Richard Gebler, Franziska Bathelt, Martin Sedlmayr, Ines Reinecke
2023 conf
MIE
Ines Reinecke, Elisa Henke, Yuan Peng, Martin Sedlmayr, Franziska Bathelt
2023 conf
MIE
Elisa Henke, Michéle Zoch, Ines Reinecke, Melissa Spoden, Thomas Ruhnke, Christian Günster, Martin Sedlmayr, Franziska Bathelt
2023 J jnl
J. Am. Medical Informatics Assoc.
Anna Ostropolets, Yasser Albogami, Mitchell Conover, Juan M. Banda, William A. Baumgartner Jr., Clair Blacketer, Priyamvada Desai, Scott L. DuVall, Stephen P. Fortin, James P. Gilbert, Asieh Golozar, Joshua Ide, Andrew S. Kanter, David M. Kern, Chungsoo Kim, Lana Y. H. Lai, Chenyu Li, Feifan Liu, Kristine E. Lynch, Evan Minty, Maria Inês Neves, Ding Quan Ng, Tontel Obene, Victor Pera, Nicole Pratt, Gowtham Rao, Nadav Rappoport, Ines Reinecke, Paola Saroufim, Azza Shoaibi, Katherine Simon, Marc A. Suchard, Joel N. Swerdel, Erica A. Voss, James Weaver, Linying Zhang, George Hripcsak, Patrick B. Ryan
2022 conf
ICIMTH
Ines Reinecke, Mirko Gruhl, Martin Pinnau, Fatma Betül Altun, Michael Folz, Michéle Zoch, Franziska Bathelt, Martin Sedlmayr
2022 conf
MIE
Ines Reinecke, Michael Kallfelz, Martin Sedlmayr, Joscha Siebel, Franziska Bathelt
2022 conf
EFMI-STC
Ines Reinecke, Franziska Bathelt, Martin Sedlmayr, Andreas Kühn
2022 conf
MIE
Elisa Henke, Ines Reinecke, Michele Zoch, Martin Sedlmayr, Franziska Bathelt
2022 conf
EFMI-STC
Franziska Bathelt, Ines Reinecke, Brita Sedlmayr, Sepp Höhne, Christian Gierschner, Martin Sedlmayr
2021 conf
MedInfo
Albert Vass, Ines Reinecke, Martin Boeker, Hans-Ulrich Prokosch, Christian Gulden
2021 conf
GMDS
Ines Reinecke, Michéle Zoch, Christian Reich, Martin Sedlmayr, Franziska Bathelt
2021 conf
EFMI-STC
Ines Reinecke, Michéle Zoch, Markus Wilhelm, Martin Sedlmayr, Franziska Bathelt
2020 conf
MIE
Ines Reinecke, Christian Gulden, Michéle Kümmel, Azadeh Nassirian, Romina Blasini, Martin Sedlmayr
2020 conf
MIE
Mirko Gruhl, Ines Reinecke, Martin Sedlmayr
redb/extractors/js_extractors/js_patterns.py
← Index redb/extractors/js_extractors/js_patterns.py python
"""Canonical, compiled JavaScript regex patterns shared across JS extractors.

All suspicious-API patterns and the few feature-only patterns live here so each
expression is compiled exactly once per Python process and so any pattern that
was previously duplicated across `js_features.py` and `js_suspicious_apis.py`
now resolves to a single shared compiled object.

JavaScript is case-sensitive at runtime, but every suspicious-API pattern matches
either a literal-case identifier (`\\beval\\s*\\(`, `String\\.fromCharCode`, etc.)
or a string-quoted token (`"powershell"`). Compiling them with `re.IGNORECASE`
matches the historical behaviour of `JSSuspiciousAPIsExtractor` and is safe for
the patterns that historically came from `JSFeaturesExtractor` — those literals
are spelled in real-world JS exactly as written.

`scan_source()` is the entry point used by extractors: it walks the source once
per pattern using the pre-compiled regexes and returns a flat
`{name: {"count": N, "lines": [unique_line_numbers_sorted]}}` dict. Both
`JSFeaturesExtractor` and `JSSuspiciousAPIsExtractor` consume the same dict so
the per-pattern × per-line loops they used to run independently collapse to a
single shared scan.
"""

import bisect
import re
from typing import Dict, Iterable, List, Mapping

_FLAGS = re.IGNORECASE

# Canonical compiled patterns, keyed by their human-readable name. The name is
# also the value emitted into `redb_js_suspicious_apis.api_name`.
PATTERNS = {
    # ---- code execution ----
    "eval": re.compile(r"\beval\s*\(", _FLAGS),
    "Function constructor": re.compile(r"\bnew\s+Function\s*\(", _FLAGS),
    "execScript": re.compile(r"\bexecScript\s*\(", _FLAGS),
    "document.write": re.compile(r"\bdocument\.write(?:ln)?\s*\(", _FLAGS),
    "innerHTML assignment": re.compile(r"\.innerHTML\s*=", _FLAGS),
    "outerHTML assignment": re.compile(r"\.outerHTML\s*=", _FLAGS),
    "insertAdjacentHTML": re.compile(r"\.insertAdjacentHTML\s*\(", _FLAGS),
    # ---- network ----
    "XMLHttpRequest": re.compile(r"\bnew\s+XMLHttpRequest\b", _FLAGS),
    "fetch": re.compile(r"\bfetch\s*\(", _FLAGS),
    "WebSocket": re.compile(r"\bnew\s+WebSocket\s*\(", _FLAGS),
    "navigator.sendBeacon": re.compile(r"\bnavigator\.sendBeacon\s*\(", _FLAGS),
    "ActiveXObject XMLHTTP": re.compile(
        r"ActiveXObject\s*\(\s*[\"\'](?:MSXML2\.XMLHTTP|Microsoft\.XMLHTTP)", _FLAGS
    ),
    "require network module": re.compile(
        r"require\s*\(\s*[\"\'](?:http|https|net|dgram)[\"\']", _FLAGS
    ),
    "axios": re.compile(r"\baxios\b", _FLAGS),
    # ---- filesystem ----
    "require fs": re.compile(r"require\s*\(\s*[\"\']fs[\"\']", _FLAGS),
    "require path": re.compile(r"require\s*\(\s*[\"\']path[\"\']", _FLAGS),
    "FileSystemObject": re.compile(r"Scripting\.FileSystemObject", _FLAGS),
    "ADODB.Stream": re.compile(r"ADODB\.Stream", _FLAGS),
    "Shell.Application": re.compile(r"Shell\.Application", _FLAGS),
    "WScript.CreateObject": re.compile(r"WScript\.CreateObject", _FLAGS),
    # ---- process ----
    "require child_process": re.compile(r"require\s*\(\s*[\"\']child_process[\"\']", _FLAGS),
    "child_process exec": re.compile(r"child_process\.(?:exec|spawn|execFile|fork)\s*\(", _FLAGS),
    "WScript.Shell": re.compile(r"WScript\.Shell", _FLAGS),
    "WScript.Shell.Run": re.compile(r"\.Run\s*\(", _FLAGS),
    "WScript.Shell.Exec": re.compile(r"\.Exec\s*\(", _FLAGS),
    "ShellExecute": re.compile(r"\bShellExecute\b", _FLAGS),
    "PowerShell reference": re.compile(r"[\"\']powershell[\"\']", _FLAGS),
    "cmd.exe reference": re.compile(r"[\"\']cmd\.exe[\"\']", _FLAGS),
    "require os": re.compile(r"require\s*\(\s*[\"\']os[\"\']", _FLAGS),
    # ---- registry ----
    "RegRead": re.compile(r"\.RegRead\s*\(", _FLAGS),
    "RegWrite": re.compile(r"\.RegWrite\s*\(", _FLAGS),
    "RegDelete": re.compile(r"\.RegDelete\s*\(", _FLAGS),
    "StdRegProv": re.compile(r"StdRegProv", _FLAGS),
    # ---- crypto / encoding ----
    "atob": re.compile(r"\batob\s*\(", _FLAGS),
    "btoa": re.compile(r"\bbtoa\s*\(", _FLAGS),
    "String.fromCharCode": re.compile(r"String\.fromCharCode\s*\(", _FLAGS),
    "unescape": re.compile(r"\bunescape\s*\(", _FLAGS),
    "decodeURIComponent": re.compile(r"\bdecodeURIComponent\s*\(", _FLAGS),
    "Buffer.from": re.compile(r"Buffer\.from\s*\(", _FLAGS),
    "crypto module": re.compile(r"crypto\.create(?:Cipher|Decipher|Hash|Hmac)", _FLAGS),
    # ---- DOM manipulation ----
    "document.forms": re.compile(r"document\.forms", _FLAGS),
    "document.cookie": re.compile(r"document\.cookie", _FLAGS),
    "querySelector sensitive input": re.compile(
        r"document\.querySelector\s*\([^)]*(?:password|credit|card|cvv|ssn)", _FLAGS
    ),
    "submit event listener": re.compile(r"addEventListener\s*\(\s*[\"\']submit", _FLAGS),
    "createElement script/iframe": re.compile(
        r"\.createElement\s*\(\s*[\"\'](?:script|iframe)", _FLAGS
    ),
    "dynamic script src": re.compile(r"\.src\s*=\s*[\"\'](?:https?://|//)", _FLAGS),
}

# Pattern name -> category (one of code_execution / network / filesystem /
# process / registry / crypto_encoding / dom_manipulation).
CATEGORIES = {
    "eval": "code_execution",
    "Function constructor": "code_execution",
    "execScript": "code_execution",
    "document.write": "code_execution",
    "innerHTML assignment": "code_execution",
    "outerHTML assignment": "code_execution",
    "insertAdjacentHTML": "code_execution",
    "XMLHttpRequest": "network",
    "fetch": "network",
    "WebSocket": "network",
    "navigator.sendBeacon": "network",
    "ActiveXObject XMLHTTP": "network",
    "require network module": "network",
    "axios": "network",
    "require fs": "filesystem",
    "require path": "filesystem",
    "FileSystemObject": "filesystem",
    "ADODB.Stream": "filesystem",
    "Shell.Application": "filesystem",
    "WScript.CreateObject": "filesystem",
    "require child_process": "process",
    "child_process exec": "process",
    "WScript.Shell": "process",
    "WScript.Shell.Run": "process",
    "WScript.Shell.Exec": "process",
    "ShellExecute": "process",
    "PowerShell reference": "process",
    "cmd.exe reference": "process",
    "require os": "process",
    "RegRead": "registry",
    "RegWrite": "registry",
    "RegDelete": "registry",
    "StdRegProv": "registry",
    "atob": "crypto_encoding",
    "btoa": "crypto_encoding",
    "String.fromCharCode": "crypto_encoding",
    "unescape": "crypto_encoding",
    "decodeURIComponent": "crypto_encoding",
    "Buffer.from": "crypto_encoding",
    "crypto module": "crypto_encoding",
    "document.forms": "dom_manipulation",
    "document.cookie": "dom_manipulation",
    "querySelector sensitive input": "dom_manipulation",
    "submit event listener": "dom_manipulation",
    "createElement script/iframe": "dom_manipulation",
    "dynamic script src": "dom_manipulation",
}

# Patterns consumed only by JSFeaturesExtractor (no category, never surfaced as
# a suspicious-API row). Kept here so every JS regex is compiled in one place.
FEATURE_PATTERNS = {
    "hex_escape": re.compile(r"\\x[0-9a-fA-F]{2}"),
    "unicode_escape": re.compile(r"\\u[0-9a-fA-F]{4}"),
    "base64_string": re.compile(r"[A-Za-z0-9+/]{40,}={0,2}"),
    # decodeURI matches BOTH decodeURI and decodeURIComponent. The latter is also
    # a suspicious-API pattern in PATTERNS; this broader form is what the
    # `decodeuri_count` feature column has historically counted.
    "decodeURI": re.compile(r"\b(?:decodeURI|decodeURIComponent)\s*\(", _FLAGS),
    "settimeout_setinterval": re.compile(r"\b(?:setTimeout|setInterval)\s*\(", _FLAGS),
    "function_decl": re.compile(r"\bfunction\s+\w+\s*\(|\bfunction\s*\("),
    "var_decl": re.compile(r"\b(?:var|let|const)\s+"),
    "string_concat": re.compile(r"[\"\'][\s]*\+[\s]*[\"\']"),
    "comment": re.compile(r"//.*?$|/\*[\s\S]*?\*/", re.MULTILINE),
    "long_string": re.compile(r"[\"\']([^\"\']{256,})[\"\']"),
    "array_function_call": re.compile(r"\[(?:0x[0-9a-f]+|[\d]+)\]\s*\(", _FLAGS),
}

# Patterns consumed only by JSStringsExtractor for encoded-string discovery.
# Scoped to *hidden* strings only — patterns whose decoded form is not visible
# to a substring search over the raw text. Plain long literals are not
# extracted here because they're already preserved in code_text_content and
# scraped by the IOC pipeline over text_raw / text_normalized.
#
# Distinct from FEATURE_PATTERNS even where the names rhyme:
#   FEATURE_PATTERNS["hex_escape"] / ["unicode_escape"]   -> single escape
#   STRING_PATTERNS["hex_escape_seq"] / ["unicode_escape_seq"] -> 4+ / 3+ in a row
#   FEATURE_PATTERNS["base64_string"]                     -> bare base64 token
#   STRING_PATTERNS["base64_quoted"]                      -> base64 inside JS quotes
# These do not share match objects with the suspicious-API or feature scans, so
# they are not folded into JSContext.scan; the strings extractor walks them
# itself (one finditer per pattern, with shared line-offset bisect in #4b).
STRING_PATTERNS = {
    "hex_escape_seq": re.compile(r"(?:\\x[0-9a-fA-F]{2}){4,}"),
    "unicode_escape_seq": re.compile(r"(?:\\u[0-9a-fA-F]{4}){3,}"),
    "charcode_call": re.compile(r"String\.fromCharCode\s*\(\s*([\d,\s]+)\s*\)"),
    "base64_quoted": re.compile(r"[\"\']([A-Za-z0-9+/]{40,}={0,2})[\"\']"),
    "concat_chain": re.compile(r"(?:[\"\'][^\"\']+[\"\']\s*\+\s*){3,}[\"\'][^\"\']+[\"\']"),
}


def line_offsets(source: str) -> List[int]:
    """Sorted list of byte offsets for every newline in `source`, plus a final
    sentinel of len(source). Used to translate match offsets into 1-indexed
    line numbers via bisect.
    """
    offsets = [-1]  # so that bisect_right of offset 0 returns line 1
    push = offsets.append
    idx = source.find("\n")
    while idx != -1:
        push(idx)
        idx = source.find("\n", idx + 1)
    return offsets


def _scan_one(
    pattern: "re.Pattern[str]", source: str, offsets: List[int]
) -> Dict[str, object]:
    """Run a single compiled pattern over `source` and return count + unique lines."""
    count = 0
    seen_lines: "set[int]" = set()
    for m in pattern.finditer(source):
        count += 1
        seen_lines.add(bisect.bisect_right(offsets, m.start()))
    if not count:
        return None  # type: ignore[return-value]
    return {"count": count, "lines": sorted(seen_lines)}


def scan_source(
    source: str,
    patterns: Iterable[Mapping[str, "re.Pattern[str]"]] = (PATTERNS, FEATURE_PATTERNS),
) -> Dict[str, Dict[str, object]]:
    """Scan `source` against every compiled pattern in `patterns`.

    Returns a dict keyed by pattern name. Each entry has:
        "count": total number of matches in the source
        "lines": sorted list of unique 1-indexed line numbers where the pattern
                 matched (deduplicated — multiple matches on the same line
                 collapse to one entry, preserving the historical
                 line-set semantics of JSSuspiciousAPIsExtractor)
    Patterns with zero matches are absent from the dict; callers should default
    to {"count": 0, "lines": []}.
    """
    if not source:
        return {}
    offsets = line_offsets(source)
    results: Dict[str, Dict[str, object]] = {}
    for table in patterns:
        for name, pat in table.items():
            entry = _scan_one(pat, source, offsets)
            if entry is not None:
                results[name] = entry
    return results