UNPKG

ioc-extractor

Version:

IoC (Indicator of Compromise) extractor

2,526 lines (2,525 loc) 43.3 kB
Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" }); //#region \0rolldown/runtime.js var __create = Object.create; var __defProp = Object.defineProperty; var __getOwnPropDesc = Object.getOwnPropertyDescriptor; var __getOwnPropNames = Object.getOwnPropertyNames; var __getProtoOf = Object.getPrototypeOf; var __hasOwnProp = Object.prototype.hasOwnProperty; var __copyProps = (to, from, except, desc) => { if (from && typeof from === "object" || typeof from === "function") for (var keys = __getOwnPropNames(from), i = 0, n = keys.length, key; i < n; i++) { key = keys[i]; if (!__hasOwnProp.call(to, key) && key !== except) __defProp(to, key, { get: ((k) => from[k]).bind(null, key), enumerable: !(desc = __getOwnPropDesc(from, key)) || desc.enumerable }); } return to; }; var __toESM = (mod, isNodeMode, target) => (target = mod != null ? __create(__getProtoOf(mod)) : {}, __copyProps(isNodeMode || !mod || !mod.__esModule ? __defProp(target, "default", { value: mod, enumerable: true }) : target, mod)); //#endregion let lru_cache = require("lru-cache"); let punycode_js = require("punycode.js"); punycode_js = __toESM(punycode_js); //#region src/aux/tlds.ts const tlds = [ "xn--vermgensberatung-pwb", "xn--vermgensberater-ctb", "xn--clchc0ea0b2g2a9gcd", "xn--w4r85el8fhu5dnra", "northwesternmutual", "travelersinsurance", "xn--3oq18vl8pn36a", "xn--5su34j936bgsg", "xn--bck1b9a5dre4c", "xn--mgbah1a3hjkrd", "xn--mgbai9azgqp6j", "xn--mgberp4a5d4ar", "xn--xkc2dl3a5ee0h", "xn--fzys8d69uvgm", "xn--mgba7c0bbn0a", "xn--mgbcpq6gpa1a", "xn--xkc2al3hye2a", "americanexpress", "kerryproperties", "sandvikcoromant", "xn--i1b6b1a6a2e", "xn--kcrx77d1x4a", "xn--lgbbat1ad8j", "xn--mgba3a4f16a", "xn--mgbaakc7dvf", "xn--mgbc0a9azcg", "xn--nqv7fs00ema", "afamilycompany", "americanfamily", "bananarepublic", "cancerresearch", "cookingchannel", "kerrylogistics", "weatherchannel", "xn--54b7fta0cc", "xn--6qq986b3xl", "xn--80aqecdr1a", "xn--b4w605ferd", "xn--fiq228c5hs", "xn--h2breg3eve", "xn--jlq480n2rg", "xn--jlq61u9w7b", "xn--mgba3a3ejt", "xn--mgbaam7a8h", "xn--mgbayh7gpa", "xn--mgbb9fbpob", "xn--mgbbh1a71e", "xn--mgbca7dzdo", "xn--mgbi4ecexp", "xn--mgbx4cd0ab", "xn--rvc1e0am3e", "international", "lifeinsurance", "orientexpress", "spreadbetting", "travelchannel", "wolterskluwer", "xn--cckwcxetd", "xn--eckvdtc9d", "xn--fpcrj9c3d", "xn--fzc2c9e2c", "xn--h2brj9c8c", "xn--tiq49xqyj", "xn--yfro4i67o", "xn--ygbi2ammx", "construction", "lplfinancial", "pamperedchef", "scholarships", "versicherung", "xn--3e0b707e", "xn--45br5cyl", "xn--4dbrk0ce", "xn--80adxhks", "xn--80asehdb", "xn--8y0a063a", "xn--gckr3f0f", "xn--mgb9awbf", "xn--mgbab2bd", "xn--mgbgu82a", "xn--mgbpl2fh", "xn--mgbt3dhd", "xn--mk1bu44c", "xn--ngbc5azd", "xn--ngbe9e0a", "xn--ogbpf8fl", "xn--qcka1pmc", "accountants", "barclaycard", "blackfriday", "blockbuster", "bridgestone", "calvinklein", "contractors", "creditunion", "engineering", "enterprises", "foodnetwork", "investments", "kerryhotels", "lamborghini", "motorcycles", "olayangroup", "photography", "playstation", "productions", "progressive", "redumbrella", "rightathome", "williamhill", "xn--11b4c3d", "xn--1ck2e1b", "xn--1qqw23a", "xn--2scrj9c", "xn--3bst00m", "xn--3ds443g", "xn--3hcrj9c", "xn--42c2d9a", "xn--45brj9c", "xn--55qw42g", "xn--6frz82g", "xn--80ao21a", "xn--9krt00a", "xn--cck2b3b", "xn--czr694b", "xn--d1acj3b", "xn--efvy88h", "xn--estv75g", "xn--fct429k", "xn--fjq720a", "xn--flw351e", "xn--g2xx48c", "xn--gecrj9c", "xn--gk3at1e", "xn--h2brj9c", "xn--hxt814e", "xn--imr513n", "xn--j6w193g", "xn--jvr189m", "xn--kprw13d", "xn--kpry57d", "xn--kpu716f", "xn--mgbbh1a", "xn--mgbtx2b", "xn--mix891f", "xn--nyqy26a", "xn--otu796d", "xn--pbt977c", "xn--pgbs0dh", "xn--q9jyb4c", "xn--rhqv96g", "xn--rovu88b", "xn--s9brj9c", "xn--ses554g", "xn--t60b56a", "xn--vuq861b", "xn--w4rs40l", "xn--xhq521b", "xn--zfr164b", "accountant", "apartments", "associates", "basketball", "bnpparibas", "boehringer", "capitalone", "consulting", "creditcard", "cuisinella", "eurovision", "extraspace", "foundation", "healthcare", "immobilien", "industries", "management", "mitsubishi", "nationwide", "newholland", "nextdirect", "onyourside", "properties", "protection", "prudential", "realestate", "republican", "restaurant", "schaeffler", "swiftcover", "tatamotors", "technology", "telefonica", "university", "vistaprint", "vlaanderen", "volkswagen", "xn--30rr7y", "xn--3pxu8k", "xn--45q11c", "xn--4gbrim", "xn--55qx5d", "xn--5tzm5g", "xn--80aswg", "xn--90a3ac", "xn--9dbq2a", "xn--9et52u", "xn--c2br7g", "xn--cg4bki", "xn--czrs0t", "xn--czru2d", "xn--fiq64b", "xn--fiqs8s", "xn--fiqz9s", "xn--io0a7i", "xn--kput3i", "xn--mxtq1m", "xn--o3cw4h", "xn--pssy2u", "xn--q7ce6a", "xn--unup4y", "xn--wgbh1c", "xn--wgbl6a", "xn--y9a3aq", "accenture", "alfaromeo", "allfinanz", "amsterdam", "analytics", "aquarelle", "barcelona", "bloomberg", "christmas", "community", "directory", "education", "equipment", "fairwinds", "financial", "firestone", "fresenius", "frontdoor", "fujixerox", "furniture", "goldpoint", "goodhands", "hisamitsu", "homedepot", "homegoods", "homesense", "honeywell", "institute", "insurance", "kuokgroup", "ladbrokes", "lancaster", "landrover", "lifestyle", "marketing", "marshalls", "melbourne", "microsoft", "montblanc", "panasonic", "passagens", "pramerica", "richardli", "scjohnson", "shangrila", "solutions", "statebank", "statefarm", "stockholm", "travelers", "vacations", "xn--90ais", "xn--c1avg", "xn--d1alf", "xn--e1a4c", "xn--fhbei", "xn--j1aef", "xn--j1amh", "xn--l1acc", "xn--ngbrx", "xn--nqv7f", "xn--p1acf", "xn--qxa6a", "xn--tckwe", "xn--vhquv", "yodobashi", "abudhabi", "airforce", "allstate", "attorney", "barclays", "barefoot", "bargains", "baseball", "boutique", "bradesco", "broadway", "brussels", "budapest", "builders", "business", "capetown", "catering", "catholic", "chrysler", "cipriani", "cityeats", "cleaning", "clinique", "clothing", "commbank", "computer", "delivery", "deloitte", "democrat", "diamonds", "discount", "discover", "download", "engineer", "ericsson", "esurance", "etisalat", "everbank", "exchange", "feedback", "fidelity", "firmdale", "flsmidth", "football", "frontier", "goodyear", "grainger", "graphics", "guardian", "hdfcbank", "helsinki", "holdings", "hospital", "infiniti", "ipiranga", "istanbul", "jpmorgan", "lighting", "lundbeck", "marriott", "maserati", "mckinsey", "memorial", "merckmsd", "mortgage", "movistar", "mutuelle", "observer", "partners", "pharmacy", "pictures", "plumbing", "property", "redstone", "reliance", "saarland", "samsclub", "security", "services", "shopping", "showtime", "softbank", "software", "stcgroup", "supplies", "symantec", "telecity", "training", "uconnect", "vanguard", "ventures", "verisign", "woodside", "xn--90ae", "xn--node", "xn--p1ai", "xn--qxam", "yokohama", "abogado", "academy", "agakhan", "alibaba", "android", "athleta", "auction", "audible", "auspost", "avianca", "banamex", "bauhaus", "bentley", "bestbuy", "booking", "brother", "bugatti", "capital", "caravan", "careers", "cartier", "channel", "charity", "chintai", "citadel", "clubmed", "college", "cologne", "comcast", "company", "compare", "contact", "cooking", "corsica", "country", "coupons", "courses", "cricket", "cruises", "dentist", "digital", "domains", "exposed", "express", "farmers", "fashion", "ferrari", "ferrero", "finance", "fishing", "fitness", "flights", "florist", "flowers", "forsale", "frogans", "fujitsu", "gallery", "genting", "godaddy", "grocery", "guitars", "hamburg", "hangout", "hitachi", "holiday", "hosting", "hoteles", "hotmail", "hyundai", "iselect", "ismaili", "jewelry", "juniper", "kitchen", "komatsu", "lacaixa", "lancome", "lanxess", "lasalle", "latrobe", "leclerc", "liaison", "limited", "lincoln", "markets", "metlife", "monster", "netbank", "netflix", "network", "neustar", "okinawa", "oldnavy", "organic", "origins", "panerai", "philips", "pioneer", "politie", "realtor", "recipes", "rentals", "reviews", "rexroth", "samsung", "sandvik", "schmidt", "schwarz", "science", "shiksha", "shriram", "singles", "spiegel", "staples", "starhub", "statoil", "storage", "support", "surgery", "systems", "temasek", "theater", "theatre", "tickets", "tiffany", "toshiba", "trading", "walmart", "wanggou", "watches", "weather", "website", "wedding", "whoswho", "windows", "winners", "xfinity", "yamaxun", "youtube", "zuerich", "abarth", "abbott", "abbvie", "active", "africa", "agency", "airbus", "airtel", "alipay", "alsace", "alstom", "amazon", "anquan", "aramco", "author", "bayern", "beauty", "berlin", "bharti", "blanco", "bostik", "boston", "broker", "camera", "career", "caseih", "casino", "center", "chanel", "chrome", "church", "circle", "claims", "clinic", "coffee", "comsec", "condos", "coupon", "credit", "cruise", "dating", "datsun", "dealer", "degree", "dental", "design", "direct", "doctor", "doosan", "dunlop", "dupont", "durban", "emerck", "energy", "estate", "events", "expert", "family", "flickr", "futbol", "gallup", "garden", "george", "giving", "global", "google", "gratis", "health", "hermes", "hiphop", "hockey", "hotels", "hughes", "imamat", "insure", "intuit", "jaguar", "joburg", "juegos", "kaufen", "kinder", "kindle", "kosher", "lancia", "latino", "lawyer", "lefrak", "living", "locker", "london", "luxury", "madrid", "maison", "makeup", "market", "mattel", "mobile", "mobily", "monash", "mormon", "moscow", "museum", "mutual", "nagoya", "natura", "nissan", "nissay", "norton", "nowruz", "office", "olayan", "online", "oracle", "orange", "otsuka", "pfizer", "photos", "physio", "piaget", "pictet", "quebec", "racing", "realty", "reisen", "repair", "report", "review", "rocher", "rogers", "ryukyu", "safety", "sakura", "sanofi", "school", "schule", "search", "secure", "select", "shouji", "soccer", "social", "stream", "studio", "supply", "suzuki", "swatch", "sydney", "taipei", "taobao", "target", "tattoo", "tennis", "tienda", "tjmaxx", "tkmaxx", "toyota", "travel", "unicom", "viajes", "viking", "villas", "virgin", "vision", "voting", "voyage", "vuelos", "walter", "warman", "webcam", "xihuan", "xperia", "yachts", "yandex", "zappos", "actor", "adult", "aetna", "amfam", "amica", "apple", "archi", "audio", "autos", "azure", "baidu", "beats", "bible", "bingo", "black", "boats", "boots", "bosch", "build", "canon", "cards", "chase", "cheap", "chloe", "cisco", "citic", "click", "cloud", "coach", "codes", "crown", "cymru", "dabur", "dance", "deals", "delta", "dodge", "drive", "dubai", "earth", "edeka", "email", "epost", "epson", "faith", "fedex", "final", "forex", "forum", "gallo", "games", "gifts", "gives", "glade", "glass", "globo", "gmail", "green", "gripe", "group", "gucci", "guide", "homes", "honda", "horse", "house", "hyatt", "iinet", "ikano", "intel", "irish", "iveco", "jetzt", "koeln", "kyoto", "lamer", "lease", "legal", "lexus", "lilly", "linde", "lipsy", "lixil", "loans", "locus", "lotte", "lotto", "lupin", "macys", "mango", "media", "miami", "money", "mopar", "movie", "music", "nadex", "nexus", "nikon", "ninja", "nokia", "nowtv", "omega", "osaka", "paris", "parts", "party", "phone", "photo", "pizza", "place", "poker", "praxi", "press", "prime", "promo", "quest", "radio", "rehab", "reise", "ricoh", "rocks", "rodeo", "rugby", "salon", "sener", "seven", "sharp", "shell", "shoes", "skype", "sling", "smart", "smile", "solar", "space", "sport", "stada", "store", "study", "style", "sucks", "swiss", "tatar", "tires", "tirol", "tmall", "today", "tokyo", "tools", "toray", "total", "tours", "trade", "trust", "tunes", "tushu", "ubank", "vegas", "video", "vista", "vodka", "volvo", "wales", "watch", "weber", "weibo", "works", "world", "xerox", "yahoo", "zippo", "aarp", "able", "adac", "aero", "aigo", "akdn", "ally", "amex", "arab", "army", "arpa", "arte", "asda", "asia", "audi", "auto", "baby", "band", "bank", "bbva", "beer", "best", "bike", "bing", "blog", "blue", "bofa", "bond", "book", "buzz", "cafe", "call", "camp", "care", "cars", "casa", "case", "cash", "cbre", "cern", "chat", "citi", "city", "club", "cool", "coop", "cyou", "data", "date", "dclk", "deal", "dell", "desi", "diet", "dish", "docs", "doha", "duck", "duns", "dvag", "erni", "fage", "fail", "fans", "farm", "fast", "fiat", "fido", "film", "fire", "fish", "flir", "food", "ford", "free", "fund", "game", "gbiz", "gent", "ggee", "gift", "gmbh", "gold", "golf", "goog", "guge", "guru", "hair", "haus", "hdfc", "help", "here", "hgtv", "host", "hsbc", "icbc", "ieee", "imdb", "immo", "info", "itau", "java", "jeep", "jobs", "jprs", "kddi", "kids", "kiwi", "kpmg", "kred", "land", "lego", "lgbt", "lidl", "life", "like", "limo", "link", "live", "loan", "loft", "love", "ltda", "luxe", "maif", "meet", "meme", "menu", "mini", "mint", "mobi", "moda", "moto", "mtpc", "name", "navy", "news", "next", "nico", "nike", "ollo", "open", "page", "pars", "pccw", "pics", "ping", "pink", "play", "plus", "pohl", "porn", "post", "prod", "prof", "qpon", "raid", "read", "reit", "rent", "rest", "rich", "rmit", "room", "rsvp", "ruhr", "safe", "sale", "sapo", "sarl", "save", "saxo", "scor", "scot", "seat", "seek", "sexy", "shaw", "shia", "shop", "show", "silk", "sina", "site", "skin", "sncf", "sohu", "song", "sony", "spot", "star", "surf", "talk", "taxi", "team", "tech", "teva", "tiaa", "tips", "town", "toys", "tube", "vana", "visa", "viva", "vivo", "vote", "voto", "wang", "weir", "wien", "wiki", "wine", "work", "xbox", "yoga", "zara", "zero", "zone", "aaa", "abb", "abc", "aco", "ads", "aeg", "afl", "aig", "anz", "aol", "app", "art", "aws", "axa", "bar", "bbc", "bbt", "bcg", "bcn", "bet", "bid", "bio", "biz", "bms", "bmw", "bnl", "bom", "boo", "bot", "box", "buy", "bzh", "cab", "cal", "cam", "car", "cat", "cba", "cbn", "cbs", "ceb", "ceo", "cfa", "cfd", "com", "cpa", "crs", "csc", "dad", "day", "dds", "dev", "dhl", "diy", "dnp", "dog", "dot", "dtv", "dvr", "eat", "eco", "edu", "esq", "eus", "fan", "fit", "fly", "foo", "fox", "frl", "ftr", "fun", "fyi", "gal", "gap", "gay", "gdn", "gea", "gle", "gmo", "gmx", "goo", "gop", "got", "gov", "hbo", "hiv", "hkt", "hot", "how", "htc", "ibm", "ice", "icu", "ifm", "inc", "ing", "ink", "int", "ist", "itv", "iwc", "jcb", "jcp", "jio", "jlc", "jll", "jmp", "jnj", "jot", "joy", "kfh", "kia", "kim", "kpn", "krd", "lat", "law", "lds", "llc", "llp", "lol", "lpl", "ltd", "man", "map", "mba", "med", "men", "meo", "mil", "mit", "mlb", "mls", "mma", "moe", "moi", "mom", "mov", "msd", "mtn", "mtr", "nab", "nba", "nec", "net", "new", "nfl", "ngo", "nhk", "now", "nra", "nrw", "ntt", "nyc", "obi", "off", "one", "ong", "onl", "ooo", "org", "ott", "ovh", "pay", "pet", "phd", "pid", "pin", "pnc", "pro", "pru", "pub", "pwc", "qvc", "red", "ren", "ril", "rio", "rip", "run", "rwe", "sap", "sas", "sbi", "sbs", "sca", "scb", "ses", "sew", "sex", "sfr", "ski", "sky", "soy", "spa", "srl", "srt", "stc", "tab", "tax", "tci", "tdk", "tel", "thd", "tjx", "top", "trv", "tui", "tvs", "ubs", "uno", "uol", "ups", "vet", "vig", "vin", "vip", "wed", "win", "wme", "wow", "wtc", "wtf", "xin", "xxx", "xyz", "you", "yun", "zip", "ac", "ad", "ae", "af", "ag", "ai", "al", "am", "ao", "aq", "ar", "as", "at", "au", "aw", "ax", "az", "ba", "bb", "bd", "be", "bf", "bg", "bh", "bi", "bj", "bm", "bn", "bo", "br", "bs", "bt", "bv", "bw", "by", "bz", "ca", "cc", "cd", "cf", "cg", "ch", "ci", "ck", "cl", "cm", "cn", "co", "cr", "cu", "cv", "cw", "cx", "cy", "cz", "de", "dj", "dk", "dm", "do", "dz", "ec", "ee", "eg", "er", "es", "et", "eu", "fi", "fj", "fk", "fm", "fo", "fr", "ga", "gb", "gd", "ge", "gf", "gg", "gh", "gi", "gl", "gm", "gn", "gp", "gq", "gr", "gs", "gt", "gu", "gw", "gy", "hk", "hm", "hn", "hr", "ht", "hu", "id", "ie", "il", "im", "in", "io", "iq", "ir", "is", "it", "je", "jm", "jo", "jp", "ke", "kg", "kh", "ki", "km", "kn", "kp", "kr", "kw", "ky", "kz", "la", "lb", "lc", "li", "lk", "lr", "ls", "lt", "lu", "lv", "ly", "ma", "mc", "md", "me", "mg", "mh", "mk", "ml", "mm", "mn", "mo", "mp", "mq", "mr", "ms", "mt", "mu", "mv", "mw", "mx", "my", "mz", "na", "nc", "ne", "nf", "ng", "ni", "nl", "no", "np", "nr", "nu", "nz", "om", "pa", "pe", "pf", "pg", "ph", "pk", "pl", "pm", "pn", "pr", "ps", "pt", "pw", "py", "qa", "re", "ro", "rs", "ru", "rw", "sa", "sb", "sc", "sd", "se", "sg", "sh", "si", "sj", "sk", "sl", "sm", "sn", "so", "sr", "ss", "st", "su", "sv", "sx", "sy", "sz", "tc", "td", "tf", "tg", "th", "tj", "tk", "tl", "tm", "tn", "to", "tr", "tt", "tv", "tw", "tz", "ua", "ug", "uk", "us", "uy", "uz", "va", "vc", "ve", "vg", "vi", "vn", "vu", "wf", "ws", "ye", "yt", "za", "zm", "zw" ]; //#endregion //#region src/aux/regexes.ts const alphabets = "a-z"; const labelLetters = `${alphabets}0-9`; const oneOrMoreLabel = `[${labelLetters}]{1,63}`; const zeroOrMoreLabel = `[${labelLetters}]{0,63}`; const zeroOrMoreLabelWithHyphen = `[${labelLetters}-]{0,63}`; const nonDigitTwoOrMoreLabelWithHyphen = `[${alphabets}-]{2,63}`; const idnPrefix = "xn--"; const nonStrictTld = nonDigitTwoOrMoreLabelWithHyphen; const strictTld = tlds.join("|"); const asnRegex = /\b(AS|ASN)\d+\b/gi; const btcRegex = /\b[13][a-km-zA-HJ-NP-Z1-9]{26,33}\b/g; const cveRegExp = /(CVE-(19|20)\d{2}-\d{4,7})/gi; const ethRegex = /\b0x[a-fA-F0-9]{40}\b/g; const gaPubIDRegex = /\bpub-\d{16}\b/gi; const gaTrackIDRegex = /\bUA-\d{4,9}(-\d{1,2})?\b/gi; const macAddressRegex = /\b[A-F0-9]{2}([-:])(?:[A-F0-9]{2}\1){4}[A-F0-9]{2}\b/gi; const md5Regex = /\b[A-F0-9]{32}\b/gi; const sha1Regex = /\b[A-F0-9]{40}\b/gi; const sha256Regex = /\b[A-F0-9]{64}\b/gi; const sha512Regex = /\b[A-F0-9]{128}\b/gi; const ssdeepRegex = /\b\d+:[A-Z0-9/+]{3,}:[A-Z0-9/+]{3,}/gi; const xmrRegex = /\b4[0-9AB][1-9A-HJ-NP-Za-km-z]{93}\b/g; //#endregion //#region src/aux/domain.ts const domainRegexCache = new lru_cache.LRUCache({ max: 2 }); function buildDomainRegex(strict) { const tld = strict ? strictTld : nonStrictTld; const regex = `(?=[${labelLetters}.\\-]{1,252}\\.(${tld})\\b)((${idnPrefix}${zeroOrMoreLabel}|${oneOrMoreLabel})((?!.{0,63}--)${zeroOrMoreLabelWithHyphen}[${labelLetters}])?\\.)+(${tld})\\b`; const result = new RegExp(regex, "gi"); domainRegexCache.set(strict, result); return result; } function domainRegex(options = { strict: true }) { const strict = options.strict ?? true; return domainRegexCache.get(strict) ?? buildDomainRegex(strict); } //#endregion //#region src/aux/email.ts const emailRegexCache = new lru_cache.LRUCache({ max: 2 }); function buildEmailRegex(options) { const domainPart = domainRegex(options).source; const result = new RegExp(`[a-zA-Z0-9.!#\$%&'*+/=?^_\`{|}~-]+@${domainPart}`, "gi"); emailRegexCache.set(options.strict ?? true, result); return result; } function emailRegex(options = { strict: true }) { const strict = options.strict ?? true; return emailRegexCache.get(strict) ?? buildEmailRegex(options); } //#endregion //#region src/aux/ip.ts const word = "[a-fA-F\\d:]"; const boundary = (options) => options && options.includeBoundaries ? `(?:(?<=\\s|^)(?=${word})|(?<=${word})(?=\\s|$))` : ""; const v4 = "(?:25[0-5]|2[0-4]\\d|1\\d\\d|[1-9]\\d|\\d)(?:\\.(?:25[0-5]|2[0-4]\\d|1\\d\\d|[1-9]\\d|\\d)){3}"; const v6segment = "[a-fA-F\\d]{1,4}"; const v6 = ` (?: (?:${v6segment}:){7}(?:${v6segment}|:)| // 1:2:3:4:5:6:7:: 1:2:3:4:5:6:7:8 (?:${v6segment}:){6}(?:${v4}|:${v6segment}|:)| // 1:2:3:4:5:6:: 1:2:3:4:5:6::8 1:2:3:4:5:6::8 1:2:3:4:5:6::1.2.3.4 (?:${v6segment}:){5}(?::${v4}|(?::${v6segment}){1,2}|:)| // 1:2:3:4:5:: 1:2:3:4:5::7:8 1:2:3:4:5::8 1:2:3:4:5::7:1.2.3.4 (?:${v6segment}:){4}(?:(?::${v6segment}){0,1}:${v4}|(?::${v6segment}){1,3}|:)| // 1:2:3:4:: 1:2:3:4::6:7:8 1:2:3:4::8 1:2:3:4::6:7:1.2.3.4 (?:${v6segment}:){3}(?:(?::${v6segment}){0,2}:${v4}|(?::${v6segment}){1,4}|:)| // 1:2:3:: 1:2:3::5:6:7:8 1:2:3::8 1:2:3::5:6:7:1.2.3.4 (?:${v6segment}:){2}(?:(?::${v6segment}){0,3}:${v4}|(?::${v6segment}){1,5}|:)| // 1:2:: 1:2::4:5:6:7:8 1:2::8 1:2::4:5:6:7:1.2.3.4 (?:${v6segment}:){1}(?:(?::${v6segment}){0,4}:${v4}|(?::${v6segment}){1,6}|:)| // 1:: 1::3:4:5:6:7:8 1::8 1::3:4:5:6:7:1.2.3.4 (?::(?:(?::${v6segment}){0,5}:${v4}|(?::${v6segment}){1,7}|:)) // ::2:3:4:5:6:7:8 ::2:3:4:5:6:7:8 ::8 ::1.2.3.4 )(?:%[0-9a-zA-Z]{1,})? // %eth0 %1 `.replace(/\s*\/\/.*$/gm, "").replace(/\n/g, "").trim(); const ipRegex = (options) => new RegExp(`(?:${boundary(options)}${v4}${boundary(options)})|(?:${boundary(options)}${v6}${boundary(options)})`, "g"); const _v4Regex = new RegExp(`${v4}`, "g"); const _v6Regex = new RegExp(`${v6}`, "g"); ipRegex.v4 = (options) => options?.includeBoundaries ? new RegExp(`${boundary(options)}${v4}${boundary(options)}`, "g") : _v4Regex; ipRegex.v6 = (options) => options?.includeBoundaries ? new RegExp(`${boundary(options)}${v6}${boundary(options)}`, "g") : _v6Regex; //#endregion //#region src/aux/url.ts const urlRegexCache = new lru_cache.LRUCache({ max: 2 }); function buildUrlRegex(options) { const domainPart = domainRegex(options).source; const result = new RegExp(`(?:(?:(?:https?)://))(?:\\S+(?::\\S*)?@)?(?:${domainPart}|localhost|${ipRegex.v4().source})(?::\\d{2,5})?(?:[/?#][^\\s"]*)?`, "gi"); urlRegexCache.set(options.strict ?? true, result); return result; } function urlRegex(options = { strict: true }) { const strict = options.strict ?? true; return urlRegexCache.get(strict) ?? buildUrlRegex(options); } //#endregion //#region src/aux/utils.ts /** * Reject duplications from an array * * @param {string[]} array An array of strings * @returns {string[]} A set of strings */ function dedup(array) { return Array.from(new Set(array)); } /** * Soar an array by value * * @param {string[]} array An array of strings * @returns {string[]} A sorted array */ function sortByValue(array) { return array.sort(); } const DOT_RE = new RegExp([ /\s\.\s/, /([[({])\.([\])}])/, /([[({])\./, /\.([\])}])/, /\\\./, /([[({])dot([\])}])/ ].map((r) => r.source).join("|"), "gi"); const COLON_RE = /[[({]:[\])}]/g; const SLASH_RE = /[[({]\/[\])}]/g; const COLON_SLASH_RE = /[[({]:\/\/[\])}]/g; const AT_RE = /[[({](?:at|@)[\])}]/gi; const HTTP_RE = /h(?:xx|\*\*)p(s?):\/\//gi; function hasDot(s) { return [ "\\.", " . ", "[.", "(.", "{.", "[dot", "(dot", "{dot" ].some((x) => s.includes(x)); } function hasColon(s) { return [ "[:", "(:", "{:" ].some((x) => s.includes(x)); } function hasSlash(s) { return [ "[/", "(/", "{/" ].some((x) => s.includes(x)); } function hasColonDoubleSlash(s) { return [ "[://", "(://", "{://" ].some((x) => s.includes(x)); } function hasAt(s) { return [ "[@", "(@", "{@", "[at", "(at", "{at" ].some((x) => s.includes(x)); } function hasHttp(s) { return ["hxxp", "h**p"].some((x) => s.includes(x)); } /** * Remove defanged symbols from a string * * @param {string} s A string * @returns {string} A cleaned (aka refanged) string */ function refang(s) { if (hasDot(s)) s = s.replace(DOT_RE, "."); if (hasColon(s)) s = s.replace(COLON_RE, ":"); if (hasSlash(s)) s = s.replace(SLASH_RE, "/"); if (hasColonDoubleSlash(s)) s = s.replace(COLON_SLASH_RE, "://"); if (hasAt(s)) s = s.replace(AT_RE, "@"); if (hasHttp(s)) s = s.replace(HTTP_RE, "http$1://"); return s; } function unicodeToASCII(s) { return punycode_js.default.toASCII(s); } //#endregion //#region src/aux/extractors.ts /** * Perform String match() by using a regexp * * @param {string} s A string * @param {RegExp} regexp A regexp to use * @param {SortOptions} options * @returns {string[]} An array of matched strings, returns an empty array if not matched */ function matchesWithRegExp(s, regexp, options = { sort: true }) { const matched = s.match(regexp); const values = matched === null ? [] : dedup(matched); return options.sort ? sortByValue(values) : values; } const nonGlobalRegexCache = /* @__PURE__ */ new Map(); function getFirstMatchedValue(s, regexp) { if (regexp.global) { let cached = nonGlobalRegexCache.get(regexp.source); if (!cached) { cached = new RegExp(regexp.source, regexp.flags.replace("g", "")); nonGlobalRegexCache.set(regexp.source, cached); } regexp = cached; } const matched = s.match(regexp); return matched === null ? null : matched[0]; } /** * Extract MD5s from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of MD5s */ function extractMD5s(s, options = { sort: true }) { return matchesWithRegExp(s, md5Regex, options); } /** * Extract MD5 from a string * * @param {string} s A string * @returns {string | null} MD5 */ function extractMD5(s) { return getFirstMatchedValue(s, md5Regex); } /** * Extract SHA1s from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of SHA1s */ function extractSHA1s(s, options = { sort: true }) { return matchesWithRegExp(s, sha1Regex, options); } /** * Extract SHA1 from a string * * @param {string} s A string * @returns {string | null } SHA1 */ function extractSHA1(s) { return getFirstMatchedValue(s, sha1Regex); } /** * Extract SHA256s from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of SHA256s */ function extractSHA256s(s, options = { sort: true }) { return matchesWithRegExp(s, sha256Regex, options); } /** * Extract SHA256 from a string * * @param {string} s A string * @returns {string | null } SHA256 */ function extractSHA256(s) { return getFirstMatchedValue(s, sha256Regex); } /** * Extract SHA512s from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of SHA512s */ function extractSHA512s(s, options = { sort: true }) { return matchesWithRegExp(s, sha512Regex, options); } /** * Extract SHA512 from a string * * @param {string} s A string * @returns {string | null} SHA512 */ function extractSHA512(s) { return getFirstMatchedValue(s, sha512Regex); } /** * Extract SSDEEPs from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of SSDEEPs */ function extractSSDEEPs(s, options = { sort: true }) { return matchesWithRegExp(s, ssdeepRegex, options); } /** * Extract SSDEEP from a string * * @param {string} s A string * @returns {string | null} SSDEEP */ function extractSSDEEP(s) { return getFirstMatchedValue(s, ssdeepRegex); } /** * Extract ASNs from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of ASNs */ function extractASNs(s, options = { sort: true }) { if (!s.includes("AS")) return []; return matchesWithRegExp(s, asnRegex, options); } /** * Extract ASN from a string * * @param {string} s A string * @returns {string[]} ASN */ function extractASN(s) { if (!s.includes("AS")) return null; return getFirstMatchedValue(s, asnRegex); } /** * Extract domains from a string * * @param {string} s A string * @param {StrictSortOptions} options * @returns {string[]} An array of domains */ function extractDomains(s, options = { strict: true, sort: true }) { if (!s.includes(".")) return []; const values = matchesWithRegExp(s, domainRegex(options)); return options.sort ? sortByValue(values) : values; } /** * Extract domain from a string * * @param {string} s A string * @param {StrictOptions} options * @returns {string | null} Domain */ function extractDomain(s, options = { strict: true }) { if (!s.includes(".")) return null; return getFirstMatchedValue(s, domainRegex(options)); } /** * Extract emails from a string * * @param {string} s A string * @param {StrictSortOptions} options * @returns {string[]} An array of emails */ function extractEmails(s, options = { strict: true, sort: true }) { if (!s.includes("@") && !s.includes(".")) return []; const values = matchesWithRegExp(s, emailRegex(options)); return options.sort ? sortByValue(values) : values; } /** * Extract email from a string * * @param {string} s A string * @param {StrictOptions} options * @returns {string | null} Email */ function extractEmail(s, options = { strict: true }) { if (!s.includes("@") && !s.includes(".")) return null; return getFirstMatchedValue(s, emailRegex(options)); } /** * Extract IPv4s from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of IPv4s */ function extractIPv4s(s, options = { sort: true }) { if (!s.includes(".")) return []; return matchesWithRegExp(s, ipRegex.v4(), options); } /** * Extract IPv4 from a string * * @param {string} s A string * @returns {string | null} IPv4 */ function extractIPv4(s) { if (!s.includes(".")) return null; return getFirstMatchedValue(s, ipRegex.v4()); } /** * Extract IPv6s from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of IPv6s */ function extractIPv6s(s, options = { sort: true }) { return matchesWithRegExp(s, ipRegex.v6(), options); } /** * Extract IPv6 from a string * * @param {string} s A string * @returns {string | null} IPv6 */ function extractIPv6(s) { return getFirstMatchedValue(s, ipRegex.v6()); } /** * Extract URLs from a string * * @param {string} s A string * @param {StrictSortOptions} options * @returns {string[]} An array of URLs */ function extractURLs(s, options = { strict: true, sort: true }) { return matchesWithRegExp(s, urlRegex(options), options); } /** * Extract URL from a string * * @param {string} s A string * @param {StrictOptions} options * @returns {string | null} URL */ function extractURL(s, options = { strict: true }) { return getFirstMatchedValue(s, urlRegex(options)); } /** * Extract CVEs from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of CVEs */ function extractCVEs(s, options = { sort: true }) { return matchesWithRegExp(s, cveRegExp, options); } /** * Extract CVE from a string * * @param {string} s A string * @returns {string | null} CVE */ function extractCVE(s) { return getFirstMatchedValue(s, cveRegExp); } /** * Extract BTCs from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of BTCs */ function extractBTCs(s, options = { sort: true }) { return matchesWithRegExp(s, btcRegex, options); } /** * Extract BTC from a string * * @param {string} s A string * @returns {string | null} BTC */ function extractBTC(s) { return getFirstMatchedValue(s, btcRegex); } /** * Extract XMRs from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of XMRs */ function extractXMRs(s, options = { sort: true }) { return matchesWithRegExp(s, xmrRegex, options); } /** * Extract XMR from a string * * @param {string} s A string * @returns {string[]} XMR */ function extractXMR(s) { return getFirstMatchedValue(s, xmrRegex); } /** * Extract Google Adsense Publisher IDs from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of Google Adsense Publisher IDs */ function extractGAPubIDs(s, options = { sort: true }) { return matchesWithRegExp(s, gaPubIDRegex, options); } /** * Extract Google Adsense Publisher IDs from a string * * @param {string} s A string * @returns {string | null} Adsense Publisher ID */ function extractGAPubID(s) { return getFirstMatchedValue(s, gaPubIDRegex); } /** * Extract Google Analytics tracking IDs from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of Google Analytics tracking IDs */ function extractGATrackIDs(s, options = { sort: true }) { return matchesWithRegExp(s, gaTrackIDRegex, options); } /** * Extract Google Analytics tracking ID from a string * * @param {string} s A string * @returns {string[]} Google Analytics tracking ID */ function extractGATrackID(s) { return getFirstMatchedValue(s, gaTrackIDRegex); } /** * Extract mac addresses from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of mac addresses */ function extractMacAddresses(s, options = { sort: true }) { return matchesWithRegExp(s, macAddressRegex, options); } /** * Extract mac address from a string * * @param {string} s A string * @returns {string[]} Mac address */ function extractMacAddress(s) { return getFirstMatchedValue(s, macAddressRegex); } /** * Extract ETH addresses from a string * * @param {string} s A string * @param {SortOptions} options * @returns {string[]} An array of ETH addresses */ function extractETHs(s, options = { sort: true }) { return matchesWithRegExp(s, ethRegex, options); } /** * Extract ETH address from a string * * @param {string} s A string * @returns {string | null} ETH address */ function extractETH(s) { return getFirstMatchedValue(s, ethRegex); } //#endregion //#region src/aux/validators.ts /** * Check whether a string matches with a regexp or not * * @param {string} s A string * @param {RegExp} regexp A regexp * @returns {boolean} returns true if a string matches with a regexp */ function check(s, regexp) { const match = s.match(regexp); if (match === null) return false; return match[0].length == s.length; } /** * Check whether a string is a MD5 or not * * @param {string} s A string * @returns {boolean} return true if a string is MD5 */ function isMD5(s) { return check(s, md5Regex); } /** * Check whether a string is a SHA1 or not * * @param {string} s A string * @returns {boolean} return true if a string is a SHA1 */ function isSHA1(s) { return check(s, sha1Regex); } /** * Check whether a string is a SHA256 or not * * @param {string} s A string * @returns {boolean} return true if a string is a SHA256 */ function isSHA256(s) { return check(s, sha256Regex); } /** * Check whether a string is a SHA512 or not * * @param {string} s A string * @returns {boolean} return true if a string is a SHA512 */ function isSHA512(s) { return check(s, sha512Regex); } /** * Check whether a string is a SSDEEP or not * * @param {string} s A string * @returns {boolean} return true if a string is a SSDEEP */ function isSSDEEP(s) { return check(s, ssdeepRegex); } /** * Check whether a string is an ASN or not * * @param {string} s A string * @returns {boolean} return true if a string is an ASN */ function isASN(s) { return check(s, asnRegex); } /** * Check whether a string is a domain or not * * @param {string} s A string * @param {StrictOptions} options * @returns {boolean} return true if a string is a domain */ function isDomain(s, options = { strict: true }) { return check(s, domainRegex(options)); } /** * Check whether a string is an email or not * * @param {string} s A string * @param {StrictOptions} options * @returns {boolean} true if a string is a domain */ function isEmail(s, options = { strict: true }) { return check(s, emailRegex(options)); } /** * Check whether a string is an IPv4 or not * * @param {string} s A string * @returns {boolean} true if a string is an IPv4 */ function isIPv4(s) { return check(s, ipRegex.v4()); } /** * Check whether a string is an IPv6 or not * * @param {string} s A string * @returns {boolean} true if a string is an IPv6 */ function isIPv6(s) { return check(s, ipRegex.v6()); } /** * Check whether a string is a URL or not * * @param {string} s A string * @param {StrictOptions} options * @returns {boolean} true if a string is a URL */ function isURL(s, options = { strict: true }) { return check(s, urlRegex(options)); } /** * Check whether a string is a CVE or not * * @param {string} s A string * @returns {boolean} true if a string is a CVE */ function isCVE(s) { return check(s, cveRegExp); } /** * Check whether a string is a BTC or not * * @param {string} s A string * @returns {boolean} return true if a string is a BTC */ function isBTC(s) { return check(s, btcRegex); } /** * Check whether a string is an XMR or not * * @param {string} s A string * @returns {boolean} true if a string is an XMR */ function isXMR(s) { return check(s, xmrRegex); } /** * Check whether a string is a Google Adsense Publisher ID or not * * @param {string} s A string * @returns {boolean} true if a string is a Google Adsense Publisher ID */ function isGAPubID(s) { return check(s, gaPubIDRegex); } /** * Check whether a string is a Google Analytics tracking ID or not * * @param {string} s A string * @returns {boolean} true if a string is a Google Analytics tracking ID */ function isGATrackID(s) { return check(s, gaTrackIDRegex); } /** * Check whether a string is a mac address or not * * @param {string} s A string * @returns {boolean} true if a string is a mac address */ function isMacAddress(s) { return check(s, macAddressRegex); } /** * Check whether a string is an ETH address or not * * @param {string} s A string * @returns {boolean} true if a string is an ETH address */ function isETH(s) { return check(s, ethRegex); } //#endregion //#region src/index.ts var IOCExtractor = class { s; constructor(s) { this.s = s; } /** * Extract IoCs from a string * * @returns {IOC} * @param {Options} options */ extractIOC(options = { strict: true, refang: true, punycode: false, sort: true }) { let normalized = options.refang ? refang(this.s) : this.s; normalized = options.punycode ? unicodeToASCII(normalized) : normalized; return { asns: extractASNs(normalized, options), btcs: extractBTCs(normalized, options), cves: extractCVEs(normalized, options), domains: extractDomains(normalized, options), emails: extractEmails(normalized, options), eths: extractETHs(normalized, options), gaPubIDs: extractGAPubIDs(normalized, options), gaTrackIDs: extractGATrackIDs(normalized, options), ipv4s: extractIPv4s(normalized, options), ipv6s: extractIPv6s(normalized, options), macAddresses: extractMacAddresses(normalized, options), md5s: extractMD5s(normalized, options), sha1s: extractSHA1s(normalized, options), sha256s: extractSHA256s(normalized, options), sha512s: extractSHA512s(normalized, options), ssdeeps: extractSSDEEPs(normalized, options), urls: extractURLs(normalized, options), xmrs: extractXMRs(normalized, options) }; } /** * Partially extract IoCs a string * * @returns {IOC} * @param {IOCKey[]} only * @param {Options} options */ partialExtractIOC(only, options = { strict: true, refang: true, punycode: false, sort: true }) { let normalized = options.refang ? refang(this.s) : this.s; normalized = options.punycode ? unicodeToASCII(normalized) : normalized; const funcByType = { asns: extractASNs, btcs: extractBTCs, cves: extractCVEs, domains: extractDomains, emails: extractEmails, eths: extractETHs, gaPubIDs: extractGAPubIDs, gaTrackIDs: extractGATrackIDs, ipv4s: extractIPv4s, ipv6s: extractIPv6s, macAddresses: extractMacAddresses, md5s: extractMD5s, sha1s: extractSHA1s, sha256s: extractSHA256s, sha512s: extractSHA512s, ssdeeps: extractSSDEEPs, urls: extractURLs, xmrs: extractXMRs }; return Object.fromEntries(only.map((key) => [key, funcByType[key](normalized, options)])); } }; /** * Extract IoCs from a string * * @param {string} s A string * @param {Options} options * @returns {IOC} */ function extractIOC(s, options = { strict: true, refang: true, punycode: false, sort: true }) { return new IOCExtractor(s).extractIOC(options); } /** * Partially extract IoCs from a string * * @param {string} s A string * @param {Options} options * @returns {IOC} */ function partialExtractIOC(s, only, options = { strict: true, refang: true, punycode: false, sort: true }) { return new IOCExtractor(s).partialExtractIOC(only, options); } //#endregion exports.IOCExtractor = IOCExtractor; exports.extractASN = extractASN; exports.extractASNs = extractASNs; exports.extractBTC = extractBTC; exports.extractBTCs = extractBTCs; exports.extractCVE = extractCVE; exports.extractCVEs = extractCVEs; exports.extractDomain = extractDomain; exports.extractDomains = extractDomains; exports.extractETH = extractETH; exports.extractETHs = extractETHs; exports.extractEmail = extractEmail; exports.extractEmails = extractEmails; exports.extractGAPubID = extractGAPubID; exports.extractGAPubIDs = extractGAPubIDs; exports.extractGATrackID = extractGATrackID; exports.extractGATrackIDs = extractGATrackIDs; exports.extractIOC = extractIOC; exports.extractIPv4 = extractIPv4; exports.extractIPv4s = extractIPv4s; exports.extractIPv6 = extractIPv6; exports.extractIPv6s = extractIPv6s; exports.extractMD5 = extractMD5; exports.extractMD5s = extractMD5s; exports.extractMacAddress = extractMacAddress; exports.extractMacAddresses = extractMacAddresses; exports.extractSHA1 = extractSHA1; exports.extractSHA1s = extractSHA1s; exports.extractSHA256 = extractSHA256; exports.extractSHA256s = extractSHA256s; exports.extractSHA512 = extractSHA512; exports.extractSHA512s = extractSHA512s; exports.extractSSDEEP = extractSSDEEP; exports.extractSSDEEPs = extractSSDEEPs; exports.extractURL = extractURL; exports.extractURLs = extractURLs; exports.extractXMR = extractXMR; exports.extractXMRs = extractXMRs; exports.isASN = isASN; exports.isBTC = isBTC; exports.isCVE = isCVE; exports.isDomain = isDomain; exports.isETH = isETH; exports.isEmail = isEmail; exports.isGAPubID = isGAPubID; exports.isGATrackID = isGATrackID; exports.isIPv4 = isIPv4; exports.isIPv6 = isIPv6; exports.isMD5 = isMD5; exports.isMacAddress = isMacAddress; exports.isSHA1 = isSHA1; exports.isSHA256 = isSHA256; exports.isSHA512 = isSHA512; exports.isSSDEEP = isSSDEEP; exports.isURL = isURL; exports.isXMR = isXMR; exports.partialExtractIOC = partialExtractIOC; exports.refang = refang; exports.unicodeToASCII = unicodeToASCII;