diff --git a/Pipfile b/Pipfile index 85f8a1b..faa48f7 100644 --- a/Pipfile +++ b/Pipfile @@ -9,14 +9,14 @@ verify_ssl = true tqdm = "*" loguru = "*" dnspython = "*" -requests = {extras = ["socks"],version = "*"} -tldextract = "*" exrex = "*" fire = "*" bs4 = "*" tenacity = "*" treelib = "*" sqlalchemy = "*" +requests = "*" +pysocks = "*" [requires] python_version = "3.8" diff --git a/Pipfile.lock b/Pipfile.lock index 1d3f135..dd1c207 100644 --- a/Pipfile.lock +++ b/Pipfile.lock @@ -1,7 +1,7 @@ { "_meta": { "hash": { - "sha256": "7ad607a807779ea8674fd9d46eb4474f11349f766d94d721c23ad2b29a033105" + "sha256": "b593ca54685d3e36cb21706fd3ef3ee747c0d87fa7043fa483fcc30d316960eb" }, "pipfile-spec": 6, "requires": { @@ -90,11 +90,11 @@ }, "loguru": { "hashes": [ - "sha256:5aecbf13bc8e2f6e5a5d0475460a345b44e2885464095ea7de44e8795857ad33", - "sha256:a5e5e196b9958feaf534ac2050171d16576bae633074ce3e73af7dda7e9a58ae" + "sha256:b28e72ac7a98be3d28ad28570299a393dfcd32e5e3f6a353dec94675767b6319", + "sha256:f8087ac396b5ee5f67c963b495d615ebbceac2796379599820e324419d53667c" ], "index": "pypi", - "version": "==0.5.2" + "version": "==0.5.3" }, "pysocks": { "hashes": [ @@ -102,12 +102,10 @@ "sha256:2725bd0a9925919b9b51739eea5f9e2bae91e83288108a9ad338b2e3a4435ee5", "sha256:3f8804571ebe159c380ac6de37643bb4685970655d3bba243530d6558b799aa0" ], + "index": "pypi", "version": "==1.7.1" }, "requests": { - "extras": [ - "socks" - ], "hashes": [ "sha256:b3559a131db72c33ee969480840fff4bb6dd111de7dd27c8ee1f820f4f00231b", "sha256:fe75cc94a9443b9246fc7049224f75604b113c36acb93f87b80ed42c44cbb898" @@ -115,13 +113,6 @@ "index": "pypi", "version": "==2.24.0" }, - "requests-file": { - "hashes": [ - "sha256:07d74208d3389d01c38ab89ef403af0cfec63957d53a0081d8eca738d0247d8e", - "sha256:dfe5dae75c12481f68ba353183c53a65e6044c923e64c24b2209f6c7570ca953" - ], - "version": "==1.5.1" - }, "six": { "hashes": [ "sha256:30639c035cdb23534cd4aa2dd52c3bf48f06e5f4a941509c8bafd8ce11080259", @@ -188,21 +179,13 @@ ], "version": "==1.1.0" }, - "tldextract": { - "hashes": [ - "sha256:ab0e38977a129c72729476d5f8c85a8e1f8e49e9202e1db8dca76e95da7be9a8", - "sha256:c2a8a392edf3ea6fa8be80930f04c3ac29e91fa604cb2139bdf6a37fc1e1ac6d" - ], - "index": "pypi", - "version": "==2.2.3" - }, "tqdm": { "hashes": [ - "sha256:1a336d2b829be50e46b84668691e0a2719f26c97c62846298dd5ae2937e4d5cf", - "sha256:564d632ea2b9cb52979f7956e093e831c28d441c11751682f84c86fc46e4fd21" + "sha256:8f3c5815e3b5e20bc40463fa6b42a352178859692a68ffaa469706e6d38342a5", + "sha256:faf9c671bd3fad5ebaeee366949d969dca2b2be32c872a7092a1e1a9048d105b" ], "index": "pypi", - "version": "==4.48.2" + "version": "==4.49.0" }, "treelib": { "hashes": [ diff --git a/README.md b/README.md index fd02582..b9e54d0 100644 --- a/README.md +++ b/README.md @@ -113,7 +113,7 @@ pipenv run python oneforall.py --target example.com run ``` 3. 开启爆破模块运行(使用massdns进行爆破,网络占用极大,可能会阻塞网络) ```bash -python3 run python oneforall.py --target example.com --brute True run +python3 oneforall.py --target example.com --brute True run # or pipenv run python oneforall.py --target example.com --brute True run ``` @@ -200,7 +200,7 @@ ARGUMENTS FLAGS --brute=BRUTE - 使用爆破模块(默认False) + s --dns=DNS DNS解析子域(默认True) --req=REQ diff --git a/common/domain.py b/common/domain.py index 49cfab8..cd1a68b 100644 --- a/common/domain.py +++ b/common/domain.py @@ -1,5 +1,5 @@ import re -import tldextract +from common import tldextract from config import settings @@ -38,7 +38,7 @@ class Domain(object): """ data_storage_dir = settings.data_storage_dir extract_cache_file = data_storage_dir.joinpath('public_suffix_list.dat') - ext = tldextract.TLDExtract(extract_cache_file, None) + ext = tldextract.TLDExtract(extract_cache_file) result = self.match() if result: return ext(result) diff --git a/common/ipreg.py b/common/ipreg.py index b65d99e..32cb992 100644 --- a/common/ipreg.py +++ b/common/ipreg.py @@ -1,4 +1,3 @@ -# encoding = uft-8 """ " ip2region python searcher client module " @@ -6,9 +5,9 @@ " Date : 2015-11-06 """ import io +import sys import socket import struct -import sys from config import settings @@ -55,7 +54,7 @@ class IpRegInfo(object): else: eip = self.get_long(self.__dbBinStr, p + 4) if ip > eip: - l = m + 1; + l = m + 1 else: data_ptr = self.get_long(self.__dbBinStr, p + 8) break @@ -65,132 +64,6 @@ class IpRegInfo(object): return self.return_data(data_ptr) - def binary_search(self, ip): - """ - " binary search method - " param: ip - """ - if not ip.isdigit(): - ip = self.ip2long(ip) - - if self.__indexCount == 0: - self.__f.seek(0) - super_block = self.__f.read(8) - self.__indexSPtr = self.get_long(super_block, 0) - self.__indexLPtr = self.get_long(super_block, 4) - self.__indexCount = int((self.__indexLPtr - self.__indexSPtr) / - self.__INDEX_BLOCK_LENGTH) + 1 - - l, h, data_ptr = (0, self.__indexCount, 0) - while l <= h: - m = int((l + h) >> 1) - p = m * self.__INDEX_BLOCK_LENGTH - - self.__f.seek(self.__indexSPtr + p) - buffer = self.__f.read(self.__INDEX_BLOCK_LENGTH) - sip = self.get_long(buffer, 0) - if ip < sip: - h = m - 1 - else: - eip = self.get_long(buffer, 4) - if ip > eip: - l = m + 1 - else: - data_ptr = self.get_long(buffer, 8) - break - - if data_ptr == 0: - raise Exception("Data pointer not found") - - return self.return_data(data_ptr) - - def btree_search(self, ip): - """ - " b-tree search method - " param: ip - """ - if not ip.isdigit(): - ip = self.ip2long(ip) - - if len(self.__headerSip) < 1: - header_len = 0 - # pass the super block - self.__f.seek(8) - # read the header block - b = self.__f.read(self.__TOTAL_HEADER_LENGTH) - # parse the header block - for i in range(0, len(b), 8): - sip = self.get_long(b, i) - ptr = self.get_long(b, i + 4) - if ptr == 0: - break - self.__headerSip.append(sip) - self.__headerPtr.append(ptr) - header_len += 1 - self.__headerLen = header_len - - l, h, sptr, eptr = (0, self.__headerLen, 0, 0) - while l <= h: - m = int((l + h) >> 1) - - if ip == self.__headerSip[m]: - if m > 0: - sptr = self.__headerPtr[m - 1] - eptr = self.__headerPtr[m] - else: - sptr = self.__headerPtr[m] - eptr = self.__headerPtr[m + 1] - break - - if ip < self.__headerSip[m]: - if m == 0: - sptr = self.__headerPtr[m] - eptr = self.__headerPtr[m + 1] - break - elif ip > self.__headerSip[m - 1]: - sptr = self.__headerPtr[m - 1] - eptr = self.__headerPtr[m] - break - h = m - 1 - else: - if m == self.__headerLen - 1: - sptr = self.__headerPtr[m - 1] - eptr = self.__headerPtr[m] - break - elif ip <= self.__headerSip[m + 1]: - sptr = self.__headerPtr[m] - eptr = self.__headerPtr[m + 1] - break - l = m + 1 - - if sptr == 0: - raise Exception("Index pointer not found") - - index_len = eptr - sptr - self.__f.seek(sptr) - index = self.__f.read(index_len + self.__INDEX_BLOCK_LENGTH) - - l, h, data_prt = (0, int(index_len / self.__INDEX_BLOCK_LENGTH), 0) - while l <= h: - m = int((l + h) >> 1) - offset = int(m * self.__INDEX_BLOCK_LENGTH) - sip = self.get_long(index, offset) - - if ip < sip: - h = m - 1 - else: - eip = self.get_long(index, offset + 4) - if ip > eip: - l = m + 1; - else: - data_prt = self.get_long(index, offset + 8) - break - - if data_prt == 0: - raise Exception("Data pointer not found") - - return self.return_data(data_prt) - def init_database(self, db_file): """ " initialize the database for search @@ -205,7 +78,7 @@ class IpRegInfo(object): def return_data(self, data_ptr): """ " get ip data from db file by data start ptr - " param: dsptr + " param: data ptr """ data_len = (data_ptr >> 24) & 0xFF data_ptr = data_ptr & 0x00FFFFFF @@ -255,16 +128,8 @@ class IpRegData(IpRegInfo): path = settings.data_storage_dir.joinpath('ip2region.db') IpRegInfo.__init__(self, path) - def query(self, ip, algorithm='memory'): - algorithms = ['memory', 'binary', 'btree'] - if algorithm not in algorithms: - raise Exception(f"Only three query algorithms are supported: {algorithms}") - if algorithm == 'memory': - result = self.memory_search(ip) - elif algorithm == 'binary': - result = self.binary_search(ip) - else: - result = self.btree_search(ip) + def query(self, ip): + result = self.memory_search(ip) addr_list = result.get('region').split('|') addr = ''.join(filter(lambda x: x != '0', addr_list[:-1])) isp = addr_list[-1] diff --git a/common/module.py b/common/module.py index b04835b..b56be1d 100644 --- a/common/module.py +++ b/common/module.py @@ -87,7 +87,7 @@ class Module(object): verify=self.verify, **kwargs) except Exception as e: - logger.log('ERROR', e.args) + logger.log('ERROR', e.args[0]) return None if not check: return resp @@ -124,9 +124,9 @@ class Module(object): except Exception as e: if raise_error: if isinstance(e, requests.exceptions.ConnectTimeout): - logger.log(level, e.args) + logger.log(level, e.args[0]) raise e - logger.log(level, e.args) + logger.log(level, e.args[0]) return None if not check: return resp @@ -156,7 +156,7 @@ class Module(object): verify=self.verify, **kwargs) except Exception as e: - logger.log('ERROR', e.args) + logger.log('ERROR', e.args[0]) return None if not check: return resp @@ -184,7 +184,7 @@ class Module(object): verify=self.verify, **kwargs) except Exception as e: - logger.log('ERROR', e.args) + logger.log('ERROR', e.args[0]) return None if not check: return resp diff --git a/common/tldextract.py b/common/tldextract.py new file mode 100644 index 0000000..68a8e6f --- /dev/null +++ b/common/tldextract.py @@ -0,0 +1,240 @@ +# -*- coding: utf-8 -*- +"""`tldextract` accurately separates the gTLD or ccTLD (generic or country code +top-level domain) from the registered domain and subdomains of a URL. + + >>> import tldextract + + >>> tldextract.extract('http://forums.news.cnn.com/') + ExtractResult(subdomain='forums.news', domain='cnn', suffix='com') + + >>> tldextract.extract('http://forums.bbc.co.uk/') # United Kingdom + ExtractResult(subdomain='forums', domain='bbc', suffix='co.uk') + + >>> tldextract.extract('http://www.worldbank.org.kg/') # Kyrgyzstan + ExtractResult(subdomain='www', domain='worldbank', suffix='org.kg') + +`ExtractResult` is a namedtuple, so it's simple to access the parts you want. + + >>> ext = tldextract.extract('http://forums.bbc.co.uk') + >>> (ext.subdomain, ext.domain, ext.suffix) + ('forums', 'bbc', 'co.uk') + >>> # rejoin subdomain and domain + >>> '.'.join(ext[:2]) + 'forums.bbc' + >>> # a common alias + >>> ext.registered_domain + 'bbc.co.uk' + +Note subdomain and suffix are _optional_. Not all URL-like inputs have a +subdomain or a valid suffix. + + >>> tldextract.extract('google.com') + ExtractResult(subdomain='', domain='google', suffix='com') + + >>> tldextract.extract('google.notavalidsuffix') + ExtractResult(subdomain='google', domain='notavalidsuffix', suffix='') + + >>> tldextract.extract('http://127.0.0.1:8080/deployed/') + ExtractResult(subdomain='', domain='127.0.0.1', suffix='') + +If you want to rejoin the whole namedtuple, regardless of whether a subdomain +or suffix were found: + + >>> ext = tldextract.extract('http://127.0.0.1:8080/deployed/') + >>> # this has unwanted dots + >>> '.'.join(ext) + '.127.0.0.1.' +""" + + +import os +import re +import json +import collections +from urllib.parse import scheme_chars +from functools import wraps + +import idna + +from common import utils + +IP_RE = re.compile(r'^(([0-9]|[1-9][0-9]|1[0-9]{2}|2[0-4][0-9]|25[0-5])\.){3}([0-9]|[1-9][0-9]|1[0-9]{2}|2[0-4][0-9]|25[0-5])$') # pylint: disable=line-too-long + +SCHEME_RE = re.compile(r'^([' + scheme_chars + ']+:)?//') + + +class ExtractResult(collections.namedtuple('ExtractResult', 'subdomain domain suffix')): + """namedtuple of a URL's subdomain, domain, and suffix.""" + + # Necessary for __dict__ member to get populated in Python 3+ + __slots__ = () + + @property + def registered_domain(self): + """ + Joins the domain and suffix fields with a dot, if they're both set. + + >>> extract('http://forums.bbc.co.uk').registered_domain + 'bbc.co.uk' + >>> extract('http://localhost:8080').registered_domain + '' + """ + if self.domain and self.suffix: + return self.domain + '.' + self.suffix + return '' + + @property + def fqdn(self): + """ + Returns a Fully Qualified Domain Name, if there is a proper domain/suffix. + + >>> extract('http://forums.bbc.co.uk/path/to/file').fqdn + 'forums.bbc.co.uk' + >>> extract('http://localhost:8080').fqdn + '' + """ + if self.domain and self.suffix: + # self is the namedtuple (subdomain domain suffix) + return '.'.join(i for i in self if i) + return '' + + @property + def ipv4(self): + """ + Returns the ipv4 if that is what the presented domain/url is + + >>> extract('http://127.0.0.1/path/to/file').ipv4 + '127.0.0.1' + >>> extract('http://127.0.0.1.1/path/to/file').ipv4 + '' + >>> extract('http://256.1.1.1').ipv4 + '' + """ + if not (self.suffix or self.subdomain) and IP_RE.match(self.domain): + return self.domain + return '' + + +class TLDExtract(object): + """A callable for extracting, subdomain, domain, and suffix components from a URL.""" + + def __init__(self, cache_file=None): + """ + Constructs a callable for extracting subdomain, domain, and suffix + components from a URL. + """ + + self.cache_file = os.path.expanduser(cache_file or '') + self._extractor = None + + def __call__(self, url): + """ + Takes a string URL and splits it into its subdomain, domain, and + suffix (effective TLD, gTLD, ccTLD, etc.) component. + + >>> ext = TLDExtract() + >>> ext('http://forums.news.cnn.com/') + ExtractResult(subdomain='forums.news', domain='cnn', suffix='com') + >>> ext('http://forums.bbc.co.uk/') + ExtractResult(subdomain='forums', domain='bbc', suffix='co.uk') + """ + netloc = SCHEME_RE.sub("", url) \ + .partition("/")[0] \ + .partition("?")[0] \ + .partition("#")[0] \ + .split("@")[-1] \ + .partition(":")[0] \ + .strip() \ + .rstrip(".") + + labels = netloc.split(".") + + translations = [_decode_punycode(label) for label in labels] + suffix_index = self._get_tld_extractor().suffix_index(translations) + + suffix = ".".join(labels[suffix_index:]) + if not suffix and netloc and utils.looks_like_ip(netloc): + return ExtractResult('', netloc, '') + + subdomain = ".".join(labels[:suffix_index - 1]) if suffix_index else "" + domain = labels[suffix_index - 1] if suffix_index else "" + return ExtractResult(subdomain, domain, suffix) + + @property + def tlds(self): + return self._get_tld_extractor().tlds + + def _get_tld_extractor(self): + """Get or compute this object's TLDExtractor. Looks up the TLDExtractor + in roughly the following order, based on the settings passed to + __init__: + + 1. Memoized on `self` + 2. Local system cache file""" + # pylint: disable=no-else-return + + if self._extractor: + return self._extractor + tlds = self._get_cached_tlds() + if tlds: + self._extractor = _PublicSuffixListTLDExtractor(tlds) + return self._extractor + else: + raise Exception("tlds is empty, cannot proceed without tlds.") + + def _get_cached_tlds(self): + """Read the local TLD cache file. Returns None on IOError or other + error, or if this object is not set to use the cache + file.""" + if not self.cache_file: + return None + + with open(self.cache_file) as cache_file: + return json.loads(cache_file.read()) + + +TLD_EXTRACTOR = TLDExtract() + + +@wraps(TLD_EXTRACTOR.__call__) +def extract(url): + return TLD_EXTRACTOR(url) + + +class _PublicSuffixListTLDExtractor(object): + """Wrapper around this project's main algo for PSL + lookups. + """ + def __init__(self, tlds): + self.tlds = frozenset(tlds) + + def suffix_index(self, lower_spl): + """Returns the index of the first suffix label. + Returns len(spl) if no suffix is found + """ + length = len(lower_spl) + for i in range(length): + maybe_tld = '.'.join(lower_spl[i:]) + exception_tld = '!' + maybe_tld + if exception_tld in self.tlds: + return i + 1 + + if maybe_tld in self.tlds: + return i + + wildcard_tld = '*.' + '.'.join(lower_spl[i + 1:]) + if wildcard_tld in self.tlds: + return i + + return length + + +def _decode_punycode(label): + lowered = label.lower() + looks_like_puny = lowered.startswith('xn--') + if looks_like_puny: + try: + return idna.decode(label.encode('ascii')).lower() + except (UnicodeError, IndexError): + pass + return lowered diff --git a/common/utils.py b/common/utils.py index 2d9f949..5a08d10 100644 --- a/common/utils.py +++ b/common/utils.py @@ -1,12 +1,14 @@ -import json import os -import platform -import random import re -import string -import subprocess import sys import time +import json +import socket +import random +import string +import platform +import subprocess +from urllib.parse import scheme_chars from ipaddress import IPv4Address, ip_address from pathlib import Path from stat import S_IXUSR @@ -34,6 +36,9 @@ user_agents = [ 'Gecko/20100101 Firefox/68.0', 'Mozilla/5.0 (X11; Linux i586; rv:31.0) Gecko/20100101 Firefox/68.0'] +IP_RE = re.compile(r'^(([0-9]|[1-9][0-9]|1[0-9]{2}|2[0-4][0-9]|25[0-5])\.){3}([0-9]|[1-9][0-9]|1[0-9]{2}|2[0-4][0-9]|25[0-5])$') # pylint: disable=line-too-long +SCHEME_RE = re.compile(r'^([' + scheme_chars + ']+:)?//') + def gen_random_ip(): """ @@ -748,12 +753,18 @@ def ping_avg_time(nameserver): logger.log('ALERT', f'100.0% packet loss, ping {nameserver} failed.') return None elif platform.system() in ('Darwin', 'Linux'): - avg_time = re.findall(r'(?:min/avg/max/.+ )(?:\d+\.\d+)/(\d+\.\d+)/', text)[0] - logger.log('INFOR', f'ping {nameserver} average time {avg_time} ms.') + try: + avg_time = re.findall(r'(?:min/avg/max/.+ )(?:\d+\.\d+)/(\d+\.\d+)/', text)[0] + logger.log('INFOR', f'ping {nameserver} average time {avg_time} ms.') + except IndexError: + return None return avg_time elif platform.system() == 'Windows': - avg_time = re.findall(r'(?:Average|平均).+(\d.)ms', text)[0] - logger.log('INFOR', f'ping {nameserver} average time {avg_time} ms.') + try: + avg_time = re.findall(r'(?:Average|平均).+(\d.?)ms', text)[0] + logger.log('INFOR', f'ping {nameserver} average time {avg_time} ms.') + except IndexError: + return None return avg_time else: logger.log('ALERT', f'{text}') @@ -807,3 +818,18 @@ def default_nameserver(): logger.log('ERROR', 'Resolver configuration could not be read ' 'or specified no nameservers.') exit(1) + + +def looks_like_ip(maybe_ip): + """Does the given str look like an IP address?""" + if not maybe_ip[0].isdigit(): + return False + + try: + socket.inet_aton(maybe_ip) + return True + except (AttributeError, UnicodeError): + if IP_RE.match(maybe_ip): + return True + except socket.error: + return False \ No newline at end of file diff --git a/data/cdn_header_keys.json b/data/cdn_header_keys.json index 2f5e264..9c4fda9 100644 --- a/data/cdn_header_keys.json +++ b/data/cdn_header_keys.json @@ -2,6 +2,8 @@ "xcs", "via", "x-via", + "x-cdn", + "x-cdn-forward", "x-ser", "x-cf1", "cache", @@ -11,6 +13,7 @@ "x-hit-cache", "x-cache-status", "x-cache-hits", + "x-cache-lookup", "cc_cache", "webcache", "chinacache", @@ -20,14 +23,20 @@ "x-github-request-id", "x-sucuri-id", "x-amz-cf-id", + "x-airee-node", "x-cdn-provider", "x-fastly", + "x-iinfo", + "x-llid", + "sozu-id", + "x-cf-tsc", "x-ws-request-id", "fss-cache", "powered-by-chinacache", "verycdn", "yunjiasu", "skyparkcdn", + "x-beluga-cache-status", "x-content-type-options", "x-download-options", "x-proxy-node", @@ -36,5 +45,6 @@ "etag", "expires", "pragma", - "cache-control" + "cache-control", + "last-modified" ] diff --git a/docs/en-us/README.md b/docs/en-us/README.md index a6295f4..b74ed48 100644 --- a/docs/en-us/README.md +++ b/docs/en-us/README.md @@ -106,7 +106,7 @@ pipenv run python oneforall.py --target example.com run 3. Turn on brute modules, run the following command(Use massdns for enumerating subdomains, the network may be blocked): ```bash -python3 run python oneforall.py --target example.com --brute True run +python3 oneforall.py --target example.com --brute True run # or pipenv run python oneforall.py --target example.com --brute True run ``` @@ -191,7 +191,7 @@ ARGUMENTS FLAGS --brute=BRUTE - Use brute module (default False) + Use brute module (default True) --dns=DNS Use DNS resolution (default True) --req=REQ diff --git a/docs/usage_help.md b/docs/usage_help.md index 39231c9..491a574 100644 --- a/docs/usage_help.md +++ b/docs/usage_help.md @@ -48,7 +48,7 @@ OneForAll命令行界面基于[Fire](https://github.com/google/python-fire/)实 --targets=TARGETS 每行一个域名的文件路径 --brute=BRUTE - 使用爆破模块(默认False) + 使用爆破模块(默认True) --dns=DNS 开启子域解析(默认True) --req=REQ diff --git a/requirements.txt b/requirements.txt index ce88e6a..9aba14b 100644 --- a/requirements.txt +++ b/requirements.txt @@ -9,17 +9,15 @@ exrex==0.10.5 fire==0.3.1 future==0.18.2 idna==2.10 -loguru==0.5.2 +loguru==0.5.3 pysocks==1.7.1 -requests-file==1.5.1 -requests[socks]==2.24.0 +requests==2.24.0 six==1.15.0 soupsieve==2.0.1 sqlalchemy==1.3.19 tenacity==6.2.0 termcolor==1.1.0 -tldextract==2.2.3 -tqdm==4.48.2 +tqdm==4.49.0 treelib==1.6.1 urllib3==1.25.10 win32-setctime==1.0.2 ; sys_platform == 'win32' \ No newline at end of file