| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281 |
- # -*- coding: utf-8 -*-
- """
- -------------------------------------------------
- File Name: proxyFetcher
- Description :
- Author : JHao
- date: 2016/11/25
- -------------------------------------------------
- Change Activity:
- 2016/11/25: proxyFetcher
- -------------------------------------------------
- """
- __author__ = 'JHao'
- import re
- from time import sleep
- from util.webRequest import WebRequest
- class ProxyFetcher(object):
- """
- proxy getter
- """
- @staticmethod
- def freeProxy01():
- """
- 无忧代理 http://www.data5u.com/
- 几乎没有能用的
- :return:
- """
- url_list = [
- 'http://www.data5u.com/',
- 'http://www.data5u.com/free/gngn/index.shtml',
- 'http://www.data5u.com/free/gnpt/index.shtml'
- ]
- key = 'ABCDEFGHIZ'
- for url in url_list:
- html_tree = WebRequest().get(url).tree
- ul_list = html_tree.xpath('//ul[@class="l2"]')
- for ul in ul_list:
- try:
- ip = ul.xpath('./span[1]/li/text()')[0]
- classnames = ul.xpath('./span[2]/li/attribute::class')[0]
- classname = classnames.split(' ')[1]
- port_sum = 0
- for c in classname:
- port_sum *= 10
- port_sum += key.index(c)
- port = port_sum >> 3
- yield '{}:{}'.format(ip, port)
- except Exception as e:
- print(e)
- @staticmethod
- def freeProxy02():
- """
- 代理66 http://www.66ip.cn/
- :return:
- """
- url = "http://www.66ip.cn/mo.php"
- resp = WebRequest().get(url, timeout=10)
- proxies = re.findall(r'(\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}:\d{1,5})', resp.text)
- for proxy in proxies:
- yield proxy
- @staticmethod
- def freeProxy03(page_count=1):
- """
- 西刺代理 http://www.xicidaili.com 网站已关闭
- :return:
- """
- url_list = [
- 'http://www.xicidaili.com/nn/', # 高匿
- 'http://www.xicidaili.com/nt/', # 透明
- ]
- for each_url in url_list:
- for i in range(1, page_count + 1):
- page_url = each_url + str(i)
- tree = WebRequest().get(page_url).tree
- proxy_list = tree.xpath('.//table[@id="ip_list"]//tr[position()>1]')
- for proxy in proxy_list:
- try:
- yield ':'.join(proxy.xpath('./td/text()')[0:2])
- except Exception as e:
- pass
- @staticmethod
- def freeProxy04():
- """
- 全网代理 http://www.goubanjia.com/
- :return:
- """
- url = "http://www.goubanjia.com/"
- tree = WebRequest().get(url).tree
- proxy_list = tree.xpath('//td[@class="ip"]')
- # 此网站有隐藏的数字干扰,或抓取到多余的数字或.符号
- # 需要过滤掉<p style="display:none;">的内容
- xpath_str = """.//*[not(contains(@style, 'display: none'))
- and not(contains(@style, 'display:none'))
- and not(contains(@class, 'port'))
- ]/text()
- """
- # port是class属性值加密得到
- def _parse_port(port_element):
- port_list = []
- for letter in port_element:
- port_list.append(str("ABCDEFGHIZ".find(letter)))
- _port = "".join(port_list)
- return int(_port) >> 0x3
- for each_proxy in proxy_list:
- try:
- ip_addr = ''.join(each_proxy.xpath(xpath_str))
- port_str = each_proxy.xpath(".//span[contains(@class, 'port')]/@class")[0].split()[-1]
- port = _parse_port(port_str.strip())
- yield '{}:{}'.format(ip_addr, int(port))
- except Exception:
- pass
- @staticmethod
- def freeProxy05(page_count=1):
- """
- 快代理 https://www.kuaidaili.com
- """
- url_pattern = [
- 'https://www.kuaidaili.com/free/inha/{}/',
- 'https://www.kuaidaili.com/free/intr/{}/'
- ]
- url_list = []
- for page_index in range(1, page_count + 1):
- for pattern in url_pattern:
- url_list.append(pattern.format(page_index))
- for url in url_list:
- tree = WebRequest().get(url).tree
- proxy_list = tree.xpath('.//table//tr')
- sleep(1) # 必须sleep 不然第二条请求不到数据
- for tr in proxy_list[1:]:
- yield ':'.join(tr.xpath('./td/text()')[0:2])
- @staticmethod
- def freeProxy06():
- """
- 代理盒子 https://proxy.coderbusy.com/
- :return:
- """
- urls = ['https://proxy.coderbusy.com/zh-hans/ops/country/cn.html']
- for url in urls:
- tree = WebRequest().get(url).tree
- proxy_list = tree.xpath('.//table//tr')
- for tr in proxy_list[1:]:
- proxy = '{}:{}'.format("".join(tr.xpath("./td[1]/text()")).strip(),
- "".join(tr.xpath("./td[2]//text()")).strip())
- if proxy:
- yield proxy
- @staticmethod
- def freeProxy07():
- """
- 云代理 http://www.ip3366.net/free/
- :return:
- """
- urls = ['http://www.ip3366.net/free/?stype=1',
- "http://www.ip3366.net/free/?stype=2"]
- for url in urls:
- r = WebRequest().get(url, timeout=10)
- proxies = re.findall(r'<td>(\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3})</td>[\s\S]*?<td>(\d+)</td>', r.text)
- for proxy in proxies:
- yield ":".join(proxy)
- @staticmethod
- def freeProxy08():
- """
- IP海 http://www.iphai.com/free/ng
- :return:
- """
- urls = [
- 'http://www.iphai.com/free/ng',
- 'http://www.iphai.com/free/np',
- 'http://www.iphai.com/free/wg',
- 'http://www.iphai.com/free/wp'
- ]
- for url in urls:
- r = WebRequest().get(url, timeout=10)
- proxies = re.findall(r'<td>\s*?(\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3})\s*?</td>[\s\S]*?<td>\s*?(\d+)\s*?</td>',
- r.text)
- for proxy in proxies:
- yield ":".join(proxy)
- @staticmethod
- def freeProxy09(page_count=1):
- """
- http://ip.jiangxianli.com/?page=
- 免费代理库
- :return:
- """
- for i in range(1, page_count + 1):
- url = 'http://ip.jiangxianli.com/?country=中国&page={}'.format(i)
- html_tree = WebRequest().get(url).tree
- for index, tr in enumerate(html_tree.xpath("//table//tr")):
- if index == 0:
- continue
- yield ":".join(tr.xpath("./td/text()")[0:2]).strip()
- # @staticmethod
- # def freeProxy10():
- # """
- # 墙外网站 cn-proxy
- # :return:
- # """
- # urls = ['http://cn-proxy.com/', 'http://cn-proxy.com/archives/218']
- # request = WebRequest()
- # for url in urls:
- # r = request.get(url, timeout=10)
- # proxies = re.findall(r'<td>(\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3})</td>[\w\W]<td>(\d+)</td>', r.text)
- # for proxy in proxies:
- # yield ':'.join(proxy)
- # @staticmethod
- # def freeProxy11():
- # """
- # https://proxy-list.org/english/index.php
- # :return:
- # """
- # urls = ['https://proxy-list.org/english/index.php?p=%s' % n for n in range(1, 10)]
- # request = WebRequest()
- # import base64
- # for url in urls:
- # r = request.get(url, timeout=10)
- # proxies = re.findall(r"Proxy\('(.*?)'\)", r.text)
- # for proxy in proxies:
- # yield base64.b64decode(proxy).decode()
- # @staticmethod
- # def freeProxy12():
- # urls = ['https://list.proxylistplus.com/Fresh-HTTP-Proxy-List-1']
- # request = WebRequest()
- # for url in urls:
- # r = request.get(url, timeout=10)
- # proxies = re.findall(r'<td>(\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3})</td>[\s\S]*?<td>(\d+)</td>', r.text)
- # for proxy in proxies:
- # yield ':'.join(proxy)
- @staticmethod
- def freeProxy13(max_page=2):
- """
- http://www.89ip.cn/index.html
- 89免费代理
- :param max_page:
- :return:
- """
- base_url = 'http://www.89ip.cn/index_{}.html'
- for page in range(1, max_page + 1):
- url = base_url.format(page)
- r = WebRequest().get(url, timeout=10)
- proxies = re.findall(
- r'<td.*?>[\s\S]*?(\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3})[\s\S]*?</td>[\s\S]*?<td.*?>[\s\S]*?(\d+)[\s\S]*?</td>',
- r.text)
- for proxy in proxies:
- yield ':'.join(proxy)
- @staticmethod
- def freeProxy14():
- """
- http://www.xiladaili.com/
- 西拉代理
- :return:
- """
- urls = ['http://www.xiladaili.com/putong/',
- "http://www.xiladaili.com/gaoni/",
- "http://www.xiladaili.com/http/",
- "http://www.xiladaili.com/https/"]
- for url in urls:
- r = WebRequest().get(url, timeout=10)
- ips = re.findall(r"\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}:\d{1,5}", r.text)
- for ip in ips:
- yield ip.strip()
|