工具介绍
- 新增更多伪User-Agent
- 新增快代理进行随机IP+端口访问
- 因为CxxN和谐,移除了并发访问
- 后续增加UI界面、定时启动等前端功能操作
使用方法
复制代码进代码编辑器,main函数中的字符串修改成自己的UID,右键运行
完整代码
# coding:utf-8nimport jsonnimport randomnimport timennimport lxml.htmlnimport requestsnn# headersnHEADERS = {n "Accept": "*/*",n "Accept-Encoding": "gzip, deflate, br",n "Accept-Language": "zh-CN,zh;q=0.8,en-US;q=0.5,en;q=0.3",n "Cookie": "l=AurqcPuigwQdnQv7WvAfCoR1OlrRQW7h; isg=BHp6mNB79CHqYXpVEiRteXyyyKNcg8YEwjgLqoRvCI3ddxqxbLtOFUBGwwOrZ3ad; thw=cn; cna=VsJQERAypn0CATrXFEIahcz8; t=0eed37629fe7ef5ec0b8ecb6cd3a3577; tracknick=tb830309_22; _cc_=UtASsssmfA%3D%3D; tg=0; ubn=p; ucn=unzbyun; x=e%3D1%26p%3D*%26s%3D0%26c%3D0%26f%3D0%26g%3D0%26t%3D0%26__ll%3D-1%26_ato%3D0; miid=981798063989731689; hng=CN%7Czh-CN%7CCNY%7C156; um=0712F33290AB8A6D01951C8161A2DF2CDC7C5278664EE3E02F8F6195B27229B88A7470FD7B89F7FACD43AD3E795C914CC2A8BEB1FA88729A3A74257D8EE4FBBC; enc=1UeyOeN0l7Fkx0yPu7l6BuiPkT%2BdSxE0EqUM26jcSMdi1LtYaZbjQCMj5dKU3P0qfGwJn8QqYXc6oJugH%2FhFRA%3D%3D; ali_ab=58.215.20.66.1516409089271.6; mt=ci%3D-1_1; cookie2=104f8fc9c13eb24c296768a50cabdd6e; _tb_token_=ee7e1e1e7dbe7; v=0",n "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64;` rv:46.0) Gecko/20100102 Firefox/46.0"n}nn# user agentnUSER_AGENTS = [n 'Opera/8.0 (Windows NT 5.1; U; en)',n 'Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; en) Opera 9.50',n 'Mozilla/5.0 (Windows NT 6.1; WOW64; rv:34.0) Gecko/20100101 Firefox/34.0',n 'Mozilla/5.0 (Windows NT 5.1; U; en; rv:1.8.1) Gecko/20061208 Firefox/2.0.0 Opera 9.50',n 'Mozilla/5.0 (Windows NT 5.1) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1063.0 Safari/536.3',n 'Mozilla/5.0 (Windows NT 6.2) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1062.0 Safari/536.3',n 'Mozilla/5.0 (Windows NT 6.2) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1061.0 Safari/536.3',n 'Mozilla/5.0 (Windows NT 6.2) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1061.1 Safari/536.3',n 'Mozilla/5.0 (Windows NT 6.2) AppleWebKit/536.6 (KHTML, like Gecko) Chrome/20.0.1090.0 Safari/536.6',n 'Mozilla/5.0 (Windows NT 6.1) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1061.1 Safari/536.3',n 'Mozilla/5.0 (Windows NT 6.0) AppleWebKit/536.5 (KHTML, like Gecko) Chrome/19.0.1084.36 Safari/536.5',n 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/536.5 (KHTML, like Gecko) Chrome/19.0.1084.9 Safari/536.5',n 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/535.24 (KHTML, like Gecko) Chrome/19.0.1055.1 Safari/535.24',n 'Mozilla/5.0 (Windows NT 6.2; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/19.77.34.5 Safari/537.1',n 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.11 (KHTML, like Gecko) Chrome/23.0.1271.64 Safari/537.11',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/22.0.1207.1 Safari/537.1'n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1063.0 Safari/536.3',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/536.6 (KHTML, like Gecko) Chrome/20.0.1092.0 Safari/536.6',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1061.1 Safari/536.3',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1062.0 Safari/536.3',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/534.57.2 (KHTML, like Gecko) Version/5.1.7 Safari/534.57.2',n 'Mozilla/5.0 (Windows NT 6.2; WOW64) AppleWebKit/535.24 (KHTML, like Gecko) Chrome/19.0.1055.1 Safari/535.24',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/39.0.2171.71 Safari/537.36',n 'Mozilla/5.0 (X11; U; Linux x86_64; zh-CN; rv:1.9.2.10) Gecko/20100922 Ubuntu/10.10 (maverick) Firefox/3.6.10',n 'Mozilla/5.0 (X11; CrOS i686 2268.111.0) AppleWebKit/536.11 (KHTML, like Gecko) Chrome/20.0.1132.57 Safari/536.11',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.71 Safari/537.1 LBBROWSER',n 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_8_0) AppleWebKit/536.3 (KHTML, like Gecko) Chrome/19.0.1063.0 Safari/536.3',n 'Mozilla/5.0 (Windows NT 5.1) AppleWebKit/535.11 (KHTML, like Gecko) Chrome/17.0.963.84 Safari/535.11 SE 2.X MetaSr 1.0',n 'Mozilla/5.0 (Windows; U; Windows NT 6.1; en-US) AppleWebKit/534.16 (KHTML, like Gecko) Chrome/10.0.648.133 Safari/534.16',n 'Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 5.1; Trident/4.0; SV1; QQDownload 732; .NET4.0C; .NET4.0E; SE 2.X MetaSr 1.0)',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/536.11 (KHTML, like Gecko) Chrome/20.0.1132.11 TaoBrowser/2.0 Safari/536.11',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/39.0.2171.95 Safari/537.36 OPR/26.0.1656.60',n 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/38.0.2125.122 UBrowser/4.0.3214.0 Safari/537.36',n 'Mozilla/5.0 (compatible; MSIE 9.0; Windows NT 6.1; WOW64; Trident/5.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; .NET4.0C; .NET4.0E; LBBROWSER)'n]nnnclass CsdnBrosAdd:n # 实例化n def __init__(self, uid):n self.page = 0n self.proxy = []n self.url_list = self.get_doc_list(uid)nn # 获取用户uid的所有文章urln def get_doc_list(self, uid):n doc_set = set()n csdn_doc_list_uri = 'https://blog.csdn.net/community/home-api/v1/get-business-list?page=1&size=30&businessType=lately&noMore=false&username='n resp = requests.request("GET", csdn_doc_list_uri + uid, data='', headers=HEADERS)n resp.encoding = resp.apparent_encodingn js_rt = resp.textn json_rt = json.loads(js_rt)n for js_term in json_rt['data']['list']:n if uid in js_term['url']:n doc_set.add(js_term['url'])n print(js_term['url'])n return doc_setnn # 使用快代理获取代理IP和代理端口n def get_proxy(self):n self.page += 1n request = requests.get("https://www.kuaidaili.com/free/inha/" + str(self.page), headers=HEADERS)n html = request.contentn etree = lxml.html.etreen content = etree.HTML(html)n # print(html)n ip = content.xpath('//td[@data-title="IP"]/text()')n port = content.xpath('//td[@data-title="PORT"]/text()')n # 将对应的ip和port进行拼接n for i in range(len(ip)):n for p in range(len(port)):n if i == p:n if ip[i] + ':' + port[p] not in self.proxy:n self.proxy.append(ip[i] + ':' + port[p])n # print self.proxyn if self.proxy:n # print("this use" + str(self.page) + "page IP")n self.bros_add()nn # 遍历文章url列表伪装访问n def bros_add(self):n num = 0 # 用于访问计数n err_num = 0 # 用于异常错误计数n while True:n # 从列表中随机选择UA和代理n user_agent = random.choice(USER_AGENTS)n proxy = random.choice(self.proxy)n referer = random.choice(list(self.url_list)) # 随机选择访问url地址n headers = {n "Host": "blog.csdn.net",n "Connection": "keep-alive",n "Cache-Control": "max-age=0",n "Upgrade-Insecure-Requests": "1",n "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8",n "User-Agent": user_agent,n "Accept-Language": "zh-CN,zh;q=0.9",n "Cookie": "smidV2=2018061109563875e1d84afd822e152ccb71c406cf4e44009bb058182434650; uuid_tt_dd=10_37292961410-1553592218451-585832; UN=RUSH00000; Hm_ct_6bcd52f51e9b3dce32bec4a3997715ac=6525*1*10_37292961410-1553592218451-585832!5744*1*RUSH00000; acw_tc=2760824015580561068492252e3b68add4b06393ed45927064f907a6c5015c; __yadk_uid=szPOcQVFaUokWOOM2tzYYKteKJO5RAK6; acw_sc__v3=5ce219647dc08d09158df14c726ba51f46249bfd; acw_sc__v2=5ce21963f50279c4296b98184746d6a85a5f877d; dc_session_id=10_1558321509369.910879; Hm_lvt_6bcd52f51e9b3dce32bec4a3997715ac=1558054651,1558321511; SESSION=25351c44-6b87-48dc-b04b-a336aab6db77; UserName=RUSH00000; UserInfo=338ce29d15c840c98dbe43f5da028079; UserToken=338ce29d15c840c98dbe43f5da028079; UserNick=RUSH00000; AU=AAC; BT=1558321532133; dc_tos=prs8ld; Hm_lpvt_6bcd52f51e9b3dce32bec4a3997715ac=1558321538"n }n try:n # 构建一个Handler处理器对象,参数是一个字典类型,包括代理类型和代理服务器IP+PROTn request = requests.get(referer, headers=headers, proxies={"http": proxy})n html = request.contentn etree = lxml.html.etreen content = etree.HTML(html)n # 使用xpath匹配阅读量n read_num = content.xpath('//span[@class="read-count"]/text()')n # 将列表转为字符串n new_read_num = ''.join(read_num)n # 通过xpath匹配的页面为blog.csdn.net所以匹配其他页面返回的为空n if len(new_read_num) != 0:n print(new_read_num)n num += 1n # print('The' + str(num) + 'few visits')n # print(request.url + " proxy ip: " + str(proxy))n # print request.headersn time.sleep(5)n # 当访问数量达到100时,退出循环,并调用get_proxy方法获取第二页的代理n if num > 100:n breakn except Exception as result:n err_num += 1n # print("error message(%d):%s" % (err_num, result))n # 当错误信息大于等于30时,初始化代理页面page,重新从第一页开始获取代理ip,并退出循环n if err_num >= 30:n self.__init__()n breakn # 当退出循环时,看就会执行get_proxy获取代理的方法n # print("Re acquiring agent proxy IP")n self.get_proxy()nnnif __name__ == '__main__':n # qq_43376286n CsdnBrosAdd('weixin_44378305').get_proxy()
Comments NOTHING