#!/usr/bin/env python # coding=utf-8 # This tool was created by alard at https://gist.github.com/4fa302a54ea8b5aa5c28 import json import re import time import urllib from collections import deque from tornado import ioloop, httpclient, gen USER_AGENT = "Mozilla/5.0 (Windows; U; Windows NT 6.1; en-US) AppleWebKit/533.20.25 (KHTML, like Gecko) Version/5.0.4 Safari/533.20.27" class YahooBlogFinder(object): BLOCK_TIMEOUT = 120 WAIT = 10 def __init__(self, words_to_search, concurrent=1): self.queue = deque([ (1,w) for w in words_to_search ]) self.running = 0 self.concurrent = concurrent self.blogs_found = set() self.http_client = httpclient.AsyncHTTPClient() self.blocked_until = 0 def start(self): if self.blocked_until > time.time(): ioloop.IOLoop.instance().add_timeout(self.blocked_until, self.start) return while len(self.queue) > 0 and self.running < self.concurrent: begin, word = self.queue.popleft() self.check_one(begin, word) if self.running == 0: ioloop.IOLoop.instance().stop() def check_one(self, begin, word): self.running += 1 enc_word = urllib.quote(word.encode("utf-8")) url = ("http://vn.blog.search.yahoo.com/search?p=%s&fr=uh-yblog&n=100&b=%d" % (enc_word, begin)) print "(%s, %d) %s" % (word, begin, url) req = httpclient.HTTPRequest( url, connect_timeout=10, request_timeout=30, use_gzip=True, user_agent=USER_AGENT) req.begin = begin req.word = word self.http_client.fetch(req, self.handle_response) def handle_response(self, response): word = response.request.word begin = response.request.begin if response.error: if response.error.code == 999: print "(%s, %d): Error %d, blocked!" % (word, begin, response.error.code) self.queue.append((begin, word)) self.blocked_until = time.time() + self.BLOCK_TIMEOUT else: print "(%s, %d): Error %d" % (word, begin, response.error.code) self.queue.append((begin, word)) else: for url in re.findall(r'http://blog.yahoo.com/([^/"\'&<>]+)', response.body): self.blogs_found.add(url) if re.search(r'id="pg-next"', response.body): self.queue.append(( begin+100, word )) print "Blogs found: %d" % len(self.blogs_found) self.running -= 1 ioloop.IOLoop.instance().add_timeout(time.time() + self.WAIT, self.start) print "Loading keywords..." http_client = httpclient.HTTPClient() res = http_client.fetch("http://tracker.archiveteam.org:8124/request", method="POST", body="") task = json.loads(res.body) ybf = YahooBlogFinder(task["words"]) ybf.start() ioloop.IOLoop.instance().start() print "Submitting %d blogs." % len(ybf.blogs_found) json_body = json.dumps({ "id": task["id"], "found": [ a for a in ybf.blogs_found ], "done": task["words"] }) res = http_client.fetch("http://tracker.archiveteam.org:8124/submit", method="POST", body=json_body, headers={"Content-Type": "application/json"})