#!/usr/bin/env python
# coding=utf-8
# This tool was created by alard at https://gist.github.com/4fa302a54ea8b5aa5c28
import json
import re
import time
import urllib
from collections import deque
from tornado import ioloop, httpclient, gen
USER_AGENT = "Mozilla/5.0 (Windows; U; Windows NT 6.1; en-US) AppleWebKit/533.20.25 (KHTML, like Gecko) Version/5.0.4 Safari/533.20.27"
class YahooBlogFinder(object):
BLOCK_TIMEOUT = 120
WAIT = 10
def __init__(self, words_to_search, concurrent=1):
self.queue = deque([ (1,w) for w in words_to_search ])
self.running = 0
self.concurrent = concurrent
self.blogs_found = set()
self.http_client = httpclient.AsyncHTTPClient()
self.blocked_until = 0
def start(self):
if self.blocked_until > time.time():
ioloop.IOLoop.instance().add_timeout(self.blocked_until, self.start)
return
while len(self.queue) > 0 and self.running < self.concurrent:
begin, word = self.queue.popleft()
self.check_one(begin, word)
if self.running == 0:
ioloop.IOLoop.instance().stop()
def check_one(self, begin, word):
self.running += 1
enc_word = urllib.quote(word.encode("utf-8"))
url = ("http://vn.blog.search.yahoo.com/search?p=%s&fr=uh-yblog&n=100&b=%d" % (enc_word, begin))
print "(%s, %d) %s" % (word, begin, url)
req = httpclient.HTTPRequest(
url,
connect_timeout=10, request_timeout=30,
use_gzip=True, user_agent=USER_AGENT)
req.begin = begin
req.word = word
self.http_client.fetch(req, self.handle_response)
def handle_response(self, response):
word = response.request.word
begin = response.request.begin
if response.error:
if response.error.code == 999:
print "(%s, %d): Error %d, blocked!" % (word, begin, response.error.code)
self.queue.append((begin, word))
self.blocked_until = time.time() + self.BLOCK_TIMEOUT
else:
print "(%s, %d): Error %d" % (word, begin, response.error.code)
self.queue.append((begin, word))
else:
for url in re.findall(r'http://blog.yahoo.com/([^/"\'&<>]+)', response.body):
self.blogs_found.add(url)
if re.search(r'id="pg-next"', response.body):
self.queue.append(( begin+100, word ))
print "Blogs found: %d" % len(self.blogs_found)
self.running -= 1
ioloop.IOLoop.instance().add_timeout(time.time() + self.WAIT, self.start)
print "Loading keywords..."
http_client = httpclient.HTTPClient()
res = http_client.fetch("http://tracker.archiveteam.org:8124/request", method="POST", body="")
task = json.loads(res.body)
ybf = YahooBlogFinder(task["words"])
ybf.start()
ioloop.IOLoop.instance().start()
print "Submitting %d blogs." % len(ybf.blogs_found)
json_body = json.dumps({ "id": task["id"], "found": [ a for a in ybf.blogs_found ], "done": task["words"] })
res = http_client.fetch("http://tracker.archiveteam.org:8124/submit", method="POST",
body=json_body, headers={"Content-Type": "application/json"})
Comments