tuankiet65 icon

Yahoo Blog search

tuankiet65 | PRO | 12/07/12 11:56:40 AM UTC | 0 ⭐ | 315 👁️ | Never ⏰ | []
Python |

3.07 KB

|

None

|

0 👍

/

0 👎

#!/usr/bin/env python
# coding=utf-8
# This tool was created by alard at https://gist.github.com/4fa302a54ea8b5aa5c28
 
import json
import re
import time
import urllib
from collections import deque
 
from tornado import ioloop, httpclient, gen
 
USER_AGENT = "Mozilla/5.0 (Windows; U; Windows NT 6.1; en-US) AppleWebKit/533.20.25 (KHTML, like Gecko) Version/5.0.4 Safari/533.20.27"
 
class YahooBlogFinder(object):
  BLOCK_TIMEOUT = 120
  WAIT = 10
 
  def __init__(self, words_to_search, concurrent=1):
    self.queue = deque([ (1,w) for w in words_to_search ])
    self.running = 0
    self.concurrent = concurrent
    self.blogs_found = set()
    self.http_client = httpclient.AsyncHTTPClient()
    self.blocked_until = 0
 
  def start(self):
    if self.blocked_until > time.time():
      ioloop.IOLoop.instance().add_timeout(self.blocked_until, self.start)
      return
 
    while len(self.queue) > 0 and self.running < self.concurrent:
      begin, word = self.queue.popleft()
      self.check_one(begin, word)
 
    if self.running == 0:
      ioloop.IOLoop.instance().stop()
 
  def check_one(self, begin, word):
    self.running += 1
 
    enc_word = urllib.quote(word.encode("utf-8"))
 
    url = ("http://vn.blog.search.yahoo.com/search?p=%s&fr=uh-yblog&n=100&b=%d" % (enc_word, begin))
    print "(%s, %d) %s" % (word, begin, url)
 
    req = httpclient.HTTPRequest(
        url,
        connect_timeout=10, request_timeout=30,
        use_gzip=True, user_agent=USER_AGENT)
    req.begin = begin
    req.word = word
    self.http_client.fetch(req, self.handle_response)
 
  def handle_response(self, response):
    word = response.request.word
    begin = response.request.begin
 
    if response.error:
      if response.error.code == 999:
        print "(%s, %d): Error %d, blocked!" % (word, begin, response.error.code)
        self.queue.append((begin, word))
        self.blocked_until = time.time() + self.BLOCK_TIMEOUT
 
      else:
        print "(%s, %d): Error %d" % (word, begin, response.error.code)
        self.queue.append((begin, word))
 
    else:
      for url in re.findall(r'http://blog.yahoo.com/([^/"\'&<>]+)', response.body):
        self.blogs_found.add(url)
 
      if re.search(r'id="pg-next"', response.body):
        self.queue.append(( begin+100, word ))
 
      print "Blogs found: %d" % len(self.blogs_found)
 
    self.running -= 1
    ioloop.IOLoop.instance().add_timeout(time.time() + self.WAIT, self.start)
 
 
 
 
print "Loading keywords..."
http_client = httpclient.HTTPClient()
res = http_client.fetch("http://tracker.archiveteam.org:8124/request", method="POST", body="")
task = json.loads(res.body)
 
ybf = YahooBlogFinder(task["words"])
ybf.start()
ioloop.IOLoop.instance().start()
 
print "Submitting %d blogs." % len(ybf.blogs_found)
json_body = json.dumps({ "id": task["id"], "found": [ a for a in ybf.blogs_found ], "done": task["words"] })
res = http_client.fetch("http://tracker.archiveteam.org:8124/submit", method="POST",
                        body=json_body, headers={"Content-Type": "application/json"})

Comments