#!/usr/bin/python import httplib from HTMLParser import HTMLParser from random import randint import brathuhn import sys class WikiQuoteParser(HTMLParser): in_ul=0 in_li=0 h2_num=0 in_h2=0 stop_h2=False stop_div=True quote="" quotes=[] div_hist=0 in_div=0 forbidden_h2 = ["External links","Web","Filminfo", "Contents","Inhaltsverzeichnis"] def handle_starttag(self, tag, attrs): if tag=="ul": self.in_ul+=1 if tag=="li": self.in_li+=1 if tag=="h2": self.h2_num+=1 self.in_h2+=1 self.stop_h2=False if tag=="div": self.in_div+=1 for at in attrs: if at[0]=="id" and at[1]=="mw-content-text": self.div_hist=self.in_div self.stop_div=False def handle_endtag(self, tag): if tag=="div": if self.div_hist==self.in_div: self.div_hist=0 self.stop_div=True self.in_div-=1 if tag=="h2": self.in_h2-=1 if tag=="ul": self.in_ul-=1 if tag=="li": self.in_li-=1 if self.in_li==0: if self.quote!="": self.quotes.append(self.quote) self.quote="" def handle_data(self, data): if self.in_h2>0: for i in self.forbidden_h2: if data.startswith(i): self.stop_h2=True if self.stop_div or self.stop_h2: return if self.in_li==1 and self.in_ul==1 and self.h2_num>=1: self.quote+=data.replace("\n","") #http://en.wikiquote.org/wiki/Douglas_Adams def quote(msg,args): if len(args)!=3: print "Usage: "+args[0]+" language (two letter code) topic" else: try: print "start" conn = httplib.HTTPConnection(args[1][0:2]+".wikiquote.org") conn.request("GET", "/wiki/"+args[2]) print conn res = conn.getresponse() if res.status==200: # OK data=res.read() print "data:", args[1][0:2], args[2] wqp=WikiQuoteParser() wqp.feed(data) #for i in wqp.quotes: # print "-- ",i if len(wqp.quotes)==1: print wqp.quotes[0] elif len(wqp.quotes)>1: print wqp.quotes[randint(0,len(wqp.quotes)-1)] else: print "No quotes" else: print "Error %d %s" % (res.status, res.reason) conn.close() print "end" except: print "Unknown Error"