import urllib2, exceptions from HTMLParser import HTMLParser from random import randint import brathuhn class WikiQuoteParser(HTMLParser): FORBIDDEN_H2 = ["External links","Web","Filminfo", "Contents","Inhaltsverzeichnis","Cast", "See also"] def __init__(self): HTMLParser.__init__(self) self.in_ul=0 self.in_li=0 self.h2_num=0 self.in_h2=0 self.stop_h2=False self.stop_div=True self.quote="" self.quotes=[] self.div_hist=0 self.in_div=0 def handle_starttag(self, tag, attrs): if tag=="ul": self.in_ul+=1 if tag=="li": self.in_li+=1 if tag=="h2": self.h2_num+=1 self.in_h2+=1 self.stop_h2=False if tag=="div": self.in_div+=1 for at in attrs: if at[0]=="id" and at[1]=="mw-content-text": self.div_hist=self.in_div self.stop_div=False def handle_endtag(self, tag): if tag=="div": if self.div_hist==self.in_div: self.div_hist=0 self.stop_div=True self.in_div-=1 if tag=="h2": self.in_h2-=1 if tag=="ul": self.in_ul-=1 if tag=="li": self.in_li-=1 if self.in_li==0: if self.quote!="": self.quotes.append(self.quote) self.quote="" def handle_data(self, data): if self.in_h2>0: for i in self.FORBIDDEN_H2: if data.startswith(i): self.stop_h2=True if self.stop_div or self.stop_h2: return if self.in_li==1 and self.in_ul==1 and self.h2_num>=1: self.quote+=data.replace("\n","") def search_wiki(code,term): results = [] url = "http://"+code[0:2]+".wikiquote.org/w/api.php?action=opensearch&search="+urllib2.quote(term) req = urllib2.Request(url, headers={'User-Agent' : "Magic Browser"}) con = urllib2.urlopen( req ) data=con.read() data=data[data[1:].index("[")+2:data.rindex("]")-1] for dat in data.split(","): results.append( dat[1:-1].decode('string_escape') ) return results def quote(msg,args): if len(args)!=3: brathuhn.sendMsg(brathuhn.ROOM,"Usage: "+args[0]+" language (two letter code) topic",'groupchat') else: try: search_results = search_wiki(args[1],args[2]) if len( search_results )==0: brathuhn.sendMsg(brathuhn.ROOM,"No quotes found",'groupchat') url = "http://"+args[1][0:2]+".wikiquote.org/wiki/" + urllib2.quote(search_results[0]) req = urllib2.Request(url, headers={'User-Agent' : "Magic Browser"}) con = urllib2.urlopen( req ) data=con.read() print "data:", args[1][0:2], args[2],"->",search_results[0],url wqp=WikiQuoteParser() wqp.feed(data) #for i in wqp.quotes: # print "-- ",i if len(wqp.quotes)>=1: brathuhn.sendMsg(brathuhn.ROOM,'Quoting "'+ search_results[0] +'"') brathuhn.sendMsg(brathuhn.ROOM,wqp.quotes[randint(0,len(wqp.quotes)-1)],'groupchat') else: brathuhn.sendMsg(brathuhn.ROOM,'Could not crawl quotes from > '+url+' <','groupchat') con.close() except exceptions.Exception as e: brathuhn.sendMsg(brathuhn.ROOM,"Network Error "+unicode(e),'groupchat') brathuhn.addCommand("quote","Usage: !quote language (two letter code) topic",quote)