Jabber/XMPP conference bot & framework
You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
brathuhn/modules/active/get_quote.py

111 lines
3.2 KiB

import urllib2, exceptions
from HTMLParser import HTMLParser
from random import randint
import brathuhn
class WikiQuoteParser(HTMLParser):
FORBIDDEN_H2 = ["External links","Web","Filminfo", "Contents","Inhaltsverzeichnis","Cast", "See also"]
def __init__(self):
HTMLParser.__init__(self)
self.in_ul=0
self.in_li=0
self.h2_num=0
self.in_h2=0
self.stop_h2=False
self.stop_div=True
self.quote=""
self.quotes=[]
self.div_hist=0
self.in_div=0
def handle_starttag(self, tag, attrs):
if tag=="ul":
self.in_ul+=1
if tag=="li":
self.in_li+=1
if tag=="h2":
self.h2_num+=1
self.in_h2+=1
self.stop_h2=False
if tag=="div":
self.in_div+=1
for at in attrs:
if at[0]=="id" and at[1]=="mw-content-text":
self.div_hist=self.in_div
self.stop_div=False
def handle_endtag(self, tag):
if tag=="div":
if self.div_hist==self.in_div:
self.div_hist=0
self.stop_div=True
self.in_div-=1
if tag=="h2":
self.in_h2-=1
if tag=="ul":
self.in_ul-=1
if tag=="li":
self.in_li-=1
if self.in_li==0:
if self.quote!="":
self.quotes.append(self.quote)
self.quote=""
def handle_data(self, data):
if self.in_h2>0:
for i in self.FORBIDDEN_H2:
if data.startswith(i):
self.stop_h2=True
if self.stop_div or self.stop_h2:
return
if self.in_li==1 and self.in_ul==1 and self.h2_num>=1:
self.quote+=data.replace("\n","")
def search_wiki(code,term):
results = []
url = "http://"+code[0:2]+".wikiquote.org/w/api.php?action=opensearch&search="+urllib2.quote(term)
req = urllib2.Request(url, headers={'User-Agent' : "Magic Browser"})
con = urllib2.urlopen( req )
data=con.read()
data=data[data[1:].index("[")+2:data.rindex("]")-1]
for dat in data.split(","):
results.append( dat[1:-1].decode('string_escape') )
return results
def quote(msg,args):
if len(args)!=2 or len(args)!=3:
brathuhn.sendMsg(brathuhn.ROOM,"Usage: "+args[0]+" language (two letter code) [topic]",'groupchat')
else:
try:
if len(args)==2:
brathuhn.sendMsg(brathuhn.ROOM,'fetching random page...','groupchat')
url = "http://"+args[1][0:2]+".wikiquote.org/wiki/Special:Random"
else:
brathuhn.sendMsg(brathuhn.ROOM,'searching for "'+ args[2] +'"...','groupchat')
search_results = search_wiki(args[1],args[2])
if len( search_results )==0:
brathuhn.sendMsg(brathuhn.ROOM,"No quotes found",'groupchat')
url = "http://"+args[1][0:2]+".wikiquote.org/wiki/" + urllib2.quote(search_results[0])
req = urllib2.Request(url, headers={'User-Agent' : "Magic Browser"})
con = urllib2.urlopen( req )
data=con.read()
print "data:", args[1][0:2], args[2],"->",search_results[0],url
wqp=WikiQuoteParser()
wqp.feed(data)
#for i in wqp.quotes:
# print "-- ",i
if len(wqp.quotes)>=1:
brathuhn.sendMsg(brathuhn.ROOM,'Quoting "'+ search_results[0] +'"','groupchat')
brathuhn.sendMsg(brathuhn.ROOM,wqp.quotes[randint(0,len(wqp.quotes)-1)],'groupchat')
else:
brathuhn.sendMsg(brathuhn.ROOM,'Could not crawl quotes from > '+url+' <','groupchat')
con.close()
except exceptions.Exception as e:
brathuhn.sendMsg(brathuhn.ROOM,"Network Error "+unicode(e),'groupchat')
brathuhn.addCommand("quote","Usage: !quote language (two letter code) [topic]",quote)