from sgmllib import SGMLParser
import urllib2
class sgm(SGMLParser):
def reset(self):
SGMLParser.reset(self)
self.srcs=[]
self.ISTRUE=True
def start_div(self,artts):
for k,v in artts:
if v=="author":
self.ISTRUE=False
def end_div(self):
self.ISTRUE=True
def start_img(self,artts):
for k,v in artts:
if k=="src" and self.ISTRUE==True:
self.srcs.append(v)
def download(self):
i=1
for src in self.srcs:
f=open("%d.jpg"%i,"wb")
print src
img=urllib2.urlopen(src)
f.write(img.read())
f.close()
i+=1
sgm=sgm()
for page in range(300,500):
url="http://www.qiushibaike.com/late/page/%s?s=4622726" % page
data=urllib2.urlopen(url).read()
sgm.feed(data)
sgm.download()
原文地址 http://www.bcwhy.com/thread-21111-1-1.html