-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathcsdnToPdf.py
More file actions
122 lines (114 loc) · 3.24 KB
/
Copy pathcsdnToPdf.py
File metadata and controls
122 lines (114 loc) · 3.24 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
# -*- coding: utf-8 -*-
import urllib,urllib2,cookielib,re,socket
import os,sys,time
from bs4 import BeautifulSoup
#import BeautifulSoup
#防止编码乱码#
reload(sys)
sys.setdefaultencoding('utf-8')
####
#url='http://blog.csdn.net/Luoshengyang/'# csdn的账号 default
#blogName='Luoshengyang/' #default
#url='http://blog.csdn.net/jerryjbiao/article/category/870957/'
url='http://blog.csdn.net/xubin341719'
blogName='clk/'
blogDir='./csdn_blog/'
headers={
'User-Agent':'Mozilla/5.0 (Windows NT 6.1; WOW64; rv:30.0) Gecko/20100101 Firefox/30.0'
}
#读取htmml
def login(url=url):
#socket.setdefaulttim #单位为秒
time.sleep(0.5)# 防止封IP
req= urllib2.Request(url=url,headers=headers)
html = urllib2.urlopen(req).read()
return html
#return html.decode('GBK','ignore').encode('UTF-8')
StringPrefix=''
StringSurfix=' '
def fixSynaxHilghLighter(html):
soup = BeautifulSoup(html,from_encoding='utf-8')
userSoup = soup.find(name="div", attrs={"id":"body"})
classes=userSoup.findAll(name="pre")
try:
for cla in classes:
if(cla.get('class')==0):
continue
s = cla['class'][0]
tmp="brush: "+ s + ";"
cla['class'][0] = tmp
except KeyError ,e:
print e
str = userSoup.__str__()
dest = StringPrefix+str+StringSurfix
return dest
'''
if __name__ == '__main__':
artical_url='http://blog.csdn.net/tx3344/article/details/8476669'
html=login(artical_url)
fixSynaxHilghLighter(html)
'''
if __name__ == '__main__':
#def main():
state=True
pageNum=0
listNum=0
surfixFd= open('./Surfix.txt','r')
prefixFd = open('./prefix.txt','r')
StringPrefix = prefixFd.read()
StringSurfix = surfixFd.read()
surfixFd.close()
prefixFd.close()
html=login()
isExist = os.path.exists(blogDir+blogName)
if not isExist:
os.makedirs(blogDir+blogName)
os.system("cd "+blogDir+blogName+" && ln -s ../../scripts scripts"
+ " && ln -s ../../styles styles")
os.system("cd -")
while state:
soup=BeautifulSoup(html)
articals=soup.findAll(name='div',attrs={'class' : 'list_item article_item'})
for artical in articals:
listNum+=1
title=artical.find('a')
artical_url='http://blog.csdn.net/'+title['href']
print artical_url
artNum=artical_url.split('/')
artNum=artNum[-1]
print artNum
s=title.text.replace('\r\n',' ')#去掉回车符
s=s.lstrip()#去掉首空格
s=s.rstrip()#去掉尾空格
s=s.strip() #过滤字符串中所有的转义符
#s.s.find('/')
s=s.replace('/','or')
s=s.replace(' ','')
s=s.decode('UTF-8','ignore').encode('UTF-8');
s='P%02d_%02d_%s'%(pageNum,listNum,s)
print s
destHtml=blogDir+blogName+artNum+'.htm'
destPdf=blogDir+blogName+artNum+'.pdf'
realNamePdf=blogDir+blogName+s+'.pdf'
if os.path.isfile(realNamePdf):
continue;
f=file(destHtml, 'w') #保存的目录
f.write(fixSynaxHilghLighter(login(artical_url)))
f.close()
print destHtml
print destPdf
os.system('wkhtmltopdf '+'\"'+destHtml+'\"'+' '+'\"'+destPdf+'\"')
os.rename(destPdf,realNamePdf)
#print artical_url
##换页转换
pagelist= soup.find(name='div',id='papelist')
next=pagelist.findAll('a')
state=False
for i in next :
if i.text.encode('utf-8')==str('下一页') :
pageNum+=1
listNum=0;
url='http://blog.csdn.net/'+i['href']
html=login(url)
state=True
break;