看别人用异步请求快的飞起, 忍不住手痒尝试了下, 但过程并不美好, 一不小心就是满屏的"红色警告" 不过成功的喜悦总是让人陶醉的, 这也是学习的魅力吧
先看下运行过程
需要插入一个视频, 暂时不知道视频怎么上传 好像没啥问题, 1秒十章的下载速度, 挺感人了. 异步运行任务每次只添加20个, 最终耗时相比同步运行快了10多倍
再看下运行结果,
有些章节标题漏掉了, 前几章内容变成了最新章节的内容, 应该是正则匹配出了的偏差, 继续往下看 第一章怎么跑到这么靠后的地方了, 原因不明, 继续往下看 第二章是空的, 继续往下检查了一些章节, 好像没啥问题了. 总体上就是第一章前面多了一些章节, 部分章节为空, 其余部分正常. 应该是前面这几个玩意对应的html结构不同导致的 所有章节内容都保存在“小说名.txt”中了, 进行适当的修改, 处理结果保存在"modify.txt"中 快速浏览了整个文件, 章节内容没啥问题, 顺序也对的上
上代码:
import requests
import asyncio
import aiohttp
import json
import re
import time
import os
import sys
from bs4
import BeautifulSoup
from docx
import Document
from docx
.shared
import Cm
headers
= {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/77.0.3865.120 Safari/537.36'}
url
= input('please input url:')
if len(url
) < 24:
url
= 'https://www.xbiquge.cc/book/14779/'
rootPath
= r
'C:\Users\QQ\Desktop\ls\py\{}'
url
= url
.replace
(' ','')
def getCatalog():
def saveCatalog():
rep
= requests
.get
(url
, headers
=headers
)
print(rep
.text
[:10])
rep
.encoding
= 'gbk'
soup
= BeautifulSoup
(rep
.text
, 'lxml')
title
= soup
.title
.contents
[0]
print(title
)
global name
name
= (re
.findall
('(.+?) ', title
))[0] + ' ' + (re
.findall
('_(.+?)_', title
))[0]
print(name
)
mkDir
(path
=rootPath
.format(name
))
f1
= r
'C:\Users\QQ\Desktop\ls\py\{}\{}.txt'.format(name
, '目录')
with open(f1
, 'w') as f
:
f
.write
(rep
.text
)
saveCatalog
()
def findAllChapter():
f1
= r
'C:\Users\QQ\Desktop\ls\py\{}\{}.txt'.format(name
, '目录')
f2
= r
'C:\Users\QQ\Desktop\ls\py\{}\{}.txt'.format(name
, '章节链接')
with open(f1
, 'r') as f
:
rep
= f
.read
()
soup
= BeautifulSoup
(rep
, 'lxml')
s
= str(soup
.find
(id='list'))
soup
= BeautifulSoup
(s
, 'lxml')
ss
= soup
.findAll
('a')[:]
print(ss
[:10], ss
[-10:])
global cul
, cnl
cul
= re
.findall
(r
'\d{6,8}.html', str(s
))
cnl
= re
.findall
(r
'.html.>(第?\d{0,4} ?章? ?.*?)</a', str(s
))
print(len(ss
), len(cul
), len(cnl
))
print(cul
[:10], cul
[-10:], cnl
[:10], cnl
[-10:])
print('len(cul):', len(cul
), 'len(cnl):', len(cnl
))
'''for i in range(0, len(ss)):
# 检查正则表达式,检查完后需注释掉
c = str(ss[i])
cu = re.search(r'\d{7,8}.html', str(c)).group()
cn = c[c.index('.html') + 7:-4]
if cu != cul[i] or cn != cnl[i]:
print(cu, cul[i], cu == cul[i], cn, cnl[i], cn == cnl[i])
break'''
if len(cul
) == len(cnl
):
with open(f2
, 'w') as f
:
for u
, n
in zip(cul
, cnl
):
f
.write
(u
+ n
+ '\n')
print('All url and name of chapters from source have been saved in this file:{}'.format(f2
))
else:
print('Rules require changes the regular expression')
findAllChapter
()
def mkDir(path
):
if not os
.path
.exists
(path
):
os
.makedirs
(path
)
def missingChapter():
new
= int(re
.search
(r
'\d{1,4}', cnl
[-1]).group
())
nl
= [0]
ml
= []
for i
in range(len(cnl
)):
nl
.append
(int(re
.search
(r
'\d{1,4}', cnl
[i
]).group
()))
d
= nl
[i
] - nl
[i
- 1] - 1
while d
> 0:
ml
.append
(nl
[i
] - d
)
d
-= 1
return nl
'''
for i in ml:
if str(i) in str(cnl):
print(i,True)
else:
print(i,False)
'''
def saveChapter():
f3
= r
'C:\Users\QQ\Desktop\ls\py\{}\{}.txt'.format(name
, name
)
lencnl
= len(cnl
)
nlc
= modify
()
with open(f3
, 'a') as f
:
async def getreptext(session
, cu
, cn
, serialcontent
):
async with session
.get
(url
+ cu
, headers
=headers
) as rep
:
rep
= await rep
.text
(encoding
= 'gbk')
serialcontent
.update
(await cache
(rep
, cu
, cn
))
return serialcontent
async def cache(rep
, cu
, cn
, nlc
=nlc
):
start
= time
.perf_counter
()
content
= ''
for s
in rep
.splitlines
():
if 'span' in s
:
continue
test1
= re
.findall
(r
' (.+)<', s
)
if test1
:
content
+= test1
[0] + '\n'
if len(content
) > 1200:
content
+= '\n'
print('contents has been writen to cache which from : {} {}'.format(cu
, cn
))
else:
print(content
)
content
= '\n'
print('There are problems in this chapter : {} {} !!!'.format(cu
, cn
))
end
= time
.perf_counter
()
rt
= end
- start
trt
= rt
* (lencnl
- nlc
)
print('estimate rest of runtime : {} minutes {} seconds'.format(trt
// 60, trt
%60))
nlc
+= 1
return {cu
: content
}
pervolume
= 20
size
= lencnl
- nlc
tail
= size
% pervolume
iterations
= size
// pervolume
async def getandcache(cu
, cn
, serialcontent
):
async with aiohttp
.ClientSession
() as session
:
await getreptext
(session
, cu
, cn
, serialcontent
)
for i
in range(iterations
):
serialcontent
= {}
tasks
= [asyncio
.ensure_future
(getandcache
(cu
, cn
, serialcontent
)) for cu
, cn
in zip(cul
[nlc
:nlc
+pervolume
], cnl
[nlc
:nlc
+pervolume
])]
loop
= asyncio
.get_event_loop
()
loop
.run_until_complete
(asyncio
.wait
(tasks
))
for i
in cul
[nlc
: nlc
+pervolume
]:
f
.write
(serialcontent
[i
])
nlc
= nlc
+ pervolume
del serialcontent
pervolume
= tail
if pervolume
:
serialcontent
= {}
tasks
= [asyncio
.ensure_future
(getandcache
(cu
, cn
, serialcontent
)) for cu
, cn
in zip(cul
[nlc
:nlc
+ pervolume
], cnl
[nlc
:nlc
+ pervolume
])]
loop
= asyncio
.get_event_loop
()
loop
.run_until_complete
(asyncio
.wait
(tasks
))
for i
in cul
[nlc
: nlc
+pervolume
]:
f
.write
(serialcontent
[i
])
del serialcontent
def runlog():
pass
def modify():
f3
= r
'C:\Users\QQ\Desktop\ls\py\{}\{}.txt'.format(name
, name
)
f4
= r
'C:\Users\QQ\Desktop\ls\py\{}\{}.txt'.format(name
, 'modify')
if not os
.path
.exists
(f3
):
with open(f3
, 'w') as f
:
pass
print('saved such file : {}'.format(f3
))
else:
print('no such file : {}'.format(f3
))
with open(f3
, 'r') as f
, open(f4
, 'w') as fs
:
cc
(f
)
c
= 0
li
= f
.readlines
()
for n
, i
in enumerate(li
):
if 'span' in i
:
continue
fs
.write
(i
)
if i
== '\n' and n
< len(li
) - 1:
c
+= 1
if '第' not in li
[n
+ 1] and '章' not in li
[n
+ 1]:
fs
.write
(cnl
[c
] + '\n')
pass
return c
def cc(file):
f00
= r
'C:\Users\QQ\Desktop\ls\py\{}\{}.txt'.format(name
, 'other characters')
hs0
= {
3: '·、【】!¥—~……();‘’:“”《》,。?、',
4: ''' `~!@#$%^&*()_+-={}|:%"<>?[]\;',./×'''
}
hs
= {
1: 0,
2: 0,
3: 0,
4: 0,
5: 0,
6: 0,
7: 0,
}
string
= file.read
()
with open(f00
, 'w') as f
:
for i
in string
:
if 19968 <= ord(i
) <= 40869:
hs
[1] += 1
elif 65 <= ord(i
) <= 90 or 97 <= ord(i
) <= 122:
hs
[2] += 1
elif i
in hs0
[3]:
hs
[3] += 1
elif i
in hs0
[4]:
hs
[4] += 1
elif 48 <= ord(i
) <= 57:
hs
[5] += 1
elif i
== '\n':
hs
[6] += 1
else:
f
.write
(i
)
length
= len(string
)
hs
[7] = hs
[1] / (length
+ 1)
file.seek
(0)
l
= ['中文', 'english letter', '中文标点符号', 'english punctuation marks', '数字', '行数', '中文字数占总字符数的比例']
for i
in range(7):
if i
== 6:
print('{} : {:.2%}'.format(l
[i
], hs
[i
+ 1]))
else:
print('{} : {:.2f}万'.format(l
[i
], hs
[i
+ 1] / 10000))
print('\n总字符数:{:.0f}万.平均每章节{:.0f}字,平均每个段落{:.0f}字\n'.format(length
/ 10000, length
/ (len(cnl
) + 1), length
/ (hs
[6] + 1)))
def main():
start
= time
.perf_counter
()
getCatalog
()
saveChapter
()
modify
()
end
= time
.perf_counter
()
print('total time consuming : ', (end
- start
) // 60, 'minutes', (end
- start
) % 60, 'seconds')
main
()
几个需要注意的地方
因为异步运行有随机性, 所以不能将getreptext获取到的文本直接写入"小说名.txt"文件, 否则章节顺序就乱了. 解决方法有很多, 比如
把每个章节分别保存到本地, 全保存下来后, 按顺序读取每个文件并写入"小说名.txt"文件将章节暂时储存在字典中, 本文定义了一个字典"serialcontent", 每次将pervolume个章节的文本内容更新到serialcontent中, 按章节顺序取出serialcontent的value, 将value写入"小说名.txt"其实就是对每个getreptext得到的content做一个标记, 有了标记后面的事情就好办了
异步请求可以一次性访问所有章节链接, 但这样做毫无疑问的是网站会把你封掉
所以需要对访问任务进行分组, 每次并行pervolume个任务, 建议每次20个章实在有刚需, 可以尝试伪装IP. 伪装IP大概分三类:服务器知道你伪装了IP, 而且知道你真实的IP服务器知道你伪装了IP, 不知道你真实的IP服务器难以辨别你是否伪装了IP