欢迎您访问程序员文章站本站旨在为大家提供分享程序员计算机编程知识!
您现在的位置是: 首页  >  IT编程

pyhton 网络爬取软考题库保持txt

程序员文章站 2023-09-01 18:13:27
#-*-coding:utf-8-*-#参考文档#https://www.crummy.com/software/BeautifulSoup/bs4/doc/index.zh.html#find-all #https://m.cnitpm.com import requestsimport refr ......
#-*-coding:utf-8-*-
#参考文档
#https://www.crummy.com/software/beautifulsoup/bs4/doc/index.zh.html#find-all
#https://m.cnitpm.com

import requests
import re
from bs4 import beautifulsoup
html = requests.get('https://m.cnitpm.com/exam/examst1_1031655.htm/')
soup = beautifulsoup(html.text,'lxml')
ultag=soup.find_all('ul','tit')
for item in ultag:
a_temp=item.find_all('a')
#print(a_temp)
for aitem in a_temp:
#print (aitem.get('href'))
html2 = requests.get(aitem.get('href'))
#解决乱码问题
html2.encoding = 'utf-8'
soup2 = beautifulsoup(html2.text, 'lxml')
divtag = soup2.find_all('div', 'tm-box')
for divitem in divtag:
print(divitem.get_text())
#print(divtag.replace('[<div class="tm-box">', ''))
################################以上为爬取############################################