import requestsimport re#获取网站源码url="https://www.dytt8899.com"headers={"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " "AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.1 " "Safari/605.1.15"}response = requests.get(url,headers=headers)response.encoding='gbk's=response.text#提取2026必看电源的html部分,用searchobj1=re.compile(r'2026必看热片.*?<ul>(?P<html>.*?)</ul>',flags=re.S)html=obj1.search(s).group('html')# print(html)#第一次提取a标签中的href的值obj2=re.compile(r"<li><a href='(?P<href>.*?)' title")#第二次提取自页面的片名和下载地址obj3=re.compile(r'片 名 (?P<name>.*?)<br />.*?<td style="WORD-WRAP: ' r'break-word" bgcolor="#fdfddf"><a href="(?P<download>.*?)>',flags=re.S)result2=obj2.finditer(html)for item in result2:# print(item.group('href'))child_url=url+item.group('href') child_response=requests.get(child_url,headers=headers) child_response.encoding='gbk'# print(child_response.text)result3 = obj3.search(child_response.text) name=result3.group('name') download=result3.group('download')print(name,download)