您从子页面获取 HTML,但未转换为 soup,因此您在主页上搜索
response = requests.get(gymurl)
sub_soup = BeautifulSoup(response.text)
我也遇到了 CSS 选择器的问题
address_line = sub_soup.select('p.small.m-b-sm.p-t-1 span.btn-icon-text')
有些页面在这个地方没有元素,它会引发错误,所以我使用try/except 来捕捉它。
在 Python 3 上测试,因为在 Python 2 上 .select() 对我不起作用
import requests
from bs4 import BeautifulSoup
import urllib.parse
import csv
import time
initial_url = "https://www.lifetime.life"
response = requests.get("https://www.lifetime.life/view-all-locations.html")
soup = BeautifulSoup(response.text)
with open('gyms2.csv', 'w') as gf:
gymwriter = csv.writer(gf)
for a in soup.findAll('a'):
if '/life-time-locations/' in a['href']:
gymurl = urllib.parse.urljoin(initial_url, a.get('href'))
print(gymurl)
response = requests.get(gymurl)
sub_soup = BeautifulSoup(response.text)
try:
address_line = sub_soup.select('p.small.m-b-sm.p-t-1 span.btn-icon-text')
gymrow = [gymurl, address_line[0].text.strip()]
print(gymrow)
gymwriter.writerow(gymrow)
time.sleep(3)
except Exception as ex:
print(ex)
编辑: Python 2 使用 find() 而不是 select()
import requests
import BeautifulSoup
import csv
import urllib2
import time
initial_url = "https://www.lifetime.life"
response = requests.get("https://www.lifetime.life/view-all-locations.html")
soup = BeautifulSoup.BeautifulSoup(response.text)
with open('gyms2.csv', 'w') as gf:
gymwriter = csv.writer(gf)
for a in soup.findAll('a'):
if '/life-time-locations/' in a['href']:
gymurl = urllib2.urlparse.urljoin(initial_url, a.get('href'))
print(gymurl)
response = requests.get(gymurl)
sub_soup = BeautifulSoup.BeautifulSoup(response.text)
try:
address_line = sub_soup.find('p', {'class': 'small m-b-sm p-t-1'}).find('span', {'class': 'btn-icon-text'})
gymrow = [gymurl, address_line.text]
print(gymrow)
gymwriter.writerow(gymrow)
time.sleep(3)
except Exception as ex:
print(ex)
编辑: 页面似乎有很多版本。每个页面可能需要分隔try/except。但是,如果第一个 try 工作正常,我会使用 continue 跳过下一个 try/except,而不是将第二个 try/except 放在第一个 except 中。
import requests
from bs4 import BeautifulSoup
import urllib.parse
import csv
import time
initial_url = "https://www.lifetime.life"
response = requests.get("https://www.lifetime.life/view-all-locations.html")
soup = BeautifulSoup(response.text)
with open('gyms2.csv', 'w') as gf:
gymwriter = csv.writer(gf)
for a in soup.findAll('a'):
if '/life-time-locations/' in a['href']:
gymurl = urllib.parse.urljoin(initial_url, a.get('href'))
print(gymurl)
response = requests.get(gymurl)
sub_soup = BeautifulSoup(response.text)
try:
address_line = sub_soup.select('p.small.m-b-sm.p-t-1 span.btn-icon-text')
gymrow = [gymurl, address_line[0].text.strip()]
print('type 1:', gymrow)
gymwriter.writerow(gymrow)
time.sleep(3)
continue # go back to `for`
except Exception as ex:
print('ex:', ex)
try:
address_line = sub_soup.find('div', {'class': 'btn-resp-md'}).find('p')
gymrow = [gymurl, address_line.text.strip()]
print('type 2:', gymrow)
gymwriter.writerow(gymrow)
time.sleep(3)
continue # go back to `for`
except Exception as ex:
print('ex:', ex)
try:
address_line = sub_soup.find('p', {'class': 'm-b-grid'})
gymrow = [gymurl, address_line.text.strip()]
print('type 3:', gymrow)
gymwriter.writerow(gymrow)
time.sleep(3)
continue # go back to `for`
except Exception as ex:
print('ex:', ex)