使用 Python 抓取《穿靴子的猫2》豆瓣电影评论数据
使用 Python 抓取《穿靴子的猫2》豆瓣电影评论数据
本教程将演示如何使用 Python 和 Selenium 库,抓取《穿靴子的猫2》在豆瓣电影上的所有页面的影评数据。
步骤 1:使用 Selenium 打开全部影评页面
from selenium import webdriver
import time
driver = webdriver.Chrome()
driver.get('https://movie.douban.com/subject/25868125/')
time.sleep(2)
# 点击'全部影评'按钮
more_btn = driver.find_element_by_xpath('//a[@class="more"]')
more_btn.click()
time.sleep(2)
步骤 2:抓取第一页评论数据
import requests
from bs4 import BeautifulSoup
import json
url = 'https://movie.douban.com/subject/25868125/comments?start=0&limit=20&status=P&sort=new_score'
headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.3',
'Referer': 'https://movie.douban.com/subject/25868125/comments?status=P'
}
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, 'html.parser')
comment_list = soup.find_all('div', class_='comment-item')
comments = []
for comment in comment_list:
author = comment.find('span', class_='comment-info').find('a').text
time = comment.find('span', class_='comment-time')['title']
content = comment.find('span', class_='short').text.strip()
comments.append({
'author': author,
'time': time,
'content': content
})
# 存储为json文件
with open('comments.json', 'w', encoding='utf-8') as f:
json.dump(comments, f, ensure_ascii=False)
步骤 3:抓取后续页面评论数据
for i in range(1, 4):
url = 'https://movie.douban.com/subject/25868125/comments?start={}&limit=20&status=P&sort=new_score'.format((i-1)*20)
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, 'html.parser')
comment_list = soup.find_all('div', class_='comment-item')
for comment in comment_list:
author = comment.find('span', class_='comment-info').find('a').text
time = comment.find('span', class_='comment-time')['title']
content = comment.find('span', class_='short').text.strip()
comments.append({
'author': author,
'time': time,
'content': content
})
# 存储为json文件
with open('comments.json', 'w', encoding='utf-8') as f:
json.dump(comments, f, ensure_ascii=False)
完整代码
from selenium import webdriver
import time
import requests
from bs4 import BeautifulSoup
import json
driver = webdriver.Chrome()
driver.get('https://movie.douban.com/subject/25868125/')
time.sleep(2)
# 点击'全部影评'按钮
more_btn = driver.find_element_by_xpath('//a[@class="more"]')
more_btn.click()
time.sleep(2)
url = 'https://movie.douban.com/subject/25868125/comments?start=0&limit=20&status=P&sort=new_score'
headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.3',
'Referer': 'https://movie.douban.com/subject/25868125/comments?status=P'
}
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, 'html.parser')
comment_list = soup.find_all('div', class_='comment-item')
comments = []
for comment in comment_list:
author = comment.find('span', class_='comment-info').find('a').text
time = comment.find('span', class_='comment-time')['title']
content = comment.find('span', class_='short').text.strip()
comments.append({
'author': author,
'time': time,
'content': content
})
for i in range(1, 4):
url = 'https://movie.douban.com/subject/25868125/comments?start={}&limit=20&status=P&sort=new_score'.format((i-1)*20)
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, 'html.parser')
comment_list = soup.find_all('div', class_='comment-item')
for comment in comment_list:
author = comment.find('span', class_='comment-info').find('a').text
time = comment.find('span', class_='comment-time')['title']
content = comment.find('span', class_='short').text.strip()
comments.append({
'author': author,
'time': time,
'content': content
})
# 存储为json文件
with open('comments.json', 'w', encoding='utf-8') as f:
json.dump(comments, f, ensure_ascii=False)
driver.quit()
注意:
- 以上代码示例仅抓取了前三页的评论数据,您可以根据实际情况修改代码抓取更多页面的数据。
- 由于豆瓣网站的反爬机制,可能需要调整代码中的 User-Agent 和 Referer 信息,以避免被封禁。
- 在抓取数据时,请尊重网站的 robots.txt 协议,避免对网站造成过大的压力。
希望本教程对您有所帮助!
原文地址: https://www.cveoy.top/t/topic/oBmF 著作权归作者所有。请勿转载和采集!