使用 Python 抓取《穿靴子的猫2》豆瓣电影评论数据

本教程将演示如何使用 Python 和 Selenium 库,抓取《穿靴子的猫2》在豆瓣电影上的所有页面的影评数据。

步骤 1:使用 Selenium 打开全部影评页面

from selenium import webdriver
import time

driver = webdriver.Chrome()
driver.get('https://movie.douban.com/subject/25868125/')
time.sleep(2)

# 点击'全部影评'按钮
more_btn = driver.find_element_by_xpath('//a[@class="more"]')
more_btn.click()
time.sleep(2)

步骤 2:抓取第一页评论数据

import requests
from bs4 import BeautifulSoup
import json

url = 'https://movie.douban.com/subject/25868125/comments?start=0&limit=20&status=P&sort=new_score'
headers = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.3',
    'Referer': 'https://movie.douban.com/subject/25868125/comments?status=P'
}
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, 'html.parser')
comment_list = soup.find_all('div', class_='comment-item')

comments = []
for comment in comment_list:
    author = comment.find('span', class_='comment-info').find('a').text
    time = comment.find('span', class_='comment-time')['title']
    content = comment.find('span', class_='short').text.strip()
    comments.append({
        'author': author,
        'time': time,
        'content': content
    })

# 存储为json文件
with open('comments.json', 'w', encoding='utf-8') as f:
    json.dump(comments, f, ensure_ascii=False)

步骤 3:抓取后续页面评论数据

for i in range(1, 4):
    url = 'https://movie.douban.com/subject/25868125/comments?start={}&limit=20&status=P&sort=new_score'.format((i-1)*20)
    response = requests.get(url, headers=headers)
    soup = BeautifulSoup(response.text, 'html.parser')
    comment_list = soup.find_all('div', class_='comment-item')

    for comment in comment_list:
        author = comment.find('span', class_='comment-info').find('a').text
        time = comment.find('span', class_='comment-time')['title']
        content = comment.find('span', class_='short').text.strip()
        comments.append({
            'author': author,
            'time': time,
            'content': content
        })

# 存储为json文件
with open('comments.json', 'w', encoding='utf-8') as f:
    json.dump(comments, f, ensure_ascii=False)

完整代码

from selenium import webdriver
import time
import requests
from bs4 import BeautifulSoup
import json

driver = webdriver.Chrome()
driver.get('https://movie.douban.com/subject/25868125/')
time.sleep(2)

# 点击'全部影评'按钮
more_btn = driver.find_element_by_xpath('//a[@class="more"]')
more_btn.click()
time.sleep(2)

url = 'https://movie.douban.com/subject/25868125/comments?start=0&limit=20&status=P&sort=new_score'
headers = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.3',
    'Referer': 'https://movie.douban.com/subject/25868125/comments?status=P'
}
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, 'html.parser')
comment_list = soup.find_all('div', class_='comment-item')

comments = []
for comment in comment_list:
    author = comment.find('span', class_='comment-info').find('a').text
    time = comment.find('span', class_='comment-time')['title']
    content = comment.find('span', class_='short').text.strip()
    comments.append({
        'author': author,
        'time': time,
        'content': content
    })

for i in range(1, 4):
    url = 'https://movie.douban.com/subject/25868125/comments?start={}&limit=20&status=P&sort=new_score'.format((i-1)*20)
    response = requests.get(url, headers=headers)
    soup = BeautifulSoup(response.text, 'html.parser')
    comment_list = soup.find_all('div', class_='comment-item')

    for comment in comment_list:
        author = comment.find('span', class_='comment-info').find('a').text
        time = comment.find('span', class_='comment-time')['title']
        content = comment.find('span', class_='short').text.strip()
        comments.append({
            'author': author,
            'time': time,
            'content': content
        })

# 存储为json文件
with open('comments.json', 'w', encoding='utf-8') as f:
    json.dump(comments, f, ensure_ascii=False)

driver.quit()

注意:

  • 以上代码示例仅抓取了前三页的评论数据,您可以根据实际情况修改代码抓取更多页面的数据。
  • 由于豆瓣网站的反爬机制,可能需要调整代码中的 User-Agent 和 Referer 信息,以避免被封禁。
  • 在抓取数据时,请尊重网站的 robots.txt 协议,避免对网站造成过大的压力。

希望本教程对您有所帮助!

使用 Python 抓取《穿靴子的猫2》豆瓣电影评论数据

原文地址: https://www.cveoy.top/t/topic/oBmF 著作权归作者所有。请勿转载和采集!

免费AI点我,无需注册和登录