Here's the modified code to scrape all the images from the provided article URL:

if RUNNING_MODE == 1:
    driver2 = webdriver.Chrome()
    print(f'开始抓取页面所有文章,请稍候..')
    article_urls = driver.find_elements(By.CSS_SELECTOR, 'a')
    article_urls = [article_url.get_attribute('href') for article_url in article_urls if '_blank' in article_url.get_attribute('target')]
    article_urls = [article_url for article_url in article_urls if 'https://www.bilibili.com/read/' in article_url]
    print(f'本次共扫描到 {len(article_urls)} 篇文章!')

    total_url_count = len(article_urls)
    current_url_count = 0
    for article_url in article_urls:
        current_url_count += 1
        total_url_count = len(article_urls)
        print(f'正在加载第 {current_url_count}/{total_url_count} 篇文章内容...')

        # 进入文章页面
        driver2.get(article_url)
        time.sleep(1)
        soup = BeautifulSoup(driver2.page_source, 'html.parser')
        image_elements = soup.select('.card-image__image')
        image_urls = [img['style'].split('url(''')[1].split('');')[0] for img in image_elements]
        print(f'本篇文章共找到 {len(image_urls)} 张图片!')

        # Download the images
        for i, image_url in enumerate(image_urls):
            response = requests.get(image_url)
            with open(f'image_{current_url_count}_{i+1}.jpg', 'wb') as f:
                f.write(response.content)
            print(f'Successfully downloaded image {i+1}/{len(image_urls)}')

Please note that you need to have the necessary libraries (requests, selenium, and beautifulsoup4) installed in order to run this code successfully. Additionally, make sure to replace the CHROME_DRIVER_PATH with the path to your Chrome driver executable.


原文地址: https://www.cveoy.top/t/topic/hM7k 著作权归作者所有。请勿转载和采集!

免费AI点我,无需注册和登录