写出增加爬取书籍评论数的代码import requests from bs4 import BeautifulSoup from threading import Thread def crawl_books(start, end): for i in range(start, end): url = 'http://search.dangdang.com/?key=%BC%C6%CB%E3%BB%FA&act=input&page_index={}'.format(i) response = requests.get(url) page = response.text soup = BeautifulSoup(page, 'lxml') books = soup.find('ul', class_='bigimg') for book in books.find_all('li'): title = book.find('a', class_='pic').get('title') author = book.find('p', class_='search_book_author').text price = book.find('p', class_='price').find('span', class_='search_now_price').text print(title, author, price) threads = [] for i in range(1, 101, 10): t = Thread(target=crawl_books, args=(i, i+10)) threads.append(t) t.start() for t in threads: t.join()
时间: 2024-02-29 10:55:17 浏览: 60
import reimport requestsfrom bs4 import BeautifulSoupimport t
上面的代码只是用来爬取当当网中与搜索关键词“心理学”相关的书籍的名称、作者和价格,而没有爬取书籍评论数。如果需要爬取书籍评论数,可以先获取每本书的详情页链接,然后再进入详情页爬取评论数。具体的代码可以参考下面的示例:
```
import requests
from bs4 import BeautifulSoup
from threading import Thread
def get_book_links(start, end):
links = []
for i in range(start, end):
url = 'http://search.dangdang.com/?key=%BC%C6%CB%E3%BB%FA&act=input&page_index={}'.format(i)
response = requests.get(url)
page = response.text
soup = BeautifulSoup(page, 'lxml')
books = soup.find('ul', class_='bigimg')
for book in books.find_all('li'):
link = book.find('a', class_='pic').get('href')
links.append(link)
return links
def get_comment_count(url):
response = requests.get(url)
page = response.text
soup = BeautifulSoup(page, 'lxml')
count = soup.find('a', class_='comm_num').text
return count
def crawl_books(start, end):
links = get_book_links(start, end)
for link in links:
count = get_comment_count(link)
print(link, count)
threads = []
for i in range(1, 101, 10):
t = Thread(target=crawl_books, args=(i, i+10))
threads.append(t)
t.start()
for t in threads:
t.join()
```
上面的代码中,我们新增了两个函数:`get_book_links` 和 `get_comment_count`。`get_book_links` 用来获取每本书的详情页链接,`get_comment_count` 用来获取详情页中的评论数。在 `crawl_books` 函数中,我们先获取每个搜索结果页中的书籍链接,然后再调用 `get_comment_count` 函数获取评论数,并打印出来。最后,我们通过多线程的方式同时爬取多个搜索结果页中的书籍评论数。
阅读全文