优化这段代码 for p in range(1,1190): try: # print(browser.page_source) # 输出网页源码 time.sleep(1) html = etree.HTML(browser.page_source) content_html = html.xpath('//div[@class="table-responsive"]/table/tbody[@class="data-table"]/tr') for html_ in content_html: No_id = html_.xpath('.//td[2]/text()') url2_list = html_.xpath('.//td[3]/div/text()') url2_list = url2_list[0].replace('www.', '').rstrip('/') # url2_list=url2_list[0].replace('https://www.','').replace('http://www.','').replace('https://','').replace('http://','').rstrip('/') # language_list=html_.xpath('.//div[@class="table__summary table__summary--inventory-verified"]/div[12]/div/text()') category_list = html_.xpath('.//td[4]/div/div/div[1]/span/text()') category_list = '、'.join(category_list) # category_list='、'.join(category_list) # 提供者数量 performer_list = html_.xpath('.//td[10]/text()') performer_list = '$' + performer_list[0] last = [p,No_id[0], '', 'globexportal.net', url2_list, '', '', category_list, '', '平台', performer_list] print(last) with open('globexportal.net.1.csv', 'a+', newline='', encoding='utf_8_sig') as fp: writer = csv.writer(fp) writer.writerow(last) pyautogui.scroll(-40, x=960, y=720) except Exception as e: print(str(e))
时间: 2024-04-19 22:24:28 浏览: 162
根据您提供的代码片段,可以对其进行一些优化。首先,可以将文件写入操作放在循环之外,减少频繁的文件写入操作。另外,可以使用更简洁的方法来处理XPath选择器的结果。以下是优化后的代码:
```python
import csv
import time
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.common.action_chains import ActionChains
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.keys import Keys
from lxml import etree
# 创建一个Chrome浏览器实例
s = Service('path/to/chromedriver') # 将路径替换为你的chromedriver路径
chrome_options = Options()
chrome_options.add_argument('--headless') # 设置为无界面模式
browser = webdriver.Chrome(service=s, options=chrome_options)
# 打开登录页面,登录
for p in range(1, 1190):
try:
time.sleep(1)
html = etree.HTML(browser.page_source)
content_html = html.xpath('//div[@class="table-responsive"]/table/tbody[@class="data-table"]/tr')
result = []
for html_ in content_html:
No_id = html_.xpath('.//td[2]/text()')[0]
url2_list = html_.xpath('.//td[3]/div/text()')[0].replace('www.', '').rstrip('/')
category_list = '、'.join(html_.xpath('.//td[4]/div/div/div[1]/span/text()'))
performer_list = '$' + html_.xpath('.//td[10]/text()')[0]
last = [p, No_id, '', 'globexportal.net', url2_list, '', '', category_list, '', '平台', performer_list]
result.append(last)
print(last)
with open('globexportal.net.1.csv', 'a+', newline='', encoding='utf_8_sig') as fp:
writer = csv.writer(fp)
writer.writerows(result)
# 模拟滚动
actions = ActionChains(browser)
actions.move_by_offset(960, 720).perform()
actions.send_keys(Keys.PAGE_DOWN).perform()
actions.reset_actions()
except Exception as e:
print(str(e))
# 关闭浏览器
browser.quit()
```
在优化后的代码中,使用XPath选择器时,直接使用索引来获取结果列表中的第一个元素,避免了使用变量并进行多余的操作。另外,将需要写入CSV文件的数据存储在一个列表中,然后一次性写入文件,减少了频繁的文件写入操作。
请确保已经正确导入所需的模块,并根据您的实际情况进行适当的调整和修改。另外,需要注意根据您的实际需求进行Chrome浏览器实例的配置,如使用无界面模式等。
阅读全文