Итак, я делаю курс на автоматизацию скучных вещей и пытаюсь отсканировать цены на амазонку для автоматизации книги скучных вещей, но она возвращает пустую строку, независимо от того, что и в результате возникает ошибка индекса в элементах [0 ] .text.strip () и я не знаю, что делать
def getAmazonPrice(productUrl): headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:69.0) Gecko/20100101 Firefox/69.0'} # to make the server think its a web browser and not a bot res = requests.get(productUrl, headers=headers) res.raise_for_status() soup = bs4.BeautifulSoup(res.text, 'html.parser') elems = soup.select('#mediaNoAccordion > div.a-row > div.a-column.a-span4.a-text-right.a-span-last') return elems[0].text.strip() price = getAmazonPrice('https://www.amazon.com/Automate-Boring-Stuff-Python-2nd-ebook/dp/B07VSXS4NK/ref=sr_1_1?crid=30NW5VCV06ZMP&dchild=1&keywords=automate+the+boring+stuff+with+python&qid=1586810720&sprefix=automate+the+bo%2Caps%2C288&sr=8-1') print('The price is ' + price)
Вам нужно изменить парсер на lxml и использовать headers = {'user-agent': 'Mozilla/5.0'}
lxml
headers = {'user-agent': 'Mozilla/5.0'}
def getAmazonPrice(productUrl): headers = {'user-agent': 'Mozilla/5.0'} # to make the server think its a web browser and not a bot res = requests.get(productUrl, headers=headers) res.raise_for_status() soup = bs4.BeautifulSoup(res.text, 'lxml') elems = soup.select_one('#mediaNoAccordion > div.a-row > div.a-column.a-span4.a-text-right.a-span-last') return elems.text.strip() price = getAmazonPrice('https://www.amazon.com/Automate-Boring-Stuff-Python-2nd-ebook/dp/B07VSXS4NK/ref=sr_1_1?crid=30NW5VCV06ZMP&dchild=1&keywords=automate+the+boring+stuff+with+python&qid=1586810720&sprefix=automate+the+bo%2Caps%2C288&sr=8-1') print('The price is ' + price)
Снимок :
Если вы хотите использовать выберите, то
def getAmazonPrice(productUrl): headers = {'user-agent': 'Mozilla/5.0'} # to make the server think its a web browser and not a bot res = requests.get(productUrl, headers=headers) res.raise_for_status() soup = bs4.BeautifulSoup(res.text, 'lxml') elems = soup.select('#mediaNoAccordion > div.a-row > div.a-column.a-span4.a-text-right.a-span-last') return elems[0].text.strip() price = getAmazonPrice('https://www.amazon.com/Automate-Boring-Stuff-Python-2nd-ebook/dp/B07VSXS4NK/ref=sr_1_1?crid=30NW5VCV06ZMP&dchild=1&keywords=automate+the+boring+stuff+with+python&qid=1586810720&sprefix=automate+the+bo%2Caps%2C288&sr=8-1') print('The price is ' + price)
Попробуйте с этим.
def getAmazonPrice(productUrl): headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:69.0) Gecko/20100101 Firefox/69.0'} # to make the server think its a web browser and not a bot res = requests.get(productUrl, headers=headers) res.raise_for_status() soup = bs4.BeautifulSoup(res.text, 'lxml') elems = soup.select('#mediaNoAccordion > div.a-row > div.a-column.a-span4.a-text-right.a-span-last') return elems[0].text.strip() price = getAmazonPrice('https://www.amazon.com/Automate-Boring-Stuff-Python-2nd-ebook/dp/B07VSXS4NK/ref=sr_1_1?crid=30NW5VCV06ZMP&dchild=1&keywords=automate+the+boring+stuff+with+python&qid=1586810720&sprefix=automate+the+bo%2Caps%2C288&sr=8-1') print('The price is ' + price)
Ваш запрос вызовет ошибку 503 от Amazon. Возможно из-за антискребущих усилий Amazon. Поэтому, возможно, вам следует рассмотреть некоторые другие способы.
import requests headers = {'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:69.0) Gecko/20100101 Firefox/69.0'} # to make the server think its a web browser and not a bot productUrl = 'https://www.amazon.com/Automate-Boring-Stuff-Python-2nd-ebook/dp/B07VSXS4NK/ref=sr_1_1?crid=30NW5VCV06ZMP&dchild=1&keywords=automate+the+boring+stuff+with+python&qid=1586810720&sprefix=automate+the+bo%2Caps%2C288&sr=8-1' res = requests.get(productUrl, headers=headers) print (res)
output:
<Response [503]>