Link Extractor в Scrapy с process_value - PullRequest
1 голос
/ 26 мая 2020

Я пытаюсь извлечь данные с с помощью scrapy. Мой код до сих пор -

# -*- coding: utf-8 -*-
import scrapy
from scrapy.linkextractors import LinkExtractor
from scrapy.spiders import CrawlSpider, Rule

class VideoSpider(CrawlSpider):
    name = 'video'
    allowed_domains = ['']

    user_agent = 'Mozilla/5.0 (Windows NT 6.3; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36'

    # def __init__(self, url = ""):
    #         # self.input = input  # source file name
    #         self.url = url
    #         # self.last = last

    def start_requests(self):
        # yield scrapy.Request(url='', headers={
        #     'User-Agent': self.user_agent
        # })
        yield scrapy.Request(url=self.url, headers={
            'User-Agent': self.user_agent
        }, callback=self.parse)
    #    with open("./Input/amazon.csv") as f:
    #     for line in f:
    #         category, url = line.split(',')
    #         category = category.strip()
    #         url = url.strip()
    #         yield scrapy.Request(url=url, headers={
    #             'User-Agent': self.user_agent
    #         }, meta={'urlkey':category})

    rules = (
        Rule(LinkExtractor(restrict_xpaths="//li[@class='product-base']", process_value=lambda x :"" +x), callback='parse_item', follow=True, process_request='set_user_agent'), # have tried //li[@class='product-base']/a/@href and //li[@class='product-base']/a[1] as well for restricted_xpaths
        Rule(LinkExtractor(restrict_xpaths="//li[@class='pagination-next']/a"), process_request='set_user_agent')

    # def parse_start(self, response):
    #     print(response)
    #     all_links = response.xpath('//li[@class="product-base"]/a/@href').extract()
    #     print(all_links)
    #     for link in all_links:
    #         yield scrapy.Request(url=''+link, callback=self.parse_item)
        # return super().parse_start_url(response)
    # def parse_fail(self, response):
    #     print(response.url)
        # all_links = response.xpath('//li[@class="product-base"]/a/@href').extract()
        # print(all_links)
        # for link in all_links:
        #     yield scrapy.Request(url=''+link, callback=self.parse_item)

    def set_user_agent(self, request):
        request.headers['User-Agent'] = self.user_agent
        return request

    # def process_values(self,value):
    #     print(value)
    #     value = "" + value
    #     print(value)
    #     return value

    # def link_add(self, links):
    #     print(links)

    def parse_item(self, response):
        # yield {
        #     'title':response.xpath("normalize-space(//span[@class='a-size-large']/text())").get(),
        #     'brand':response.xpath("normalize-space(//div[@class='a-section a-spacing-none']/a/text())").get(),
        #     'product-specification':response.xpath("normalize-space(//ul[@class='a-unordered-list a-vertical a-spacing-mini']/li/span/text())").get(),
        #     'product-description':response.xpath("normalize-space(//div[@class='a-row feature']/div[2]/p/text())").get(),
        #     'user-agent':response.request.headers['User-Agent']
        # }
        item = dict()
        item['title'] = response.xpath("//div[@class='pdp-price-info']/h1/text()").extract()
        item['price'] = response.xpath("normalize-space(//span[@class='pdp-price']/strong/text())").extract()
        item['product-specification'] = response.xpath("//div[@class='index-tableContainer']/div/div/text()").extract()
        item['product-specification'] = [p.replace("\t", "") for p in item['product-specification']]
        yield item
        # yield {
        #     'title':response.xpath("normalize-space(//span[@class='a-size-large']/text())").extract(),
        #     'brand':response.xpath("normalize-space(//div[@class='a-section a-spacing-none']/a/text())").extract(),
        #     'product-specification':response.xpath("//ul[@class='a-unordered-list a-vertical a-spacing-mini']/li/span/text()").extract(),
        #     'product-description':response.xpath("normalize-space(//div[@class='a-row feature']/div[2]/p/text())").extract(),
        # }

# //div[@class="search-searchProductsContainer row-base"]//section//ul//li[@class="product-base"]//a//@href

Комментарии в коде показывают все мои попытки.

Начальный URL передан как URL в аргументе

xpath для href, который будет использоваться в экстракторе ссылок: // li [@ class = 'product-base'] / a / @ href . Но проблема в том, что к href нужно добавить перед извлеченным значением экстрактора ссылок и, следовательно, лямбда-функцией для process_value. Но код не запускается.


2020-05-26 02:52:12 [scrapy.core.engine] INFO: Spider opened
2020-05-26 02:52:12 [scrapy.extensions.logstats] INFO: Crawled 0 pages (at 0 pages/min), scraped 0 items (at 0 items/min)
2020-05-26 02:52:12 [scrapy.extensions.telnet] INFO: Telnet console listening on
2020-05-26 02:52:12 [scrapy.core.engine] DEBUG: Crawled (200) <GET> (referer: None)
2020-05-26 02:52:13 [scrapy.core.engine] DEBUG: Crawled (200) <GET> (referer: None)
2020-05-26 02:52:13 [scrapy.core.engine] INFO: Closing spider (finished)
2020-05-26 02:52:13 [scrapy.statscollectors] INFO: Dumping Scrapy stats:
{'downloader/request_bytes': 1023,
 'downloader/request_count': 2,
 'downloader/request_method_count/GET': 2,
 'downloader/response_bytes': 87336,
 'downloader/response_count': 2,
 'downloader/response_status_count/200': 2,
 'elapsed_time_seconds': 0.76699,
 'finish_reason': 'finished',
 'finish_time': datetime.datetime(2020, 5, 25, 21, 22, 13, 437855),
 'log_count/DEBUG': 2,
 'log_count/INFO': 10,
 'log_count/WARNING': 1,
 'memusage/max': 51507200,
 'memusage/startup': 51507200,
 'response_received_count': 2,
 'robotstxt/request_count': 1,
 'robotstxt/response_count': 1,
 'robotstxt/response_status_count/200': 1,
 'scheduler/dequeued': 1,
 'scheduler/dequeued/memory': 1,
 'scheduler/enqueued': 1,
 'scheduler/enqueued/memory': 1,
 'start_time': datetime.datetime(2020, 5, 25, 21, 22, 12, 670865)}
2020-05-26 02:52:13 [scrapy.core.engine] INFO: Spider closed (finished)

Любая помощь будет принята с благодарностью.

1 Ответ

2 голосов
/ 26 мая 2020

Эта страница использует JavaScript для добавления элемента, но не читает его из внешнего файла, но содержит все данные в теге <script>

import requests
from bs4 import BeautifulSoup
import json

base_url = ""

r = requests.get(base_url)

soup = BeautifulSoup(r.text, 'html.parser')

# get .text
scripts = soup.find_all('script')[8].text

# remove window.__myx = 
script = scripts.split('=', 1)[1]

# convert to dictionary
data = json.loads(script)

for item in data['searchData']['results']['products']:
    #for key, value in item.items():
    #    print(key, '=', value)

    print('product:', item['product'])
    #print('productId:', item['productId'])
    #print('brand:', item['brand'])
    print('url:', '' + item['landingPageUrl'])


product: Puma Men Black Rapid Runner IDP Running Shoes
product: Puma Men White Smash Leather Sneakers
product: Puma Unisex Grey Escaper Core Running Shoes
product: Red Tape Men Brown Leather Derbys

РЕДАКТИРОВАТЬ: То же самое с Scrapy

Вы можете поместить весь код в один файл и запустить python без создания проекта.

Он использует meta для отправки данных о продукте из одного парсера (который анализирует главную страницу) другому парсеру (который анализирует страницу продукта)

import scrapy
import json

class MySpider(scrapy.Spider):

    name = 'myspider'

    start_urls = ['']

    def parse(self, response):
        print('url:', response.url)

        scripts = response.xpath('//script/text()')[9].get()

        # remove window.__myx = 
        script = scripts.split('=', 1)[1]

        # convert to dictionary
        data = json.loads(script)

        for item in data['searchData']['results']['products']:

            info = {
                'product': item['product'],
                'productId': item['productId'],
                'brand': item['brand'],
                'url': '' + item['landingPageUrl'],

            #yield info

            yield response.follow(item['landingPageUrl'], callback=self.parse_item, meta={'item': info})

    def parse_item(self, response):
        print('url:', response.url)

        info = response.meta['item']

        # TODO: parse product page with more information

        yield info

# --- run without project and save in `output.csv` ---

from scrapy.crawler import CrawlerProcess

c = CrawlerProcess({
    'USER_AGENT': 'Mozilla/5.0',
    # save in file CSV, JSON or XML
    'FEED_FORMAT': 'csv',     # csv, json, xml
    'FEED_URI': 'output.csv', #