KnowledgeHub
Questions
Tags
Users
Search
Alex Rivera
|
Logout
Edit Question
Title
Body
I have amended the code based on solutions offered below by the great folks here; I get the error shown below the code here. from scrapy.spider import BaseSpider from scrapy.selector import HtmlXPathSelector from scrapy.utils.response import get_base_url from scrapy.utils.url import urljoin_rfc from dmoz2.items import DmozItem class DmozSpider(BaseSpider): name = "namastecopy2" allowed_domains = ["namastefoods.com"] start_urls = [ "http://www.namastefoods.com/products/cgi-bin/products.cgi?Function=show&Category_Id=4&Id=1", "http://www.namastefoods.com/products/cgi-bin/products.cgi?Function=show&Category_Id=4&Id=12", ] def parse(self, response): hxs = HtmlXPathSelector(response) sites = hxs.select('/html/body/div/div[2]/table/tr/td[2]/table/tr') items = [] for site in sites: item = DmozItem() item['manufacturer'] = 'Namaste Foods' item['productname'] = site.select('td/h1/text()').extract() item['description'] = site.select('//*[@id="info-col"]/p[7]/strong/text()').extract() item['ingredients'] = site.select('td[1]/table/tr/td[2]/text()').extract() item['ninfo'] = site.select('td[2]/ul/li[3]/img/@src').extract() #insert code that will save the above image path for ninfo as an absolute path base_url = get_base_url(response) relative_url = site.select('//*[@id="showImage"]/@src').extract() item['image_urls'] = urljoin_rfc(base_url, relative_url) items.append(item) return items My items.py looks like this: from scrapy.item import Item, Field class DmozItem(Item): # define the fields for your item here like: productid = Field() manufacturer = Field() productname = Field() description = Field() ingredients = Field() ninfo = Field() imagename = Field() image_paths = Field() relative_images = Field() image_urls = Field() pass </c
Tags (comma-separated)
Save Edits
Cancel