undefined不是有效的uri或options对象。 Nodejs的

时间:2018-07-17 15:14:05

标签: javascript node.js web-crawler

我目前正在NodeJS中构建web scraper,并且遇到了某些问题。运行代码后,我收到此错误:

  

未定义不是有效的uri或options对象。

我不确定如何绕过此错误,我看过以下示例:Example OneExample Two

这是我所有的代码:

var request = require('request');
var cheerio = require('cheerio');
var URL = require('url-parse');

var START_URL = "http://example.com";

var pagesVisited = {};
var numPagesVisited = 0;
var pagesToVisit = [];
var url = new URL(START_URL);
var baseUrl = url.protocol + "//" + url.hostname;

pagesToVisit.push(START_URL);
setInterval(crawl,5000);

function crawl() {
  var nextPage = pagesToVisit.pop();
  if (nextPage in pagesVisited) {
    // We've already visited this page, so repeat the crawl
    setInterval(crawl,5000);
  } else {
    // New page we haven't visited
    visitPage(nextPage, crawl);
  }
}

function visitPage(url, callback) {
  // Add page to our set
  pagesVisited[url] = true;
  numPagesVisited++;

  // Make the request
  console.log("Visiting page " + url);
  request(url, function(error, response, body) {
     // Check status code (200 is HTTP OK)
     console.log("Status code: " + response.statusCode);
     if(response.statusCode !== 200) {
       console.log(response.statusCode);
       callback();
       return;
     }else{
       console.log(error);
     }
     // Parse the document body
     var $ = cheerio.load(body);
       collectInternalLinks($);
       // In this short program, our callback is just calling crawl()
       callback();

  });
}

function collectInternalLinks($) {
    var relativeLinks = $("a[href^='/']");
    console.log("Found " + relativeLinks.length + " relative links on page");
    relativeLinks.each(function() {
        pagesToVisit.push(baseUrl + $(this).attr('href'));
    });
}

2 个答案:

答案 0 :(得分:0)

您的pagesToVisit清空后,该网址将是未定义的,因为在空数组上调用pop会返回该值。

我会在visitPage中添加一个网址不是未定义的检查,例如

function visitPage(url, callback) {
    if (!url) {
        // We're done
        return;
    }

或者在抓取中,检查pagesToVisit是否包含元素,例如

function crawl() {
  var nextPage = pagesToVisit.pop();
  if (!nextPage) {
      // We're done!
      console.log('Crawl complete!');
  } else if (nextPage in pagesVisited) {
    // We've already visited this page, so repeat the crawl
    setInterval(crawl,5000);
  } else {
    // New page we haven't visited
    visitPage(nextPage, crawl);
  }
}

答案 1 :(得分:0)

根据Terry Lennox's answer的提示,我对crawl()函数进行了一些修改:

function crawl() {
    var nextPage = pagesToVisit.pop();
    if (nextPage in pagesVisited) {
        // We've already visited this page, so repeat the crawl
        setInterval(crawl, 5000);
    } else if(nextPage) {
        // New page we haven't visited
        visitPage(nextPage, crawl);
    } 
} 

我要做的就是在调用visitPage()之前检查弹出的元素是否存在。

我得到以下输出:

Visiting page http://example.com
Status code: 200
response.statusCode:  200
null
Found 0 relative links on page
^C