{"_id":"@belym.a.2105/broken-link-checker","_rev":"20-ca5f42a810df824ff1f0760c49b110d7","name":"@belym.a.2105/broken-link-checker","dist-tags":{"alpha":"0.7.10-alpha.1","latest":"0.7.10"},"versions":{"0.7.9":{"name":"@belym.a.2105/broken-link-checker","version":"0.7.9","keywords":["404","html","hyperlink","links","seo","url"],"author":{"url":"https://www.svachon.com/","name":"Steven Vachon","email":"contact@svachon.com"},"license":"MIT","_id":"@belym.a.2105/broken-link-checker@0.7.9","maintainers":[{"name":"belym.a.2105","email":"belym.a.2105@gmail.com"}],"homepage":"https://github.com/AndreyBelym/broken-link-checker#readme","bugs":{"url":"https://github.com/AndreyBelym/broken-link-checker/issues"},"bin":{"blc":"bin/blc","broken-link-checker":"bin/blc"},"dist":{"shasum":"ec82ed041da5fb53246e30d9125781ba1e6bc5ae","tarball":"https://registry.npmjs.org/@belym.a.2105/broken-link-checker/-/broken-link-checker-0.7.9.tgz","fileCount":23,"integrity":"sha512-Qya54gLWSZ4OMQ9RK9/PA9Gux+paVQ5brviK0Hzqt9ibzxl7m1AjoIfwk9In3mXpLlcgQPQwggmV7LRGB5AGkA==","signatures":[{"sig":"MEUCIQDR8CY3pAR2FAaJ+HtZ1kyz4EIOufKors7a2eV+76KjDQIgYn58ISVzW8NtcweonRQA8jPazB276wCeptxGpT5DY2Y=","keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA"}],"unpackedSize":83461,"npm-signature":"-----BEGIN PGP SIGNATURE-----\r\nVersion: OpenPGP.js v3.0.4\r\nComment: https://openpgpjs.org\r\n\r\nwsFcBAEBCAAQBQJcEjbLCRA9TVsSAnZWagAAJ+IP+wV3M6jRK8XuTBXNUPMG\n1DCAdnNmAmF2jomWrF2Fb/eMZqPTjJ5FFP1P8LYdRVfjEka2z3l0XaxdWZ6K\nwEOerNr7qB7wzlgJGwA5WK2yFTKJx9xdl6B+RlJdWwzeVr4j4MlxyBtXVOEf\nj8BVK1/ui1r7BbZj6+O208r9AEbcTpxhaCSp5cFdoq8xFOs/m/mRiTtIswj0\n+0EtZsh932Deqm6oMkNLg+TlmJ/SVe4ktP2fSZqAP/BC018tgD6VBz6+7riY\nJZUjLmQONvbKvz62FJ5Nd4M+CBea2B3lukZyf9Ey+5ZH/ThdEo26L4odLEdX\nJiv+7x5X0Cl9cbxg+TMCouYUiPUhb/lA0L8hMQv/DIu3V2OeiDR2QVDtOgtU\nQY/rEz+yKKkD4DI5kbwghnRdDfEEiw5uVd8+efRdjNP3S4BnwlpCcRma93/l\ntEShzk2dBzWUuSUfdRURgth/4zjvgFqpVQs2addemFA1WPGRohxvoAM/wOGx\nbKtXHXNd4/rBWHDDsZeHI4uzv18ZTRzhcx7ARH10oFO/ka0UcENH8ys0Wmz5\nhCXv0xCEtKWPlegQSqvphS1+KR6u2I1hSpd32L41OakcDmWDXrux57vTAR2t\nc+FPneAQO2wCoxar6aW8b8b3ZXjt/hL4IxfFPp2XUfAadwEx1Cf4SiyL/Oxe\n69td\r\n=CAPC\r\n-----END PGP SIGNATURE-----\r\n"},"main":"lib","engines":{"node":">= 0.10"},"gitHead":"66ccca5e039c7cbbb0a3abbb4b2f493ac0a4185a","scripts":{"test":"mocha test/ --reporter spec --check-leaks --bail","test-watch":"mocha test/ --reporter spec --check-leaks --bail -w"},"_npmUser":{"name":"belym.a.2105","email":"belym.a.2105@gmail.com"},"deprecated":"The broken-link-checker package has reached end-of-life and will not receive further updates.","repository":{"url":"git+https://github.com/AndreyBelym/broken-link-checker.git","type":"git"},"_npmVersion":"6.4.1","description":"Find broken links, missing images, etc in your HTML.","directories":{},"_nodeVersion":"11.4.0","dependencies":{"got":"^9.4.0","chalk":"^1.1.3","errno":"~0.1.4","extend":"^3.0.0","nopter":"~0.3.0","parse5":"^3.0.2","urlobj":"0.0.11","calmcard":"~0.1.1","urlcache":"~0.7.0","is-stream":"^1.0.1","is-string":"^1.0.4","link-types":"^1.1.0","char-spinner":"^1.0.1","maybe-callback":"^2.1.0","robot-directives":"~0.3.0","robots-txt-guard":"~0.1.0","robots-txt-parse":"~0.0.4","humanize-duration":"^3.9.1","default-user-agent":"^1.0.0","http-equiv-refresh":"^1.0.0","condense-whitespace":"^1.0.0","limited-request-queue":"^2.0.0"},"_hasShrinkwrap":false,"devDependencies":{"st":"^1.2.0","chai":"^3.5.0","mocha":"^3.0.2","slashes":"^1.0.5","chai-like":"~0.2.10","chai-things":"~0.2.0","es6-promise":"^4.1.0","object.assign":"^4.0.4","chai-as-promised":"^6.0.0"},"_npmOperationalInternal":{"tmp":"tmp/broken-link-checker_0.7.9_1544697546259_0.26173340342489504","host":"s3://npm-registry-packages"}},"0.7.10-alpha.1":{"name":"@belym.a.2105/broken-link-checker","version":"0.7.10-alpha.1","keywords":["404","html","hyperlink","links","seo","url"],"author":{"url":"https://www.svachon.com/","name":"Steven Vachon","email":"contact@svachon.com"},"license":"MIT","_id":"@belym.a.2105/broken-link-checker@0.7.10-alpha.1","maintainers":[{"name":"belym.a.2105","email":"belym.a.2105@gmail.com"}],"homepage":"https://github.com/AndreyBelym/broken-link-checker#readme","bugs":{"url":"https://github.com/AndreyBelym/broken-link-checker/issues"},"bin":{"blc":"bin/blc","broken-link-checker":"bin/blc"},"dist":{"shasum":"afaf4afe195e93d8382b07c7dcb1d1134f876937","tarball":"https://registry.npmjs.org/@belym.a.2105/broken-link-checker/-/broken-link-checker-0.7.10-alpha.1.tgz","fileCount":24,"integrity":"sha512-7Q5ck/V2+fa8BJ24iGDgxrs5d+miqZhEetE+2/CYZWC0yMw1iExhCVCkFf488jmq7VASGoG49iWyhvGkvxi8iQ==","signatures":[{"sig":"MEYCIQDJtRwpKofenyedr9HZDA8ayG3ekFq6dwINw4CQxpi/qAIhAKwmtmbsahQFRHpmsb6OSeCS80vgiRczVT1yvx5b6h5/","keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA"}],"unpackedSize":84946,"npm-signature":"-----BEGIN PGP SIGNATURE-----\r\nVersion: OpenPGP.js v3.0.4\r\nComment: https://openpgpjs.org\r\n\r\nwsFcBAEBCAAQBQJdY6oiCRA9TVsSAnZWagAATfsP/3V1vq66FWKUv1ihAe4A\nlOvMEaGUk3c3VJ+LsfX3bq+WXfCu9/vbXhoGIf/A1/mA48LdlyxWf/CyDKE2\nya81rOXIWH7o/cK6zg3orqe6eMdk9xRuuQ3GCWREw0II6LOj50Eu2YVWBI9u\nWmUVppomSj0B4azDhV9ye6F7HmEgvb9Tgl6yEBhnXh8sJE8ivl2iiO4niTK/\nkvGDLU65ohBZVxjmftXxhTxWwNsX4/So8GkP07590RSgf40ZRyry8zC+PuO7\n7+gg6bdmwtZWUmFJS8LZrBeOkxtDqOuV33ld9WIs5kO3qRfYJGi3YmcSlt0q\n9Tu52VCLjYJswbfxVVzb/Kx/KrGYC9wqPhrG/rgkNMq2nijyLrsnO45MwGog\nPBQrNOMrbkHDDj/lqc1R5ZNchrhoZoKlfKfEF8lgm0FEUWL0MuWKOjy15qSe\nkYkqo+jV0nfux+nN1NCvctouGo0Yn+fmct5Ag72CPeOKu8aDTRV0e5eSOINb\nVcb74429QQ4HuArrezsu8TxBo+RSiGbcKLGgAO7xTW6ZNLWZRpD3OLGohgzr\nCRSj0C1ZXa/f/rxCfcg9b/rRw+XC6wr6soKO+DQ2ofwjdkrWUqTwZ2va2BA/\n8C6T11i5xqRUiQAlTbUWlfHVU8V2x6ZItlBT/W9tHgd2xQ8OkS2gfXhshsT1\nZrfW\r\n=XNYg\r\n-----END PGP SIGNATURE-----\r\n"},"main":"lib","readme":"# broken-link-checker [![NPM Version][npm-image]][npm-url] [![Build Status][travis-image]][travis-url] [![Dependency Status][david-image]][david-url]\r\n\r\n> Find broken links, missing images, etc in your HTML.\r\n\r\nFeatures:\r\n* Stream-parses local and remote HTML pages\r\n* Concurrently checks multiple links\r\n* Supports various HTML elements/attributes, not just `<a href>`\r\n* Supports redirects, absolute URLs, relative URLs and `<base>`\r\n* Honors robot exclusions\r\n* Provides detailed information about each link (HTTP and HTML)\r\n* URL keyword filtering with wildcards\r\n* Pause/Resume at any time\r\n\r\n\r\n## Installation\r\n\r\n[Node.js](http://nodejs.org/) `>= 0.10` is required; `< 4.0` will need `Promise` and `Object.assign` polyfills.\r\n\r\nThere're two ways to use it:\r\n\r\n### Command Line Usage\r\nTo install, type this at the command line:\r\n```shell\r\nnpm install broken-link-checker -g\r\n```\r\nAfter that, check out the help for available options:\r\n```shell\r\nblc --help\r\n```\r\nA typical site-wide check might look like:\r\n```shell\r\nblc http://yoursite.com -ro\r\n```\r\n\r\n### Programmatic API\r\nTo install, type this at the command line:\r\n```shell\r\nnpm install broken-link-checker\r\n```\r\nThe rest of this document will assist you with how to use the API.\r\n\r\n\r\n## Classes\r\n\r\n### `blc.HtmlChecker(options, handlers)`\r\nScans an HTML document to find broken links.\r\n\r\n* `handlers.complete` is fired after the last result or zero results.\r\n* `handlers.html` is fired after the HTML document has been fully parsed.\r\n  * `tree` is supplied by [parse5](https://npmjs.com/parse5)\r\n  * `robots` is an instance of [robot-directives](https://npmjs.com/robot-directives) containing any `<meta>` robot exclusions.\r\n* `handlers.junk` is fired with data on each skipped link, as configured in options.\r\n* `handlers.link` is fired with the result of each discovered link (broken or not).\r\n\r\n* `.clearCache()` will remove any cached URL responses. This is only relevant if the `cacheResponses` option is enabled.\r\n* `.numActiveLinks()` returns the number of links with active requests.\r\n* `.numQueuedLinks()` returns the number of links that currently have no active requests.\r\n* `.pause()` will pause the internal link queue, but will not pause any active requests.\r\n* `.resume()` will resume the internal link queue.\r\n* `.scan(html, baseUrl)` parses & scans a single HTML document. Returns `false` when there is a previously incomplete scan (and `true` otherwise).\r\n  * `html` can be a stream or a string.\r\n  * `baseUrl` is the address to which all relative URLs will be made absolute. Without a value, links to relative URLs will output an \"Invalid URL\" error.\r\n\r\n```js\r\nvar htmlChecker = new blc.HtmlChecker(options, {\r\n\thtml: function(tree, robots){},\r\n\tjunk: function(result){},\r\n\tlink: function(result){},\r\n\tcomplete: function(){}\r\n});\r\n\r\nhtmlChecker.scan(html, baseUrl);\r\n```\r\n\r\n### `blc.HtmlUrlChecker(options, handlers)`\r\nScans the HTML content at each queued URL to find broken links.\r\n\r\n* `handlers.end` is fired when the end of the queue has been reached.\r\n* `handlers.html` is fired after a page's HTML document has been fully parsed.\r\n  * `tree` is supplied by [parse5](https://npmjs.com/parse5).\r\n  * `robots` is an instance of [robot-directives](https://npmjs.com/robot-directives) containing any `<meta>` and `X-Robots-Tag` robot exclusions.\r\n* `handlers.junk` is fired with data on each skipped link, as configured in options.\r\n* `handlers.link` is fired with the result of each discovered link (broken or not) within the current page.\r\n* `handlers.page` is fired after a page's last result, on zero results, or if the HTML could not be retrieved.\r\n\r\n* `.clearCache()` will remove any cached URL responses. This is only relevant if the `cacheResponses` option is enabled.\r\n* `.dequeue(id)` removes a page from the queue. Returns `true` on success or an `Error` on failure.\r\n* `.enqueue(pageUrl, customData)` adds a page to the queue. Queue items are auto-dequeued when their requests are complete. Returns a queue ID on success or an `Error` on failure.\r\n  * `customData` is optional data that is stored in the queue item for the page.\r\n* `.numActiveLinks()` returns the number of links with active requests.\r\n* `.numPages()` returns the total number of pages in the queue.\r\n* `.numQueuedLinks()` returns the number of links that currently have no active requests.\r\n* `.pause()` will pause the queue, but will not pause any active requests.\r\n* `.resume()` will resume the queue.\r\n\r\n```js\r\nvar htmlUrlChecker = new blc.HtmlUrlChecker(options, {\r\n\thtml: function(tree, robots, response, pageUrl, customData){},\r\n\tjunk: function(result, customData){},\r\n\tlink: function(result, customData){},\r\n\tpage: function(error, pageUrl, customData){},\r\n\tend: function(){}\r\n});\r\n\r\nhtmlUrlChecker.enqueue(pageUrl, customData);\r\n```\r\n\r\n### `blc.SiteChecker(options, handlers)`\r\nRecursively scans (crawls) the HTML content at each queued URL to find broken links.\r\n\r\n* `handlers.end` is fired when the end of the queue has been reached.\r\n* `handlers.html` is fired after a page's HTML document has been fully parsed.\r\n  * `tree` is supplied by [parse5](https://npmjs.com/parse5).\r\n  * `robots` is an instance of [robot-directives](https://npmjs.com/robot-directives) containing any `<meta>` and `X-Robots-Tag` robot exclusions.\r\n* `handlers.junk` is fired with data on each skipped link, as configured in options.\r\n* `handlers.link` is fired with the result of each discovered link (broken or not) within the current page.\r\n* `handlers.page` is fired after a page's last result, on zero results, or if the HTML could not be retrieved.\r\n* `handlers.robots` is fired after a site's robots.txt has been downloaded and provides an instance of [robots-txt-guard](https://npmjs.com/robots-txt-guard).\r\n* `handlers.site` is fired after a site's last result, on zero results, or if the *initial* HTML could not be retrieved.\r\n\r\n* `.clearCache()` will remove any cached URL responses. This is only relevant if the `cacheResponses` option is enabled.\r\n* `.dequeue(id)` removes a site from the queue. Returns `true` on success or an `Error` on failure.\r\n* `.enqueue(siteUrl, customData)` adds [the first page of] a site to the queue. Queue items are auto-dequeued when their requests are complete. Returns a queue ID on success or an `Error` on failure.\r\n  * `customData` is optional data that is stored in the queue item for the site.\r\n* `.numActiveLinks()` returns the number of links with active requests.\r\n* `.numPages()` returns the total number of pages in the queue.\r\n* `.numQueuedLinks()` returns the number of links that currently have no active requests.\r\n* `.numSites()` returns the total number of sites in the queue.\r\n* `.pause()` will pause the queue, but will not pause any active requests.\r\n* `.resume()` will resume the queue.\r\n\r\n**Note:** `options.filterLevel` is used for determining which links are recursive.\r\n\r\n```js\r\nvar siteChecker = new blc.SiteChecker(options, {\r\n\trobots: function(robots, customData){},\r\n\thtml: function(tree, robots, response, pageUrl, customData){},\r\n\tjunk: function(result, customData){},\r\n\tlink: function(result, customData){},\r\n\tpage: function(error, pageUrl, customData){},\r\n\tsite: function(error, siteUrl, customData){},\r\n\tend: function(){}\r\n});\r\n\r\nsiteChecker.enqueue(siteUrl, customData);\r\n```\r\n\r\n### `blc.UrlChecker(options, handlers)`\r\nRequests each queued URL to determine if they are broken.\r\n\r\n* `handlers.end` is fired when the end of the queue has been reached.\r\n* `handlers.link` is fired for each result (broken or not).\r\n\r\n* `.clearCache()` will remove any cached URL responses. This is only relevant if the `cacheResponses` option is enabled.\r\n* `.dequeue(id)` removes a URL from the queue. Returns `true` on success or an `Error` on failure.\r\n* `.enqueue(url, baseUrl, customData)` adds a URL to the queue. Queue items are auto-dequeued when their requests are completed. Returns a queue ID on success or an `Error` on failure.\r\n  * `baseUrl` is the address to which all relative URLs will be made absolute. Without a value, links to relative URLs will output an \"Invalid URL\" error.\r\n  * `customData` is optional data that is stored in the queue item for the URL.\r\n* `.numActiveLinks()` returns the number of links with active requests.\r\n* `.numQueuedLinks()` returns the number of links that currently have no active requests.\r\n* `.pause()` will pause the queue, but will not pause any active requests.\r\n* `.resume()` will resume the queue.\r\n\r\n```js\r\nvar urlChecker = new blc.UrlChecker(options, {\r\n\tlink: function(result, customData){},\r\n\tend: function(){}\r\n});\r\n\r\nurlChecker.enqueue(url, baseUrl, customData);\r\n```\r\n\r\n## Options\r\n\r\n### `options.acceptedSchemes`\r\nType: `Array`  \r\nDefault value: `[\"http\",\"https\"]`  \r\nWill only check links with schemes/protocols mentioned in this list. Any others (except those in `excludedSchemes`) will output an \"Invalid URL\" error.\r\n\r\n### `options.cacheExpiryTime`\r\nType: `Number`  \r\nDefault Value: `3600000` (1 hour)  \r\nThe number of milliseconds in which a cached response should be considered valid. This is only relevant if the `cacheResponses` option is enabled.\r\n\r\n### `options.cacheResponses`\r\nType: `Boolean`  \r\nDefault Value: `true`  \r\nURL request results will be cached when `true`. This will ensure that each unique URL will only be checked once.\r\n\r\n### `options.excludedKeywords`\r\nType: `Array`  \r\nDefault value: `[]`  \r\nWill not check or output links that match the keywords and glob patterns in this list. The only wildcard supported is `*`.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.excludedSchemes`\r\nType: `Array`  \r\nDefault value: `[\"data\",\"geo\",\"javascript\",\"mailto\",\"sms\",\"tel\"]`  \r\nWill not check or output links with schemes/protocols mentioned in this list. This avoids the output of \"Invalid URL\" errors with links that cannot be checked.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.excludeExternalLinks`\r\nType: `Boolean`  \r\nDefault value: `false`  \r\nWill not check or output external links when `true`; relative links with a remote `<base>` included.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.excludeInternalLinks`\r\nType: `Boolean`  \r\nDefault value: `false`  \r\nWill not check or output internal links when `true`.\r\n\r\nThis option does *not* apply to `UrlChecker` nor `SiteChecker`'s *crawler*.\r\n\r\n### `options.excludeLinksToSamePage`\r\nType: `Boolean`  \r\nDefault value: `true`  \r\nWill not check or output links to the same page; relative and absolute fragments/hashes included.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.filterLevel`\r\nType: `Number`  \r\nDefault value: `1`  \r\nThe tags and attributes that are considered links for checking, split into the following levels:\r\n* `0`: clickable links\r\n* `1`: clickable links, media, iframes, meta refreshes\r\n* `2`: clickable links, media, iframes, meta refreshes, stylesheets, scripts, forms\r\n* `3`: clickable links, media, iframes, meta refreshes, stylesheets, scripts, forms, metadata\r\n\r\nRecursive links have a slightly different filter subset. To see the exact breakdown of both, check out the [tag map](https://github.com/stevenvachon/broken-link-checker/blob/master/lib/internal/tags.js). `<base>` is not listed because it is not a link, though it is always parsed.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.honorRobotExclusions`\r\nType: `Boolean`  \r\nDefault value: `true`  \r\nWill not scan pages that search engine crawlers would not follow. Such will have been specified with any of the following:\r\n* `<a rel=\"nofollow\" href=\"…\">`\r\n* `<area rel=\"nofollow\" href=\"…\">`\r\n* `<meta name=\"robots\" content=\"noindex,nofollow,…\">`\r\n* `<meta name=\"googlebot\" content=\"noindex,nofollow,…\">`\r\n* `<meta name=\"robots\" content=\"unavailable_after: …\">`\r\n* `X-Robots-Tag: noindex,nofollow,…`\r\n* `X-Robots-Tag: googlebot: noindex,nofollow,…`\r\n* `X-Robots-Tag: otherbot: noindex,nofollow,…`\r\n* `X-Robots-Tag: unavailable_after: …`\r\n* robots.txt\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.maxSockets`\r\nType: `Number`  \r\nDefault value: `Infinity`  \r\nThe maximum number of links to check at any given time.\r\n\r\n### `options.maxSocketsPerHost`\r\nType: `Number`  \r\nDefault value: `1`  \r\nThe maximum number of links per host/port to check at any given time. This avoids overloading a single target host with too many concurrent requests. This will not limit concurrent requests to other hosts.\r\n\r\n### `options.rateLimit`\r\nType: `Number`  \r\nDefault value: `0`  \r\nThe number of milliseconds to wait before each request.\r\n\r\n### `options.requestMethod`\r\nType: `String`  \r\nDefault value: `\"head\"`  \r\nThe HTTP request method used in checking links. If you experience problems, try using `\"get\"`, however `options.retry405Head` should have you covered.\r\n\r\n### `options.retry405Head`\r\nType: `Boolean`  \r\nDefault value: `true`  \r\nSome servers do not respond correctly to a `\"head\"` request method. When `true`, a link resulting in an HTTP 405 \"Method Not Allowed\" error will be re-requested using a `\"get\"` method before deciding that it is broken.\r\n\r\n### `options.userAgent`\r\nType: `String`  \r\nDefault value: `\"broken-link-checker/0.7.0 Node.js/5.5.0 (OS X El Capitan; x64)\"` (or similar)  \r\nThe HTTP user-agent to use when checking links as well as retrieving pages and robot exclusions.\r\n\r\n\r\n## Handling Broken/Excluded Links\r\nA broken link will have a `broken` value of `true` and a reason code defined in `brokenReason`. A link that was not checked (emitted as `\"junk\"`) will have an `excluded` value of `true` and a reason code defined in `excludedReason`.\r\n```js\r\nif (result.broken) {\r\n\tconsole.log(result.brokenReason);\r\n\t//=> HTTP_404\r\n} else if (result.excluded) {\r\n\tconsole.log(result.excludedReason);\r\n\t//=> BLC_ROBOTS\r\n}\r\n```\r\n\r\nAdditionally, more descriptive messages are available for each reason code:\r\n```js\r\nconsole.log(blc.BLC_ROBOTS);       //=> Robots Exclusion\r\nconsole.log(blc.ERRNO_ECONNRESET); //=> connection reset by peer (ECONNRESET)\r\nconsole.log(blc.HTTP_404);         //=> Not Found (404)\r\n\r\n// List all\r\nconsole.log(blc);\r\n```\r\n\r\nPutting it all together:\r\n```js\r\nif (result.broken) {\r\n\tconsole.log(blc[result.brokenReason]);\r\n} else if (result.excluded) {\r\n\tconsole.log(blc[result.excludedReason]);\r\n}\r\n```\r\n\r\n## HTML and HTTP information\r\nDetailed information for each link result is provided. Check out the [schema](https://github.com/stevenvachon/broken-link-checker/blob/master/lib/internal/linkObj.js#L16-L64) or:\r\n```js\r\nconsole.log(result);\r\n```\r\n\r\n\r\n## Roadmap Features\r\n* fix issue where same-page links are not excluded when cache is enabled, despite `excludeLinksToSamePage===true`\r\n* publicize filter handlers\r\n* add cheerio support by using parse5's htmlparser2 tree adaptor?\r\n* add `rejectUnauthorized:false` option to avoid `UNABLE_TO_VERIFY_LEAF_SIGNATURE`\r\n* load sitemap.xml at end of each `SiteChecker` site to possibly check pages that were not linked to\r\n* remove `options.excludedSchemes` and handle schemes not in `options.acceptedSchemes` as junk?\r\n* change order of checking to: tcp error, 4xx code (broken), 5xx code (undetermined), 200\r\n* abort download of body when `options.retry405Head===true`\r\n* option to retry broken links a number of times (default=0)\r\n* option to scrape `response.body` for erroneous sounding text (using [fathom](https://npmjs.com/fathom-web)?), since an error page could be presented but still have code 200\r\n* option to check broken link on archive.org for archived version (using [this lib](https://npmjs.com/archive.org))\r\n* option to run `HtmlUrlChecker` checks on page load (using [jsdom](https://npmjs.com/jsdom)) to include links added with JavaScript?\r\n* option to check if hashes exist in target URL document?\r\n* option to parse Markdown in `HtmlChecker` for links\r\n* option to play sound when broken link is found\r\n* option to hide unbroken links\r\n* option to check plain text URLs\r\n* add throttle profiles (0–9, -1 for \"custom\") for easy configuring\r\n* check [ftp:](https://nmjs.com/ftp), [sftp:](https://npmjs.com/ssh2) (for downloadable files)\r\n* check ~~mailto:~~, news:, nntp:, telnet:?\r\n* check local files if URL is relative and has no base URL?\r\n* cli json mode -- streamed or not?\r\n* cli non-tty mode -- change nesting ASCII artwork to time stamps?\r\n\r\n\r\n[npm-image]: https://img.shields.io/npm/v/broken-link-checker.svg\r\n[npm-url]: https://npmjs.org/package/broken-link-checker\r\n[travis-image]: https://img.shields.io/travis/stevenvachon/broken-link-checker.svg\r\n[travis-url]: https://travis-ci.org/stevenvachon/broken-link-checker\r\n[david-image]: https://img.shields.io/david/stevenvachon/broken-link-checker.svg\r\n[david-url]: https://david-dm.org/stevenvachon/broken-link-checker\r\n","engines":{"node":">= 0.10"},"gitHead":"f6ff3b32901d68556dd9bf6a67c51987fdb96eb8","scripts":{"test":"mocha test/ --reporter spec --check-leaks --bail","test-watch":"mocha test/ --reporter spec --check-leaks --bail -w"},"_npmUser":{"name":"belym.a.2105","email":"belym.a.2105@gmail.com"},"deprecated":"The broken-link-checker package has reached end-of-life and will not receive further updates.","repository":{"url":"git+https://github.com/AndreyBelym/broken-link-checker.git","type":"git"},"_npmVersion":"6.10.2","description":"Find broken links, missing images, etc in your HTML.","directories":{},"_nodeVersion":"12.8.0","dependencies":{"got":"^9.4.0","chalk":"^1.1.3","errno":"~0.1.4","extend":"^3.0.0","nopter":"~0.3.0","parse5":"^3.0.2","urlobj":"0.0.11","calmcard":"~0.1.1","urlcache":"~0.7.0","is-stream":"^1.0.1","is-string":"^1.0.4","link-types":"^1.1.0","char-spinner":"^1.0.1","maybe-callback":"^2.1.0","robot-directives":"~0.3.0","robots-txt-guard":"~0.1.0","robots-txt-parse":"~0.0.4","humanize-duration":"^3.9.1","default-user-agent":"^1.0.0","http-equiv-refresh":"^1.0.0","condense-whitespace":"^1.0.0","limited-request-queue":"^2.0.0"},"_hasShrinkwrap":false,"readmeFilename":"README.md","devDependencies":{"st":"^1.2.0","chai":"^3.5.0","mocha":"^3.0.2","slashes":"^1.0.5","chai-like":"~0.2.10","chai-things":"~0.2.0","es6-promise":"^4.1.0","object.assign":"^4.0.4","chai-as-promised":"^6.0.0"},"_npmOperationalInternal":{"tmp":"tmp/broken-link-checker_0.7.10-alpha.1_1566812705629_0.4525048447905651","host":"s3://npm-registry-packages"}},"0.7.10":{"name":"@belym.a.2105/broken-link-checker","version":"0.7.10","keywords":["404","html","hyperlink","links","seo","url"],"author":{"url":"https://www.svachon.com/","name":"Steven Vachon","email":"contact@svachon.com"},"license":"MIT","_id":"@belym.a.2105/broken-link-checker@0.7.10","maintainers":[{"name":"belym.a.2105","email":"belym.a.2105@gmail.com"}],"homepage":"https://github.com/AndreyBelym/broken-link-checker#readme","bugs":{"url":"https://github.com/AndreyBelym/broken-link-checker/issues"},"bin":{"blc":"bin/blc","broken-link-checker":"bin/blc"},"dist":{"shasum":"2acc90c79246657571198e7d514f7ca7293b6b3f","tarball":"https://registry.npmjs.org/@belym.a.2105/broken-link-checker/-/broken-link-checker-0.7.10.tgz","fileCount":24,"integrity":"sha512-rgskfP4cDHsaurdP3Z1tfLizhGIYFRElZ9B/llGmMj4HDQMnoC9cF2zk2X4ign93tj7bX0ks6VkG+igfvEd1sg==","signatures":[{"sig":"MEUCIAZpndDMvrHTeF8H2h5R8jCEu+TG62VikR9C2XR3T5xdAiEAwDsBpghVDqaveDu/70T1bQl6ls3yKuxFgQ9q9wfFaHg=","keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA"}],"unpackedSize":84938,"npm-signature":"-----BEGIN PGP SIGNATURE-----\r\nVersion: OpenPGP.js v3.0.4\r\nComment: https://openpgpjs.org\r\n\r\nwsFcBAEBCAAQBQJdZRkMCRA9TVsSAnZWagAAEJEP/jc0WbHZTx0YWHCfW9Vu\nI0Tm1N4u3b+K+w+rYg3CIcIiOPuceNeWVw1jwsDc2EdM8K+CPO2Vg5V0n7J6\nIUzTG2aucmRcm3Of+E7+IySppqe79UGJpVOKHDmbj90ZhoMtjpU9LTJiX5Ft\n+KA58OPfD9j9bsAkwhoQLdZ44THwpnN0Qtrxa4bN0IzqBfaZAiRgCT5Er9az\nAcpyd87kA7bzdvdzS6hVuIaOlYq5+h8ojvRc++/VOKFXvwBM9ImlSWpCRi6D\n4Xm5ufHtr+qk+n+ZA6rbEWP6Te6ZkRqj53KORuKbZo4EopGPPVBTwq8nFvq1\naXMVuMeLfhJ+GbvHzLYep3ikKbyK/FYAw0HEWi8bDnWO+dIPx2bfaHRkLi5y\nJCVzJr7YY3cNn4BU+czKlTfnCP3qDNeSI4sTveBtW74yBWbIkveqKPYN4Im+\n5r3cNdznygyk0AtzwyGTOSymADfEyQP1+ZS3RJcXvmxZsBsNTxZ0XiA97dqG\nGLm8lAKbxeExnXxZI+9oK30Nj7ntYGGj/THiXOvRsDdu5axoqLxYVe/gOI4f\npMn6JCVn/Ngred2WmRqz6CaF4ZH7p8XsoT+69Qd7UfZEOsOn+ayA5bAVtcvr\nqkW4fSYrO8xZqbDDOPBhVHc2+yj5NVoXXDdCHkDQox7iTND+gc6pVDyw0oUy\nMbi9\r\n=l/sB\r\n-----END PGP SIGNATURE-----\r\n"},"main":"lib","engines":{"node":">= 0.10"},"gitHead":"4f8821e0efb5bcf6d8422c95a46ce842d44a491c","scripts":{"test":"mocha test/ --reporter spec --check-leaks --bail","test-watch":"mocha test/ --reporter spec --check-leaks --bail -w"},"_npmUser":{"name":"belym.a.2105","email":"belym.a.2105@gmail.com"},"deprecated":"The broken-link-checker package has reached end-of-life and will not receive further updates.","repository":{"url":"git+https://github.com/AndreyBelym/broken-link-checker.git","type":"git"},"_npmVersion":"6.10.2","description":"Find broken links, missing images, etc in your HTML.","directories":{},"_nodeVersion":"12.8.0","dependencies":{"got":"^9.4.0","chalk":"^1.1.3","errno":"~0.1.4","extend":"^3.0.0","nopter":"~0.3.0","parse5":"^3.0.2","urlobj":"0.0.11","calmcard":"~0.1.1","urlcache":"~0.7.0","is-stream":"^1.0.1","is-string":"^1.0.4","link-types":"^1.1.0","char-spinner":"^1.0.1","maybe-callback":"^2.1.0","robot-directives":"~0.3.0","robots-txt-guard":"~0.1.0","robots-txt-parse":"~0.0.4","humanize-duration":"^3.9.1","default-user-agent":"^1.0.0","http-equiv-refresh":"^1.0.0","condense-whitespace":"^1.0.0","limited-request-queue":"^2.0.0"},"_hasShrinkwrap":false,"devDependencies":{"st":"^1.2.0","chai":"^3.5.0","mocha":"^3.0.2","slashes":"^1.0.5","chai-like":"~0.2.10","chai-things":"~0.2.0","es6-promise":"^4.1.0","object.assign":"^4.0.4","chai-as-promised":"^6.0.0"},"_npmOperationalInternal":{"tmp":"tmp/broken-link-checker_0.7.10_1566906636184_0.7859562284903017","host":"s3://npm-registry-packages"}}},"time":{"created":"2018-12-13T10:39:06.067Z","modified":"2025-12-03T13:13:59.208Z","0.7.9":"2018-12-13T10:39:06.395Z","0.7.10-alpha.1":"2019-08-26T09:45:05.806Z","0.7.10":"2019-08-27T11:50:36.296Z"},"bugs":{"url":"https://github.com/AndreyBelym/broken-link-checker/issues"},"author":{"url":"https://www.svachon.com/","name":"Steven Vachon","email":"contact@svachon.com"},"license":"MIT","homepage":"https://github.com/AndreyBelym/broken-link-checker#readme","keywords":["404","html","hyperlink","links","seo","url"],"repository":{"url":"git+https://github.com/AndreyBelym/broken-link-checker.git","type":"git"},"description":"Find broken links, missing images, etc in your HTML.","maintainers":[{"email":"boris.s.kirov@gmail.com","name":"kirovboris"},{"email":"romanresh@live.com","name":"romanresh"},{"email":"belym.a.2105@gmail.com","name":"belym.a.2105"}],"readme":"# broken-link-checker [![NPM Version][npm-image]][npm-url] [![Build Status][travis-image]][travis-url] [![Dependency Status][david-image]][david-url]\r\n\r\n> Find broken links, missing images, etc in your HTML.\r\n\r\nFeatures:\r\n* Stream-parses local and remote HTML pages\r\n* Concurrently checks multiple links\r\n* Supports various HTML elements/attributes, not just `<a href>`\r\n* Supports redirects, absolute URLs, relative URLs and `<base>`\r\n* Honors robot exclusions\r\n* Provides detailed information about each link (HTTP and HTML)\r\n* URL keyword filtering with wildcards\r\n* Pause/Resume at any time\r\n\r\n\r\n## Installation\r\n\r\n[Node.js](http://nodejs.org/) `>= 0.10` is required; `< 4.0` will need `Promise` and `Object.assign` polyfills.\r\n\r\nThere're two ways to use it:\r\n\r\n### Command Line Usage\r\nTo install, type this at the command line:\r\n```shell\r\nnpm install broken-link-checker -g\r\n```\r\nAfter that, check out the help for available options:\r\n```shell\r\nblc --help\r\n```\r\nA typical site-wide check might look like:\r\n```shell\r\nblc http://yoursite.com -ro\r\n```\r\n\r\n### Programmatic API\r\nTo install, type this at the command line:\r\n```shell\r\nnpm install broken-link-checker\r\n```\r\nThe rest of this document will assist you with how to use the API.\r\n\r\n\r\n## Classes\r\n\r\n### `blc.HtmlChecker(options, handlers)`\r\nScans an HTML document to find broken links.\r\n\r\n* `handlers.complete` is fired after the last result or zero results.\r\n* `handlers.html` is fired after the HTML document has been fully parsed.\r\n  * `tree` is supplied by [parse5](https://npmjs.com/parse5)\r\n  * `robots` is an instance of [robot-directives](https://npmjs.com/robot-directives) containing any `<meta>` robot exclusions.\r\n* `handlers.junk` is fired with data on each skipped link, as configured in options.\r\n* `handlers.link` is fired with the result of each discovered link (broken or not).\r\n\r\n* `.clearCache()` will remove any cached URL responses. This is only relevant if the `cacheResponses` option is enabled.\r\n* `.numActiveLinks()` returns the number of links with active requests.\r\n* `.numQueuedLinks()` returns the number of links that currently have no active requests.\r\n* `.pause()` will pause the internal link queue, but will not pause any active requests.\r\n* `.resume()` will resume the internal link queue.\r\n* `.scan(html, baseUrl)` parses & scans a single HTML document. Returns `false` when there is a previously incomplete scan (and `true` otherwise).\r\n  * `html` can be a stream or a string.\r\n  * `baseUrl` is the address to which all relative URLs will be made absolute. Without a value, links to relative URLs will output an \"Invalid URL\" error.\r\n\r\n```js\r\nvar htmlChecker = new blc.HtmlChecker(options, {\r\n\thtml: function(tree, robots){},\r\n\tjunk: function(result){},\r\n\tlink: function(result){},\r\n\tcomplete: function(){}\r\n});\r\n\r\nhtmlChecker.scan(html, baseUrl);\r\n```\r\n\r\n### `blc.HtmlUrlChecker(options, handlers)`\r\nScans the HTML content at each queued URL to find broken links.\r\n\r\n* `handlers.end` is fired when the end of the queue has been reached.\r\n* `handlers.html` is fired after a page's HTML document has been fully parsed.\r\n  * `tree` is supplied by [parse5](https://npmjs.com/parse5).\r\n  * `robots` is an instance of [robot-directives](https://npmjs.com/robot-directives) containing any `<meta>` and `X-Robots-Tag` robot exclusions.\r\n* `handlers.junk` is fired with data on each skipped link, as configured in options.\r\n* `handlers.link` is fired with the result of each discovered link (broken or not) within the current page.\r\n* `handlers.page` is fired after a page's last result, on zero results, or if the HTML could not be retrieved.\r\n\r\n* `.clearCache()` will remove any cached URL responses. This is only relevant if the `cacheResponses` option is enabled.\r\n* `.dequeue(id)` removes a page from the queue. Returns `true` on success or an `Error` on failure.\r\n* `.enqueue(pageUrl, customData)` adds a page to the queue. Queue items are auto-dequeued when their requests are complete. Returns a queue ID on success or an `Error` on failure.\r\n  * `customData` is optional data that is stored in the queue item for the page.\r\n* `.numActiveLinks()` returns the number of links with active requests.\r\n* `.numPages()` returns the total number of pages in the queue.\r\n* `.numQueuedLinks()` returns the number of links that currently have no active requests.\r\n* `.pause()` will pause the queue, but will not pause any active requests.\r\n* `.resume()` will resume the queue.\r\n\r\n```js\r\nvar htmlUrlChecker = new blc.HtmlUrlChecker(options, {\r\n\thtml: function(tree, robots, response, pageUrl, customData){},\r\n\tjunk: function(result, customData){},\r\n\tlink: function(result, customData){},\r\n\tpage: function(error, pageUrl, customData){},\r\n\tend: function(){}\r\n});\r\n\r\nhtmlUrlChecker.enqueue(pageUrl, customData);\r\n```\r\n\r\n### `blc.SiteChecker(options, handlers)`\r\nRecursively scans (crawls) the HTML content at each queued URL to find broken links.\r\n\r\n* `handlers.end` is fired when the end of the queue has been reached.\r\n* `handlers.html` is fired after a page's HTML document has been fully parsed.\r\n  * `tree` is supplied by [parse5](https://npmjs.com/parse5).\r\n  * `robots` is an instance of [robot-directives](https://npmjs.com/robot-directives) containing any `<meta>` and `X-Robots-Tag` robot exclusions.\r\n* `handlers.junk` is fired with data on each skipped link, as configured in options.\r\n* `handlers.link` is fired with the result of each discovered link (broken or not) within the current page.\r\n* `handlers.page` is fired after a page's last result, on zero results, or if the HTML could not be retrieved.\r\n* `handlers.robots` is fired after a site's robots.txt has been downloaded and provides an instance of [robots-txt-guard](https://npmjs.com/robots-txt-guard).\r\n* `handlers.site` is fired after a site's last result, on zero results, or if the *initial* HTML could not be retrieved.\r\n\r\n* `.clearCache()` will remove any cached URL responses. This is only relevant if the `cacheResponses` option is enabled.\r\n* `.dequeue(id)` removes a site from the queue. Returns `true` on success or an `Error` on failure.\r\n* `.enqueue(siteUrl, customData)` adds [the first page of] a site to the queue. Queue items are auto-dequeued when their requests are complete. Returns a queue ID on success or an `Error` on failure.\r\n  * `customData` is optional data that is stored in the queue item for the site.\r\n* `.numActiveLinks()` returns the number of links with active requests.\r\n* `.numPages()` returns the total number of pages in the queue.\r\n* `.numQueuedLinks()` returns the number of links that currently have no active requests.\r\n* `.numSites()` returns the total number of sites in the queue.\r\n* `.pause()` will pause the queue, but will not pause any active requests.\r\n* `.resume()` will resume the queue.\r\n\r\n**Note:** `options.filterLevel` is used for determining which links are recursive.\r\n\r\n```js\r\nvar siteChecker = new blc.SiteChecker(options, {\r\n\trobots: function(robots, customData){},\r\n\thtml: function(tree, robots, response, pageUrl, customData){},\r\n\tjunk: function(result, customData){},\r\n\tlink: function(result, customData){},\r\n\tpage: function(error, pageUrl, customData){},\r\n\tsite: function(error, siteUrl, customData){},\r\n\tend: function(){}\r\n});\r\n\r\nsiteChecker.enqueue(siteUrl, customData);\r\n```\r\n\r\n### `blc.UrlChecker(options, handlers)`\r\nRequests each queued URL to determine if they are broken.\r\n\r\n* `handlers.end` is fired when the end of the queue has been reached.\r\n* `handlers.link` is fired for each result (broken or not).\r\n\r\n* `.clearCache()` will remove any cached URL responses. This is only relevant if the `cacheResponses` option is enabled.\r\n* `.dequeue(id)` removes a URL from the queue. Returns `true` on success or an `Error` on failure.\r\n* `.enqueue(url, baseUrl, customData)` adds a URL to the queue. Queue items are auto-dequeued when their requests are completed. Returns a queue ID on success or an `Error` on failure.\r\n  * `baseUrl` is the address to which all relative URLs will be made absolute. Without a value, links to relative URLs will output an \"Invalid URL\" error.\r\n  * `customData` is optional data that is stored in the queue item for the URL.\r\n* `.numActiveLinks()` returns the number of links with active requests.\r\n* `.numQueuedLinks()` returns the number of links that currently have no active requests.\r\n* `.pause()` will pause the queue, but will not pause any active requests.\r\n* `.resume()` will resume the queue.\r\n\r\n```js\r\nvar urlChecker = new blc.UrlChecker(options, {\r\n\tlink: function(result, customData){},\r\n\tend: function(){}\r\n});\r\n\r\nurlChecker.enqueue(url, baseUrl, customData);\r\n```\r\n\r\n## Options\r\n\r\n### `options.acceptedSchemes`\r\nType: `Array`  \r\nDefault value: `[\"http\",\"https\"]`  \r\nWill only check links with schemes/protocols mentioned in this list. Any others (except those in `excludedSchemes`) will output an \"Invalid URL\" error.\r\n\r\n### `options.cacheExpiryTime`\r\nType: `Number`  \r\nDefault Value: `3600000` (1 hour)  \r\nThe number of milliseconds in which a cached response should be considered valid. This is only relevant if the `cacheResponses` option is enabled.\r\n\r\n### `options.cacheResponses`\r\nType: `Boolean`  \r\nDefault Value: `true`  \r\nURL request results will be cached when `true`. This will ensure that each unique URL will only be checked once.\r\n\r\n### `options.excludedKeywords`\r\nType: `Array`  \r\nDefault value: `[]`  \r\nWill not check or output links that match the keywords and glob patterns in this list. The only wildcard supported is `*`.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.excludedSchemes`\r\nType: `Array`  \r\nDefault value: `[\"data\",\"geo\",\"javascript\",\"mailto\",\"sms\",\"tel\"]`  \r\nWill not check or output links with schemes/protocols mentioned in this list. This avoids the output of \"Invalid URL\" errors with links that cannot be checked.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.excludeExternalLinks`\r\nType: `Boolean`  \r\nDefault value: `false`  \r\nWill not check or output external links when `true`; relative links with a remote `<base>` included.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.excludeInternalLinks`\r\nType: `Boolean`  \r\nDefault value: `false`  \r\nWill not check or output internal links when `true`.\r\n\r\nThis option does *not* apply to `UrlChecker` nor `SiteChecker`'s *crawler*.\r\n\r\n### `options.excludeLinksToSamePage`\r\nType: `Boolean`  \r\nDefault value: `true`  \r\nWill not check or output links to the same page; relative and absolute fragments/hashes included.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.filterLevel`\r\nType: `Number`  \r\nDefault value: `1`  \r\nThe tags and attributes that are considered links for checking, split into the following levels:\r\n* `0`: clickable links\r\n* `1`: clickable links, media, iframes, meta refreshes\r\n* `2`: clickable links, media, iframes, meta refreshes, stylesheets, scripts, forms\r\n* `3`: clickable links, media, iframes, meta refreshes, stylesheets, scripts, forms, metadata\r\n\r\nRecursive links have a slightly different filter subset. To see the exact breakdown of both, check out the [tag map](https://github.com/stevenvachon/broken-link-checker/blob/master/lib/internal/tags.js). `<base>` is not listed because it is not a link, though it is always parsed.\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.honorRobotExclusions`\r\nType: `Boolean`  \r\nDefault value: `true`  \r\nWill not scan pages that search engine crawlers would not follow. Such will have been specified with any of the following:\r\n* `<a rel=\"nofollow\" href=\"…\">`\r\n* `<area rel=\"nofollow\" href=\"…\">`\r\n* `<meta name=\"robots\" content=\"noindex,nofollow,…\">`\r\n* `<meta name=\"googlebot\" content=\"noindex,nofollow,…\">`\r\n* `<meta name=\"robots\" content=\"unavailable_after: …\">`\r\n* `X-Robots-Tag: noindex,nofollow,…`\r\n* `X-Robots-Tag: googlebot: noindex,nofollow,…`\r\n* `X-Robots-Tag: otherbot: noindex,nofollow,…`\r\n* `X-Robots-Tag: unavailable_after: …`\r\n* robots.txt\r\n\r\nThis option does *not* apply to `UrlChecker`.\r\n\r\n### `options.maxSockets`\r\nType: `Number`  \r\nDefault value: `Infinity`  \r\nThe maximum number of links to check at any given time.\r\n\r\n### `options.maxSocketsPerHost`\r\nType: `Number`  \r\nDefault value: `1`  \r\nThe maximum number of links per host/port to check at any given time. This avoids overloading a single target host with too many concurrent requests. This will not limit concurrent requests to other hosts.\r\n\r\n### `options.rateLimit`\r\nType: `Number`  \r\nDefault value: `0`  \r\nThe number of milliseconds to wait before each request.\r\n\r\n### `options.requestMethod`\r\nType: `String`  \r\nDefault value: `\"head\"`  \r\nThe HTTP request method used in checking links. If you experience problems, try using `\"get\"`, however `options.retry405Head` should have you covered.\r\n\r\n### `options.retry405Head`\r\nType: `Boolean`  \r\nDefault value: `true`  \r\nSome servers do not respond correctly to a `\"head\"` request method. When `true`, a link resulting in an HTTP 405 \"Method Not Allowed\" error will be re-requested using a `\"get\"` method before deciding that it is broken.\r\n\r\n### `options.userAgent`\r\nType: `String`  \r\nDefault value: `\"broken-link-checker/0.7.0 Node.js/5.5.0 (OS X El Capitan; x64)\"` (or similar)  \r\nThe HTTP user-agent to use when checking links as well as retrieving pages and robot exclusions.\r\n\r\n\r\n## Handling Broken/Excluded Links\r\nA broken link will have a `broken` value of `true` and a reason code defined in `brokenReason`. A link that was not checked (emitted as `\"junk\"`) will have an `excluded` value of `true` and a reason code defined in `excludedReason`.\r\n```js\r\nif (result.broken) {\r\n\tconsole.log(result.brokenReason);\r\n\t//=> HTTP_404\r\n} else if (result.excluded) {\r\n\tconsole.log(result.excludedReason);\r\n\t//=> BLC_ROBOTS\r\n}\r\n```\r\n\r\nAdditionally, more descriptive messages are available for each reason code:\r\n```js\r\nconsole.log(blc.BLC_ROBOTS);       //=> Robots Exclusion\r\nconsole.log(blc.ERRNO_ECONNRESET); //=> connection reset by peer (ECONNRESET)\r\nconsole.log(blc.HTTP_404);         //=> Not Found (404)\r\n\r\n// List all\r\nconsole.log(blc);\r\n```\r\n\r\nPutting it all together:\r\n```js\r\nif (result.broken) {\r\n\tconsole.log(blc[result.brokenReason]);\r\n} else if (result.excluded) {\r\n\tconsole.log(blc[result.excludedReason]);\r\n}\r\n```\r\n\r\n## HTML and HTTP information\r\nDetailed information for each link result is provided. Check out the [schema](https://github.com/stevenvachon/broken-link-checker/blob/master/lib/internal/linkObj.js#L16-L64) or:\r\n```js\r\nconsole.log(result);\r\n```\r\n\r\n\r\n## Roadmap Features\r\n* fix issue where same-page links are not excluded when cache is enabled, despite `excludeLinksToSamePage===true`\r\n* publicize filter handlers\r\n* add cheerio support by using parse5's htmlparser2 tree adaptor?\r\n* add `rejectUnauthorized:false` option to avoid `UNABLE_TO_VERIFY_LEAF_SIGNATURE`\r\n* load sitemap.xml at end of each `SiteChecker` site to possibly check pages that were not linked to\r\n* remove `options.excludedSchemes` and handle schemes not in `options.acceptedSchemes` as junk?\r\n* change order of checking to: tcp error, 4xx code (broken), 5xx code (undetermined), 200\r\n* abort download of body when `options.retry405Head===true`\r\n* option to retry broken links a number of times (default=0)\r\n* option to scrape `response.body` for erroneous sounding text (using [fathom](https://npmjs.com/fathom-web)?), since an error page could be presented but still have code 200\r\n* option to check broken link on archive.org for archived version (using [this lib](https://npmjs.com/archive.org))\r\n* option to run `HtmlUrlChecker` checks on page load (using [jsdom](https://npmjs.com/jsdom)) to include links added with JavaScript?\r\n* option to check if hashes exist in target URL document?\r\n* option to parse Markdown in `HtmlChecker` for links\r\n* option to play sound when broken link is found\r\n* option to hide unbroken links\r\n* option to check plain text URLs\r\n* add throttle profiles (0–9, -1 for \"custom\") for easy configuring\r\n* check [ftp:](https://nmjs.com/ftp), [sftp:](https://npmjs.com/ssh2) (for downloadable files)\r\n* check ~~mailto:~~, news:, nntp:, telnet:?\r\n* check local files if URL is relative and has no base URL?\r\n* cli json mode -- streamed or not?\r\n* cli non-tty mode -- change nesting ASCII artwork to time stamps?\r\n\r\n\r\n[npm-image]: https://img.shields.io/npm/v/broken-link-checker.svg\r\n[npm-url]: https://npmjs.org/package/broken-link-checker\r\n[travis-image]: https://img.shields.io/travis/stevenvachon/broken-link-checker.svg\r\n[travis-url]: https://travis-ci.org/stevenvachon/broken-link-checker\r\n[david-image]: https://img.shields.io/david/stevenvachon/broken-link-checker.svg\r\n[david-url]: https://david-dm.org/stevenvachon/broken-link-checker\r\n","readmeFilename":"README.md"}