{"_id":"@bpw1621/imgscrape","_rev":"4-21685232e08e2e727a24a2a0ca084b1f","name":"@bpw1621/imgscrape","dist-tags":{"latest":"0.0.3"},"versions":{"0.0.1":{"name":"@bpw1621/imgscrape","version":"0.0.1","description":"Puppeteer-based image search engine scraper.","main":"./cli/imgscrape-cli.js","bin":{"imgscrape-cli":"cli/imgscrape-cli.js"},"dependencies":{"@google-cloud/translate":"^5.3.0","dotenv":"^8.2.0","image-downloader":"^3.5.0","image-type":"^4.1.0","progress":"^2.0.3","puppeteer":"^2.1.1","yargs":"^15.3.1"},"devDependencies":{"jest":"^25.1.0"},"gitHead":"c87b598403767b43d2a194eccd4a2624860fb195","_id":"@bpw1621/imgscrape@0.0.1","_nodeVersion":"16.2.0","_npmVersion":"7.13.0","dist":{"integrity":"sha512-pXgzyzzOiJUib5rV47KGI7b5I/YZZdEBMo/KKz/C4sMOwCyXQZsT8IopEWiI8+/ty1DKuRuTOlm6sjNMKegDWw==","shasum":"54bd3d6ce6463c6313caa2b9e05b63a31f53555e","tarball":"https://registry.npmjs.org/@bpw1621/imgscrape/-/imgscrape-0.0.1.tgz","fileCount":10,"unpackedSize":17298,"npm-signature":"-----BEGIN PGP SIGNATURE-----\r\nVersion: OpenPGP.js v3.0.13\r\nComment: https://openpgpjs.org\r\n\r\nwsFcBAEBCAAQBQJguDOdCRA9TVsSAnZWagAAVmUP+wTXXzRRaXpEwRdwocPq\n1Nj8msrbyF/Bm3frKqCzVlfOy5YekLHBlqH/QTzP7HWBkFxnVUMVnDB5l5e1\nnwwyWIHXCN0s9dWQEgxSUwxMwcsy5ujaZ0zLLFEkWDVg++zW7MNCDoK9timv\n/3/N8walyiD0fgW53xqWby8IQIfqiFgQ5MtOQXVlxbJckxXL899RwwfXm+vK\nOoswhlrhTVDUanO2lWdKPqmkXdmu2PAmg6G48+uHb3SIkHkQq3si+fq66bb+\nn3MRtdKOs53lJKSPmYlXqt9eGEWq+MIvTZlNWKSiNSO/h96/fQZ5svj4lDx1\nMOa7l8IWuLGYnXgqI3matSIoUIDK9KT+Zo4m4vYV9eR1qFACM3dxe4rW8ht/\n2m3iQhrMaTarUJ1m6Am3VgXPtfQUAyJ1rjomIbaIU5w1urC9F7UR8h/1YVfH\n+CrF3r2wegC3h4I33KIc+FdzauhsQOjWy2hAibw2eKhDHhjZ1zeCWN7h6ZEm\nHkYKrclqiDyK8YRpvmy0EKqvbvcFoOr4LUVy7glbXYA+vn+E1gol0pHTeXVW\nLIPxaX0ysvyfMEJ54JZOi2OBGRsKANRXzRbjEaebdGDaf9RTaqwrB7w9QgDJ\n0QxNAH7tq4SxlLSenbSjUz2FFt3Y3VVyucd6KGgabJS14BuUxFl8f9a2M8Nx\nj8D1\r\n=SrjR\r\n-----END PGP SIGNATURE-----\r\n","signatures":[{"keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA","sig":"MEYCIQC3r5WNmzH7vmLmMlldE12eSxBF45HQ83YIl9cGk0/c5gIhAIwgWERKBGVWjArN29aOQoc0Kwqa231yW0f8iaQQt1Mh"}]},"_npmUser":{"name":"bpw1621","email":"bpw1621+npm@gmail.com"},"directories":{},"maintainers":[{"name":"bpw1621","email":"bpw1621+npm@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages","tmp":"tmp/imgscrape_0.0.1_1622684573735_0.14810075121098643"},"_hasShrinkwrap":false},"0.0.0":{"name":"@bpw1621/imgscrape","version":"0.0.0","description":"Puppeteer-based image search engine scraper.","repository":{"type":"git","url":"git+https://github.com/bpw1621/imgscrape.git"},"main":"./cli/imgscrape-cli.js","bin":{"imgscrape-cli":"cli/imgscrape-cli.js"},"dependencies":{"@google-cloud/translate":"^5.3.0","dotenv":"^8.2.0","image-downloader":"^3.5.0","image-type":"^4.1.0","progress":"^2.0.3","puppeteer":"^2.1.1","yargs":"^15.3.1"},"devDependencies":{"jest":"^25.1.0"},"gitHead":"b7c3dbe78054ff959d493673ee34e35cf201f1ab","bugs":{"url":"https://github.com/bpw1621/imgscrape/issues"},"homepage":"https://github.com/bpw1621/imgscrape#readme","_id":"@bpw1621/imgscrape@0.0.0","_nodeVersion":"16.2.0","_npmVersion":"7.13.0","dist":{"integrity":"sha512-yZ+YPSVmhuY8c2hD5CnuIsk+er1EFpebd8BuGCjAC75335U+MDiuIy+Cq9HLyc6t9T8VsoxTLZFJb4lMILWjkQ==","shasum":"9d2e520fa746d83e12e33ddeb576832a623951c8","tarball":"https://registry.npmjs.org/@bpw1621/imgscrape/-/imgscrape-0.0.0.tgz","fileCount":10,"unpackedSize":17394,"npm-signature":"-----BEGIN PGP SIGNATURE-----\r\nVersion: OpenPGP.js v3.0.13\r\nComment: https://openpgpjs.org\r\n\r\nwsFcBAEBCAAQBQJguDxuCRA9TVsSAnZWagAAZR8P/jtLY06611+y1Q7OC/Fh\ndtcAcXDBSqwuCVBd8bRHq+1SCTuZ2jUMa6Uh5e1doZlA9+a9SUZ9PuCyJiVs\nBPlDRUqXSnT0qy9khL5mrmExoIeyMmLSFHV+ish0Dac/G5W/5VyPlsEDcSAI\nht38CedpVE7CdwLNKmbpeZJQOqu9TD9zen01oumJFQOk7bBgJTubPppO8dd+\nxm8NSxbMIoTJ6U+2xryhOud/qrraJbGW1JQIXl2yyQwb4F5Zz43KK5RF/+ll\nRpwfzf1FSii9cErJ5idgj2DG9+xIMei6DR36U/UWODuGhpd1t3UrJ+sKh9YB\nVfysIgEPXktKWDgrQjMo5u6p/C1891VgL2GLwDWTlOI0uHTGOh8ixCWgxZtP\nweNUsLukeglN9iBbv13/EVXEIM7QMyi8Ur5ySdaUv+9faDQKPcS0K2VhTQ1g\nh+ne11+g0CtZzidcfNxqOdEZcfZUFH6Be3ndZGXK4tpTAh84PRadzuR84P9B\n/R5xsca5YbFdgL/Hmyf4vn9X9roni9CPcWYV6wuyhN4hYsGqsIopVSCDoPck\n8Yn3bar/GxPIdTzJWlzXLz9oDgMZoCUp9Ko4M/9Qvmkf7tPgfqnTGI0ykUH1\npkNEyeOx+YwdDW0o/XQ5dE81LUsH9CIvlX+G8/Y1fRO98JIxNOlq28gdF9z+\nTSzE\r\n=mjva\r\n-----END PGP SIGNATURE-----\r\n","signatures":[{"keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA","sig":"MEYCIQDZSYC6MRImCJXU+VOhyBrLrZTSXT/rpILBbWgGqu7adwIhAJKd+6prw7UcU3qR60Z01HD+2j5Qry4LhG35n4L/J3XL"}]},"_npmUser":{"name":"bpw1621","email":"bpw1621+npm@gmail.com"},"directories":{},"maintainers":[{"name":"bpw1621","email":"bpw1621+npm@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages","tmp":"tmp/imgscrape_0.0.0_1622686830127_0.5477666575021805"},"_hasShrinkwrap":false},"0.0.2":{"name":"@bpw1621/imgscrape","version":"0.0.2","description":"Puppeteer-based image search engine scraper.","repository":{"type":"git","url":"git+https://github.com/bpw1621/imgscrape.git"},"main":"./cli/imgscrape-cli.js","bin":{"imgscrape-cli":"cli/imgscrape-cli.js"},"dependencies":{"@google-cloud/translate":"^5.3.0","dotenv":"^8.2.0","image-downloader":"^3.5.0","image-type":"^4.1.0","progress":"^2.0.3","puppeteer":"^2.1.1","yargs":"^15.3.1"},"devDependencies":{"jest":"^25.1.0"},"gitHead":"8e7d29ebc65bda61b01a7843e89bbce107f821c7","bugs":{"url":"https://github.com/bpw1621/imgscrape/issues"},"homepage":"https://github.com/bpw1621/imgscrape#readme","_id":"@bpw1621/imgscrape@0.0.2","_nodeVersion":"16.2.0","_npmVersion":"7.13.0","dist":{"integrity":"sha512-KnxKkRtyGxwXcFM+KGbcpn4BW0UWBtRX0XJFFqEQDcOxOeeiZon+/7jyrFmUfcigrIXCJKcu8S0kIAwn2/ZB7Q==","shasum":"d5d76a8405b59b860b3a51fe5482703753e30eec","tarball":"https://registry.npmjs.org/@bpw1621/imgscrape/-/imgscrape-0.0.2.tgz","fileCount":10,"unpackedSize":17644,"npm-signature":"-----BEGIN PGP SIGNATURE-----\r\nVersion: OpenPGP.js v3.0.13\r\nComment: https://openpgpjs.org\r\n\r\nwsFcBAEBCAAQBQJgu9ObCRA9TVsSAnZWagAAKzEP/jKR+v73hLhTR52+TTCN\nGOVMHkqkzQArKDBlmDJtUEyL6jVYU7Jd+3O5+2Ojn6HGhsIeO10Q1wX0BjOK\n5p/I9M9b4LqUlwtUBPCpBxN8XWW0Wj6hqW07Dwz3K3nKEdv16+DPQQrhler4\n/4c0PeZLvkMb+h9rySUMWUWqAiswwTagh+nLZgBkL2CkeiMGiUkSTlbz7c0E\nfMQCEhfGtVJQk1DdP1/murlycT3sGnuZnc6sxWnwIczt1hp2EnLEjRZaLrpz\nrGehmImO2su8f5XCY80GcmStlvIP4XE/BErHAbviozPJZLWmr+QPKEPhvCFX\n6d2E3Lpqiwk8yBJRmAARygJYejGoJYRs8k8BNGjb3LItpuE6k4Dw8t15Xwyg\n67epN47cB254QHefLUV9EXyGIx1mSkuUecIM9s6SZk7zF/3+gE9hNOSWj67N\nzJmM/INnHuX6Jtbf/1OHwFAjTj7aL7jnK9vsgdgizoLVzNsNVkp5en5EziyK\nq+Doj3rI7DC6Birpgsz8ohgAR0gRbIIuP/JaJ+Z9fXjg1rXb3AD8u5mu0t6I\njV7TJeVkKVLOspDC62cPzgnEsmRd3515Yl80QbUfDjnxXceduDl59g0/UZn1\ny1rmNrtZhVsV9TkX9qh38Nl9eQp8LUGpn6Uwa5FCxo+kIW/0+k4yoUA/eYHP\nYxrz\r\n=9Q3m\r\n-----END PGP SIGNATURE-----\r\n","signatures":[{"keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA","sig":"MEQCIDGXKyFHrysuMak+o9JHyaQJG2D316NKsFkjNR3ajlCzAiAePVw48BKQHdZ3y2YkVaI8cXtOl3Apo5E+JG84Y4u6Vw=="}]},"_npmUser":{"name":"bpw1621","email":"bpw1621+npm@gmail.com"},"directories":{},"maintainers":[{"name":"bpw1621","email":"bpw1621+npm@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages","tmp":"tmp/imgscrape_0.0.2_1622922139303_0.8495867896964613"},"_hasShrinkwrap":false},"0.0.3":{"name":"@bpw1621/imgscrape","version":"0.0.3","description":"Puppeteer-based image search engine scraper.","repository":{"type":"git","url":"git+https://github.com/bpw1621/imgscrape.git"},"keywords":["puppeteer","pupeteer","chrome","headless","chrome-headless","crawler","web-crawler","automation","download","script","scraper","scrape","scrapping","images-scraper","web","url","page","site","html","css","js","image","images","photo","photos","google","duckduckgo","yandex","infospace","bing","flickr","instagram","data","science","mining","machine","learning","data science","data-science","datascience","data mining","data-mining","datamining","machine-learning","machine learning","machinelearning"],"author":{"name":"Dr. Bryan Patrick Wood","email":"bpw1621+imgscrape@gmail.com","url":"https://bpw1621.com/"},"license":"MIT","homepage":"https://github.com/bpw1621/imgscrape","bugs":{"url":"https://github.com/bpw1621/imgscrape/issues"},"main":"./cli/imgscrape-cli.js","bin":{"imgscrape-cli":"cli/imgscrape-cli.js"},"dependencies":{"@google-cloud/translate":"^5.3.0","dotenv":"^8.2.0","image-downloader":"^3.5.0","image-type":"^4.1.0","progress":"^2.0.3","puppeteer":"^2.1.1","yargs":"^15.3.1"},"devDependencies":{"jest":"^25.1.0"},"gitHead":"6f8cb519c5033fbcfc05d9f2318bd47797c9025d","_id":"@bpw1621/imgscrape@0.0.3","_nodeVersion":"16.2.0","_npmVersion":"7.13.0","dist":{"integrity":"sha512-+TPGkeGt5oKSs/Pco1OiS30L68LH6e4TMnDxlMeXVI1wLXu03EqQuIWNyOy2m/MqMCm0BDGdQFNJxWry7CLj2w==","shasum":"b057b88412576688f668ac5003ed36b7353c48dd","tarball":"https://registry.npmjs.org/@bpw1621/imgscrape/-/imgscrape-0.0.3.tgz","fileCount":11,"unpackedSize":19810,"npm-signature":"-----BEGIN PGP SIGNATURE-----\r\nVersion: OpenPGP.js v3.0.13\r\nComment: https://openpgpjs.org\r\n\r\nwsFcBAEBCAAQBQJgu+/HCRA9TVsSAnZWagAA5ooP/REkYVOTILYkSpiqyowp\n5mGPT9ucE+eMGTXa4D10TnSCGhp8FY2vPHN/CMVHI9Gy//wFBOEK5Mffy4/x\nFqDV5rm7tHffROQtzvWDCu7GIQcqwL2eQfy+gZNWFrU88ombona38A2/ZDy2\nKaNfZ+8O653OfC/SLpaw9SofhC1McH1iD3QofYG1fsuUBVxbHnk55GrjEyNK\nJTT4ocF0CVe+DDGaKmbAtJFUlJkCkAX0y2P4o+DS0o45e//Y2ndsenoA7Vzo\nBLIhjbP/HNhhLp7z3119xMzkug9WhGhdvcJ0x4KfRFr1KOzGQuWWJ7VPJXmW\nqP04H7GfmanoCEY+etzpbjrH9NGQ8hpH85uzS/fOGn22tKq1yTX9vk9UMwJI\nqZllk0Y6wh2cBR6pYPIlRt+CQRiNGjV2+kVwRTkg+vnTzlIXezI8hItwzhrp\ne4OmEbvixySHDyvGg5Rj8kPjVFGZOnL+k246X5jnDAGdGAjdTRYmbfDPt1Fh\n3GaCzkPJMpxWx6DjRhqZOqMw/j+SegTVx9rHDF2EM2UZUVA8o7E/Akf7g+Ij\neT5szRGRpc08xMh0OZdaUhcprNQiJZEp1/wrYE8Mlx+QjNJXAOMij4uD+PLD\nezYRIJmLvXl5ugV6isrd6kpt5grSPTPCDdG4DlFUDor/1B9CbvItt00W6cEx\n1qU1\r\n=6gsC\r\n-----END PGP SIGNATURE-----\r\n","signatures":[{"keyid":"SHA256:jl3bwswu80PjjokCgh0o2w5c2U4LhQAE57gj9cz1kzA","sig":"MEUCIQCE9YM05iD66WCp/ADPhOVkDkN23TfPJKZ6jCjnqEfY6wIgd2WY9Kr237V2gP65PORj1+oeipF1jPaf4slXRUbFQwc="}]},"_npmUser":{"name":"bpw1621","email":"bpw1621+npm@gmail.com"},"directories":{},"maintainers":[{"name":"bpw1621","email":"bpw1621+npm@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages","tmp":"tmp/imgscrape_0.0.3_1622929350945_0.32529434047776506"},"_hasShrinkwrap":false}},"time":{"created":"2021-06-03T01:42:53.681Z","0.0.1":"2021-06-03T01:42:53.848Z","modified":"2022-04-04T20:19:17.886Z","0.0.0":"2021-06-03T02:20:30.272Z","0.0.2":"2021-06-05T19:42:19.485Z","0.0.3":"2021-06-05T21:42:31.086Z"},"maintainers":[{"name":"bpw1621","email":"bpw1621+npm@gmail.com"}],"description":"Puppeteer-based image search engine scraper.","readme":"# imgscrape: scrape images from popular internet search engines\r\n\r\nimgscrape is a pretty simple puppeteer based image webscraper with a yargs based CLI interface. It was written quickly\r\nto meet a machine learning project need. As such, it is not the best example of robust extensible software at the\r\nmoment. Node and Javascript are also not the languages I use most on a daily basis, so the code may not be idiomatic\r\nor optimized Node code. I had considered writing this the Python port of puppeteer, pyppeteer, but thought better of it\r\ngiven how simple this ended up being. The main logic is all in the lib/scrapeImages.js file.\r\n\r\nIt supports a few popular search engines that support image search. These can be found in the engine section of the\r\nconfig/data.json file. Currently, that is: duckduckgo, yandex, infospace (not working), bing, google, flickr, instagram.\r\nProvided the overall logic is similar to the other engines it should be relatively easy for folks that are familiar with\r\npuppeteer to add / fix engines in the engine switch portion of the lib/scrapeImages.js file. Adding engines for search\r\nservices that behave fundamentally different from the supported ones (viz., infospace and the source code comment)\r\nis probably not going to be easy without refactoring the code.\r\n\r\nimgscrape support -h / --help options and the main functionality, imgscrape scrape, does as well with a few example\r\nusages. Usage will result in a directory being created to drop the output into. That output consists of the images of\r\ninterest and a few json files that detail the duplicate URLs encountered, the URLs that the scraper failed to download\r\nan image from, and those URLs it successfully downloaded an image from.\r\n\r\nFor those that just want to use this as a tool, because it wasn't clear to me immediately how to just install this\r\nand do that, it's as simple as, for instance, the following\r\n\r\n    npx imgscrape-cli -t narwhal -e google\r\n\r\nThere is logic in there to bail if the engine is not providing additional images on scrolling down the page, and a MD5\r\nhash based check to bail on the downloading of new images if the engine is providing too many consecutive duplicate\r\nimages. There may be better ways to implement this logic, but it worked well enough for what I was trying to do.\r\n\r\n# Caveat Utilitor\r\n\r\nCSS selector based website scraping is brittle. Many of these services seem to change up their use of CSS classes and\r\nids potentially in an effort to break these types of tools or just as a side effect of normal software refactoring.\r\nAs of 4/18/2021, verified working on all engines to some degree (again except infospace).\r\n\r\nDoing this sort of thing may also break ToS if that is a concern check before using.\r\n\r\nThis is using the search engines to find URLs and looks no different from performing a term search for images to the\r\nsearch engine itself. Since the images are dispersed around the internet a rotating proxy is probably not needed;\r\nhowever, it should be easy enough to rough in a proxy since puppeteer supports that natively but since I do not have\r\naccess to one I did not add that myself.\r\n\r\n# Future\r\n\r\nAn image data set collection tip I came across a while back (unfortunately do not remember the attribution) was to\r\ntranslate the terms you were looking for into foreign languages to do additional searches which can provide a\r\nsignificant lift (based on my own empirical observations). To that end integrating Google Translation\r\nservice is planned but not implemented. In the meantime manually adding custom translations to the config/data.json\r\ntranslation block is a workaround.\r\n\r\nThere is also a TODO.md that has a few ideas I jotted down while throwing this together with some ideas for future\r\nimprovements.\r\n","readmeFilename":"README.md","homepage":"https://github.com/bpw1621/imgscrape","repository":{"type":"git","url":"git+https://github.com/bpw1621/imgscrape.git"},"bugs":{"url":"https://github.com/bpw1621/imgscrape/issues"},"keywords":["puppeteer","pupeteer","chrome","headless","chrome-headless","crawler","web-crawler","automation","download","script","scraper","scrape","scrapping","images-scraper","web","url","page","site","html","css","js","image","images","photo","photos","google","duckduckgo","yandex","infospace","bing","flickr","instagram","data","science","mining","machine","learning","data science","data-science","datascience","data mining","data-mining","datamining","machine-learning","machine learning","machinelearning"],"author":{"name":"Dr. Bryan Patrick Wood","email":"bpw1621+imgscrape@gmail.com","url":"https://bpw1621.com/"},"license":"MIT"}