-
Notifications
You must be signed in to change notification settings - Fork 0
/
scrapper.js
97 lines (77 loc) · 2.61 KB
/
scrapper.js
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
var request = require('request');
var cheerio = require('cheerio');
var URL = require('url-parse');
var pageRoot = "https://joopic.joobot.com";
var pageToVisit = "https://joopic.joobot.com/album/381573271650305034";
console.log("Visiting page " + pageToVisit);
var SEARCH_WORD = "";
// https://joopic.joobot.com/resource/objects/mid/381577049053070346/381573271650305034
var lastTime = new Date().getTime();
var dbUrl = ""
// var dbUrl = "https://api.mlab.com/api/1/databases/prudence/collections/photos?apiKey=5q2rhWYnQ5a2YX4HF89CyyrGO6f109Gk";
function addtoDatabase(imgUrl) {
var options = {
url: dbUrl,
headers: { 'Content-type': "application/json" },
json: { "url": imgUrl, "time": lastTime, }
}
console.log("POST sent.");
request.post(options, function(error, response, body) {
if (error || response.statusCode != 200) {
console.log(error);
console.log(response.statusCode);
return
}
console.log("Added to Database");
// console.log(body)
});
}
var imgArray = new Array();
function scrap() {
request(pageToVisit, function(error, response, body) {
if (error) {
console.log("Error: " + error);
}
// Check status code (200 is HTTP OK)
console.log("Status code: " + response.statusCode);
if (response.statusCode === 200) {
// Parse the document body
var $ = cheerio.load(body);
lastTime = new Date().getTime();
var figures = $('figure')
var counter = 0;
figures.each(function(i, element) {
var a = $(this).children()[0]
var imgUrl = pageRoot + a.attribs.href;
if (!imgArray.includes(imgUrl)) {
imgArray.push(imgUrl);
addtoDatabase(imgUrl);
counter +=1;
}
});
console.log("Scrapped: " + figures.length +"; Inserted: " + counter);
}
});
}
function listDatabase() {
request(dbUrl, function(error, response, body) {
if (error) {
console.log("Error: " + error);
}
body = JSON.parse(body)
for(let i = 0;i<body.length; i++){
imgArray.push(body[i].url)
}
console.log("Initialized at " + imgArray.length)
});
}
var promise = new Promise(function(resolve, reject) {
listDatabase()
resolve("");
});
//For every second, scrap web
//If img not in imgArray, add to database and array
promise.then(function(result) {
console.log(result);
setInterval(scrap, 2000);
})