-
Notifications
You must be signed in to change notification settings - Fork 1.4k
Expand file tree
/
Copy pathDemoAnnotatedAutoNewsCrawler.java
More file actions
67 lines (55 loc) · 2.45 KB
/
Copy pathDemoAnnotatedAutoNewsCrawler.java
File metadata and controls
67 lines (55 loc) · 2.45 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
package cn.edu.hfut.dmic.webcollector.example;
import cn.edu.hfut.dmic.webcollector.model.CrawlDatums;
import cn.edu.hfut.dmic.webcollector.model.Page;
import cn.edu.hfut.dmic.webcollector.plugin.rocks.BreadthCrawler;
/**
* Crawling news from github news
*
* @author hu
*/
public class DemoAnnotatedAutoNewsCrawler extends BreadthCrawler {
/**
* @param crawlPath crawlPath is the path of the directory which maintains
* information of this crawler
* @param autoParse if autoParse is true,BreadthCrawler will auto extract
* links which match regex rules from pag
*/
public DemoAnnotatedAutoNewsCrawler(String crawlPath, boolean autoParse) {
super(crawlPath, autoParse);
/*start pages*/
this.addSeed("https://github.blog/");
/*fetch url like "https://blog.github.com/2018-07-13-graphql-for-octokit/" */
this.addRegex("https://github.blog/[0-9]{4}-[0-9]{2}-[0-9]{2}-[^/]+/");
/*do not fetch jpg|png|gif*/
//this.addRegex("-.*\\.(jpg|png|gif).*");
/*do not fetch url contains #*/
//this.addRegex("-.*#.*");
setThreads(50);
getConf().setTopN(100);
//enable resumable mode
//setResumable(true);
}
@MatchUrl(urlRegex = "https://github.blog/[0-9]{4}-[0-9]{2}-[0-9]{2}[^/]+/")
public void visitNews(Page page, CrawlDatums next) {
/*extract title and content of news by css selector*/
String title = page.select("h1.lh-condensed").first().text();
String content = page.selectText("main[id^='post']");
System.out.println("URL:\n" + page.url());
System.out.println("title:\n" + title);
System.out.println("content:\n" + content);
/*If you want to add urls to crawl,add them to nextLink*/
/*WebCollector automatically filters links that have been fetched before*/
/*If autoParse is true and the link you add to nextLinks does not match the
regex rules,the link will also been filtered.*/
//next.add("http://xxxxxx.com");
}
@Override
public void visit(Page page, CrawlDatums next) {
System.out.println("visit pages that don't match any annotation rules: " + page.url());
}
public static void main(String[] args) throws Exception {
DemoAnnotatedAutoNewsCrawler crawler = new DemoAnnotatedAutoNewsCrawler("crawl", true);
/*start crawl with depth of 4*/
crawler.start(4);
}
}