testMaven/src/com/neusoft/utils/MyCrawler.java
(开头部分) 6KB这里只显示每个文件的开头 60 行。登录后可以解锁完整代码。
package com.neusoft.utils;
import java.util.List;
import java.util.Set;
import java.util.regex.Pattern;
import javax.enterprise.inject.New;
import javax.servlet.ServletContext;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.jsoup.select.Elements;
import org.springframework.beans.BeansException;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Component;
import org.springframework.web.context.ContextLoader;
import org.springframework.web.context.WebApplicationContext;
import org.springframework.web.context.support.WebApplicationContextUtils;
import com.neusoft.po.News;
import com.neusoft.service.NewsService;
import com.neusoft.service.impl.NewsServiceImpl;
import edu.uci.ics.crawler4j.crawler.Page;
import edu.uci.ics.crawler4j.crawler.WebCrawler;
import edu.uci.ics.crawler4j.parser.HtmlParseData;
import edu.uci.ics.crawler4j.url.WebURL;
import net.sf.cglib.core.CollectionUtils;
/**
* 自定义爬虫类需要继承WebCrawler类,决定哪些url可以被爬以及处理爬取的页面信息
* @author
*
*/
@Component
public class MyCrawler extends WebCrawler {
@Autowired
private NewsService newsService;
/**
* 正则匹配指定的后缀文件
*/
private final static Pattern FILTERS = Pattern.compile(".*(\\.(css|js|gif|jpg"
+ "|png|mp3|mp3|zip|gz))$");
/**
* 这个方法主要是决定哪些url我们需要抓取,返回true表示是我们需要的,返回false表示不是我们需要的Url
* 第一个参数referringPage封装了当前爬取的页面信息
* 第二个参数url封装了当前爬取的页面url信息
*/
@Override
public boolean shouldVisit(Page referringPage, WebURL url) {
String href = url.getURL().toLowerCase(); // 得到小写的url
return !FILTERS.matcher(href).matches() // 正则匹配,过滤掉我们不需要的后缀文件
&& href.startsWith("http://www.sohu.com/a/"); // url必须是开头,规定站点
}
/**
后面还有 91 行代码,购买后查看完整代码
24 小时内免费解锁 3 个项目,之后 1 积分/个。 规则说明
AI 解读
登录后可用,每次 10 积分,解读结果公开显示在下面。
还没有人解读过这个文件。
