testMaven/src/com/neusoft/utils/MyCrawler.java

(开头部分) 6KB

这里只显示每个文件的开头 60 行。登录后可以解锁完整代码。

package com.neusoft.utils;


import java.util.List;
import java.util.Set;
import java.util.regex.Pattern;

import javax.enterprise.inject.New;
import javax.servlet.ServletContext;

import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.jsoup.select.Elements;
import org.springframework.beans.BeansException;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Component;
import org.springframework.web.context.ContextLoader;
import org.springframework.web.context.WebApplicationContext;
import org.springframework.web.context.support.WebApplicationContextUtils;

import com.neusoft.po.News;
import com.neusoft.service.NewsService;
import com.neusoft.service.impl.NewsServiceImpl;

import edu.uci.ics.crawler4j.crawler.Page;
import edu.uci.ics.crawler4j.crawler.WebCrawler;
import edu.uci.ics.crawler4j.parser.HtmlParseData;
import edu.uci.ics.crawler4j.url.WebURL;
import net.sf.cglib.core.CollectionUtils;

/**
 * 自定义爬虫类需要继承WebCrawler类,决定哪些url可以被爬以及处理爬取的页面信息
 * @author 
 *
 */

@Component
public class MyCrawler extends WebCrawler {

	@Autowired
	private NewsService newsService;
	/**
	 * 正则匹配指定的后缀文件
	 */
    private final static Pattern FILTERS = Pattern.compile(".*(\\.(css|js|gif|jpg"
                                                           + "|png|mp3|mp3|zip|gz))$");

     /**
      * 这个方法主要是决定哪些url我们需要抓取,返回true表示是我们需要的,返回false表示不是我们需要的Url
      * 第一个参数referringPage封装了当前爬取的页面信息
      * 第二个参数url封装了当前爬取的页面url信息
      */
     @Override
     public boolean shouldVisit(Page referringPage, WebURL url) {
         String href = url.getURL().toLowerCase();  // 得到小写的url
         return !FILTERS.matcher(href).matches()   // 正则匹配,过滤掉我们不需要的后缀文件
                && href.startsWith("http://www.sohu.com/a/");  // url必须是开头,规定站点
     }

     /**
后面还有 91 行代码,购买后查看完整代码

24 小时内免费解锁 3 个项目,之后 1 积分/个。 规则说明

AI 解读

登录后可用,每次 10 积分,解读结果公开显示在下面。

还没有人解读过这个文件。