标签:爬取 except mail main htm puts 汉字 cep .com
import java.io.BufferedReader;
import java.io.FileReader;
import java.io.InputStreamReader;
import java.net.URL;
import java.net.URLConnection;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
public class GetMail {
public static void main(String[] args) throws Exception {
//getMails();
getMails_url();
}
public static void getMails_url() throws Exception {
URL url = new URL("https://wenku.baidu.com/view/ce81b0a1ddccda38366baf61.html");//这里就是要爬取的网页
URLConnection conn = url.openConnection();
BufferedReader bufr = new BufferedReader(new InputStreamReader(conn.getInputStream()));
String line = null;
String maileRes = "[\u4E00-\u9FA5]+";//这里存放需要设定的规则
//匹配邮箱:"\\w+@\\w+(\\.\\w+)+"
//匹配汉字:"[\u4E00-\u9FA5]+";
//匹配QQ号:"[1-9][0-9]{4,14}"
//qq邮箱:"(.)+@(.)+(\\.[a-z]+){1,}";
Pattern p = Pattern.compile(maileRes);
while((line=bufr.readLine())!=null) {
Matcher m = p.matcher(line);
while(m.find()) {
System.out.println(m.group());
}
}
}
标签:爬取 except mail main htm puts 汉字 cep .com
原文地址:https://www.cnblogs.com/zxwm/p/9235960.html