从一段文字中提取出uri信息

时间：2017-02-20 13:12:28 阅读：241 评论：0 收藏：0 [点我收藏+]

package handle.groupby;

import java.io.BufferedReader;
import java.io.BufferedWriter;
import java.io.FileNotFoundException;
import java.io.FileReader;
import java.io.FileWriter;
import java.io.IOException;
import java.io.Reader;
import java.io.Writer;
import java.util.regex.Matcher;
import java.util.regex.Pattern;

import org.mockito.asm.tree.IntInsnNode;

public class GetUrlFromString {

    @SuppressWarnings("resource")
    public static void main(String[] args) throws IOException {
        String line="";
        Pattern pattern = Pattern.compile("([\\w.]{1}[\\w\\/.\\s]*[\\w]{1})",Pattern.CASE_INSENSITIVE);
        
        BufferedReader r= new BufferedReader(new FileReader(args[0]));
        BufferedWriter w=new BufferedWriter(new FileWriter(args[1])) ;
        while ((line=r.readLine())!=null) {
            String source = line;
             Matcher matcher = pattern.matcher(source);
                while(matcher.find()){
//                    System.out.println(matcher.group(matcher.groupCount()));
                    String url=matcher.group(matcher.groupCount());
                    if (url.contains(".")) {
                        String resUrl="";
                        String resUrl2="";
                        if (url.contains("/")) {
　　　　　　　　　　　　　　　　//这个判断是为了提取出短域名的网站级访问访问信息，不需要可以删掉。
　　　　　　　　　　　　　　　　//例如从：汉字汉字汉字t.cn/RVIIIj8汉字汉字 中提取出 t.cn/RVIIIj8而不是t.cn
                            int i =url.lastIndexOf("/");
                            int i2 =url.indexOf("/");
                            if (i==i2) {
                                resUrl=url;
                            }else {
                                resUrl =url.split("/")[0];
                            }
                        }else {
                            resUrl=url;
                        }
                        //去空格
                        resUrl= resUrl.replaceAll(" ", "");
                        w.write(source+"|"+resUrl);
                        w.write("\r\n");
                    }
                }
        }
        r.close();
        w.flush();
        w.close();
        System.out.println("执行完毕");
    }

}

从一段文字中提取出uri信息

原文：http://www.cnblogs.com/yanghaolie/p/6418779.html

踩

(0)

评论一句话评论（0）

分享档案

更多>

2021年09月23日 (328)
2021年09月24日 (313)
2021年09月17日 (191)
2021年09月15日 (369)
2021年09月16日 (411)
2021年09月13日 (439)
2021年09月11日 (398)
2021年09月12日 (393)
2021年09月10日 (160)
2021年09月08日 (222)