/* * To change this template, choose Tools | Templates * and open the template in the editor. */ package test; import java.util.HashMap; import java.util.Map; import org.htmlparser.Node; import org.htmlparser.NodeFilter; import org.htmlparser.Parser; import org.htmlparser.tags.LinkTag; import org.htmlparser.util.NodeList; /** * * @author Arjick@163.com */ public class GetLinkTest cambridge { public static void main(String[] args) { try { // 通过过滤器过滤出<A>标签 Parser parser = new Parser("http://www.lovezan.com" ); NodeList nodeList = parser.extractAllNodesThatMatch( new NodeFilter() { // 实现该方法,用以过滤标签 public boolean accept(Node node) { if (node instanceof LinkTag) // 标记 { return true ; } return false ; } }); // 打印 for ( int i = 0; i < nodeList.size(); i++ ) { LinkTag n = (LinkTag) nodeList.elementAt(i); // System.out.print(n.getStringText() + " ==>> "); // System.out.println(n.extractLink()); try { if (n.extractLink().equals("http://www.zuzwn.com" )) { System.out.println(n.extractLink()); } } catch (Exception e) { } } } catch (Exception e) { e.printStackTrace(); } } }
/* * To change this template, choose Tools | Templates cambridge * and open the template in the editor. */ package exec; import java.io.File; import java.io.IOException; import org.htmlcleaner.CleanerProperties; import org.htmlcleaner.HtmlCleaner; import org.htmlcleaner.PrettyXmlSerializer; import org.htmlcleaner.TagNode; /** * */ public class HtmlClean { public void cleanHtml(String htmlurl, String xmlurl) { try { long start = System.currentTimeMillis(); HtmlCleaner cleaner = new HtmlCleaner(); CleanerProperties props = cleaner.getProperties(); props.setUseCdataForScriptAndStyle(true); cambridge props.setRecognizeUnicodeChars(true); props.setUseEmptyElementTags(true); props.setAdvancedXmlEscape(true); props.setTranslateSpecialEntities(true); props.setBooleanAttributeValues("empty"); TagNode node = cleaner.clean(new File(htmlurl)); System.out.println("vreme:" + (System.currentTimeMillis() - start)); new PrettyXmlSerializer(props).writeXmlToFile(node, xmlurl); System.out.println("vreme:" + (System.currentTimeMillis() - start)); } catch (IOException e) { e.printStackTrace(); } } }
No comments:
Post a Comment