htmlparser常用代码 parser n. [计] 分析程序;语法剖析程式
取得一段html代码里面所有的链接 C#版本, java版本类似:
string htmlcode = "<HTML><HEAD><TITLE>AAA</TITLE></HEAD><BODY>" + ...... + "</BODY></HTML>";
Parser
parser = Parser.CreateParser(htmlcode, "GBK");
HtmlPage
page = new HtmlPage(parser);
try
{ parser.VisitAllNodesWith(page);}
catch (ParserException e1)
{ e1 = null;}
NodeList
nodelist = page.Body;
NodeFilter
filter = new TagNameFilter("A");
nodelist =
nodelist.ExtractAllNodesThatMatch(filter, true);
for (int i = 0; i < nodelist.Size(); i++)
{
LinkTag link=(LinkTag) nodelist.ElementAt(i);
System.Console.Write(
link.GetAttribute("href") + "\n");
}