近日做了一下採集某個網頁的內容,並擷取其中所有的連結地址及連結標題。
其中用到了HttpWebRequest和Regex,代碼備忘如下:
//WebClient wc = new WebClient();
//NetworkCredential nc = new NetworkCredential("使用者名稱", "密碼", "網域名稱");
//wc.Credentials = nc;
//Response.Write(Server.HtmlEncode(wc.DownloadString("地址")));
HttpWebRequest req = (HttpWebRequest)WebRequest.Create("地址");
req.Credentials = new NetworkCredential("使用者名稱", "密碼", "網域名稱");
req.Method = "GET";
IAsyncResult ir = req.BeginGetResponse(null, null);
ir.AsyncWaitHandle.WaitOne();
try {
HttpWebResponse response1 = (HttpWebResponse)req.EndGetResponse(ir);
System.IO.Stream stream = response1.GetResponseStream();
sReader = new System.IO.StreamReader(stream, System.Text.Encoding.GetEncoding("GB2312"));
if (null != sReader) {
string pattern = @"<a(?:\s*?)href=['|""](?<url>[\s\S]+?)['|""]>(?<title>[\s\S]+?)</a>";
System.Text.RegularExpressions.MatchCollection matchs = System.Text.RegularExpressions.Regex.Matches(sReader.ReadToEnd(), pattern);
if (matchs.Count <= 0)
Response.Write("沒有匹配項");
else
{
for(int i=0;i<50;i++)
{
Response.Write("連結:" + matchs[i].Groups["url"].Value+"___名稱:"+matchs[i].Groups["title"].Value+"<br />");
}
}
}
}
catch (System.Exception ex) {
Response.Write(ex.Message);
}
finally {
if (null != sReader) {
sReader.Dispose();
}
}
這其中,Regex迷糊了我一會兒:因為沒有使用惰性匹配,導致每一次都只能匹配到一條資訊。。。。