写了一个简单的抓取网页数据的小例子,代码如下:
//根据Url地址得到网页的html源码
private string GetWebContent(string Url)
{
string strResult = ""; ;
try
{
HttpWebRequest request = (HttpWebRequest)WebRequest.Create(Url);
//声明一个HttpWebRequest请求
request.Timeout = ;
//设置连接超时时间
request.Headers.Set("Pragma", "no-cache");
HttpWebResponse response = (HttpWebResponse)request.GetResponse();
Stream streamReceive = response.GetResponseStream();
Encoding encoding = Encoding.GetEncoding("GB2312");
StreamReader streamReader = new StreamReader(streamReceive, encoding);
strResult = streamReader.ReadToEnd();
}
catch
{ }
return strResult;
}
//为了使用HttpWebRequest和HttpWebResponse,需填名字空间引用 //以下是程序具体实现过程:
protected void btn_Click(object sender, EventArgs e)
{
//要抓取的URL地址
string Url = "http://www.awtrip.com/";
//得到指定Url的源码
string strWebContent = GetWebContent(Url);
//Response.Write(strWebContent);
//取出和数据有关的那段源码
int iBodyStart = strWebContent.IndexOf("<body", );
int iStart = strWebContent.IndexOf("热门目的地旅游", iBodyStart);
int iTableStart = strWebContent.IndexOf("<ul", iStart);
int iTableEnd = strWebContent.IndexOf("</ul>", iTableStart);
string strWeb = strWebContent.Substring(iTableStart, iTableEnd - iTableStart + );
//生成HtmlDocument
WebBrowser webb = new WebBrowser();
webb.Navigate("about:blank");
HtmlDocument htmldoc = webb.Document.OpenNew(true);
htmldoc.Write(strWeb);
HtmlElementCollection htmlTR = htmldoc.GetElementsByTagName("li");
StringBuilder strlist = new StringBuilder();
foreach (HtmlElement tr in htmlTR)
{
strlist.AppendFormat(tr.GetElementsByTagName("a")[].InnerText+"$");
}
Response.Write(strlist.ToString().Remove(strlist.ToString().Length-));
////最后再插入数据库 }
引用:
using System.Net;
using System.IO;
using System.Text;
using System.Windows.Forms;
运行时可能为遇到“当前线程不在单线程单元中,因此无法实例化 ActiveX 控件”的问题,把aspx页面顶部的AutoEventWireup设置为ture就可以了