以下為引用的內(nèi)容: public partial class Form2 : Form { public Form2() { InitializeComponent(); } //姓名 public static string XM = ""; //年齡 public static string nl = ""; //性別 public static string XB = ""; //身高 public static string SG = ""; //政治面貌 public static string mm = ""; //民族 public static string MZ = ""; //學(xué)歷 public static string XL = ""; //婚姻狀況 public static string HK = ""; //所學(xué)專業(yè) public static string ZY = ""; //工作經(jīng)驗(yàn) public static string GZJY = ""; //在職單位 public static string ZZDW = ""; //在職職位 public static string ZZZW = ""; //工作經(jīng)歷 public static string GZJL = ""; //要求月薪 public static string YX = ""; //工作性質(zhì) public static string GZXZ = ""; //求職意向 public static string QZYX = ""; //具體職務(wù) public static string JTZW = ""; //期望工作地 public static string QWGZD = ""; //教育情況,語言水平,技術(shù)專長 public static string QT = "";
private void button1_Click(object sender, EventArgs e) { label1.Text = "正在采集數(shù)據(jù)……";
//遍歷數(shù)據(jù)的頁數(shù) for (int i = 1; i <=50; i++) { CJ("http://www.xcjob.cn/renli.asp?pageno=" + i); }
label1.Text = "恭喜你采集完成!"; MessageBox.Show("恭喜你采集完成!"); }
//采集數(shù)據(jù) private void CJ(string Url) { //獲得頁面源文件(Html) string strWebContent = YM(Url);
//按照Html里面的標(biāo)簽 取出和數(shù)據(jù)有關(guān)的那段源碼 int iBodyStart = strWebContent.IndexOf("<body", 0); int aaa = strWebContent.IndexOf("關(guān)鍵字:", iBodyStart); int iTableStart = strWebContent.IndexOf("<table", aaa); int iTableEnd = strWebContent.IndexOf("</table>", iTableStart); string strWeb = strWebContent.Substring(iTableStart, iTableEnd - iTableStart);
//生成HtmlDocument HtmlElementCollection htmlTR = HtmlTR_Content(strWeb, "tr");
foreach (HtmlElement tr in htmlTR) { try { //姓名 XM = tr.GetElementsByTagName("a")[0].InnerText; //獲得詳細(xì)信息頁面的網(wǎng)址 string a = tr.GetElementsByTagName("a")[0].GetAttribute("href").ToString(); a = "http://www.xcjob.cn" + a.Substring(11);
Content(a); } catch { } } }
//采集詳細(xì)數(shù)據(jù) private void Content(string URL) { try { string strWebContent = YM(URL);
//按照Html里面的標(biāo)簽 取出和數(shù)據(jù)有關(guān)的那段源碼 int iBodyStart = strWebContent.IndexOf("<body", 0); int iTableStart = strWebContent.IndexOf("瀏覽次數(shù)", iBodyStart); int iTableEnd = strWebContent.IndexOf("<table", iTableStart); int dd = strWebContent.IndexOf("</table>", iTableEnd); string strWeb = strWebContent.Substring(iTableEnd, dd - iTableEnd + 8);
HtmlElementCollection htmlTR = HtmlTR_Content(strWeb, "table");
foreach (HtmlElement tr in htmlTR) { try { //年齡 nl = tr.GetElementsByTagName("tr")[1].GetElementsByTagName("td")[1].InnerText; //性別 string XB_SG = tr.GetElementsByTagName("tr")[1].GetElementsByTagName("td")[3].InnerText; XB = XB_SG.Substring(0, 1); //身高 SG = XB_SG.Substring(11); //政治面貌 mm = tr.GetElementsByTagName("tr")[2].GetElementsByTagName("td")[1].InnerText; //民族 MZ = tr.GetElementsByTagName("tr")[2].GetElementsByTagName("td")[3].InnerText; //學(xué)歷 XL = tr.GetElementsByTagName("tr")[3].GetElementsByTagName("td")[1].InnerText; //婚煙狀況 HK = tr.GetElementsByTagName("tr")[3].GetElementsByTagName("td")[3].InnerText; //所學(xué)專業(yè) ZY = tr.GetElementsByTagName("tr")[5].GetElementsByTagName("td")[1].InnerText; //工作經(jīng)驗(yàn) GZJY = tr.GetElementsByTagName("tr")[5].GetElementsByTagName("td")[3].InnerText; //在職單位 ZZDW = tr.GetElementsByTagName("tr")[6].GetElementsByTagName("td")[1].InnerText; //在職職位 ZZZW = tr.GetElementsByTagName("tr")[6].GetElementsByTagName("td")[3].InnerText; //工作經(jīng)歷 GZJY = tr.GetElementsByTagName("tr")[7].GetElementsByTagName("td")[1].InnerText; //要求月薪 YX = tr.GetElementsByTagName("tr")[9].GetElementsByTagName("td")[1].InnerText; //工作性質(zhì) GZXZ = tr.GetElementsByTagName("tr")[9].GetElementsByTagName("td")[3].InnerText; //求職意向 QZYX = tr.GetElementsByTagName("tr")[10].GetElementsByTagName("td")[1].InnerText; //具體職務(wù) JTZW = tr.GetElementsByTagName("tr")[10].GetElementsByTagName("td")[3].InnerText; //期望工作地 QWGZD = tr.GetElementsByTagName("tr")[11].GetElementsByTagName("td")[1].InnerText; //教育情況,語言水平,技術(shù)專長 QT = tr.GetElementsByTagName("tr")[13].GetElementsByTagName("td")[1].InnerText;
insert(); } catch { } } } catch { } }
//將數(shù)據(jù)插入數(shù)據(jù)庫 private void insert() { try { string str = "Provider=Microsoft.Jet.OleDb.4.0;Data Source=Data.mdb"; string sql = "insert into 人才信息 (姓名,年齡,性別,身高,政治面貌,民族,學(xué)歷,婚煙狀況,所學(xué)專業(yè),"; sql += "工作經(jīng)驗(yàn),在職單位,在職職位,工作經(jīng)歷,要求月薪,工作性質(zhì),求職意向,具體職務(wù),期望工作地,其他) values "; sql += "('" + XM + "'," + nl + ",'" + XB + "','" + SG + "','" + mm + "','" + MZ + "','" + XL + "','" + HK + "','" + ZY + "','" + GZJY + "','" + ZZDW + "','" + ZZZW + "',"; sql += "'" + GZJY + "','" + YX + "','" + GZXZ + "','" + QZYX + "','" + JTZW + "','" + QWGZD + "','" + QT + "')";
OleDbConnection con = new OleDbConnection(str); OleDbCommand com = new OleDbCommand(sql, con); con.Open(); com.ExecuteNonQuery(); con.Close(); } catch { } }
//返回一個HtmlElementCollection,然后進(jìn)行查詢內(nèi)容 private HtmlElementCollection HtmlTR_Content(string strWeb, string tj) { try { //生成HtmlDocument WebBrowser webb = new WebBrowser(); webb.Navigate("about:blank"); //window.document返回一個htmldocument對象,表示對一個html文檔的操作 //htmldocument對象是在xmldocument基礎(chǔ)上建立的,具有xmldocument的一切方法屬性 HtmlDocument htmldoc = webb.Document.OpenNew(true); htmldoc.Write(strWeb); HtmlElementCollection htmlTR = htmldoc.GetElementsByTagName(tj);
return htmlTR; } catch { return null; } }
//獲得網(wǎng)址原代碼 private string YM(string Url) { string strResult = "";
try { HttpWebRequest request = (HttpWebRequest)WebRequest.Create(Url); request.Method = "GET"; HttpWebResponse response = (HttpWebResponse)request.GetResponse(); Stream streamReceive = response.GetResponseStream(); Encoding encoding = Encoding.GetEncoding("GB2312"); StreamReader streamReader = new StreamReader(streamReceive, encoding); strResult = streamReader.ReadToEnd(); } catch { }
return strResult; } }
|