利用C#实现最基本的小说爬虫示例代码

2019-12-30 18:13:50于丽

二、下面就是用正则处理内容了,由于正则表达式不熟悉所以重复动作太多。

1.先获取网页内容


 IWebHttpRepository webHttpRepository = new WebHttpRepository();
   string html = webHttpRepository.HttpGet(Url_Txt.Text, "");

2.获取书名和文章列表

书名

C#,实现网络爬虫,小说爬虫,小说网站爬虫

文章列表

C#,实现网络爬虫,小说爬虫,小说网站爬虫


string Novel_Name = Regex.Match(html, @"(?<=<h1>)([Ss]*?)(?=</h1>)").Value; //获取书名

   Regex Regex_Menu = new Regex(@"(?is)(?<=<dl class=""book_list"">).+?(?=</dl>)");
   string Result_Menu = Regex_Menu.Match(html).Value; //获取列表内容


   Regex Regex_List = new Regex(@"(?is)(?<=<dd>).+?(?=</dd>)");
   var Result_List = Regex_List.Matches(Result_Menu); //获取列表集合

3.因为章节列表前面有多余的<dd>,所以要剔除


int i = 0; //计数
   string Menu_Content = ""; //所有章节
   foreach (var x in Result_List)
   {
    if (i < 4)
    {
     //前面五个都不是章节列表,所以剔除
    }
    else
    {
     Menu_Content += x.ToString();
    }
    i++;
   }

4.然后获取<a>的href和innerHTML,然后遍历访问获得内容和章节名称并处理,然后写入txt


Regex Regex_Href = new Regex(@"(?is)<a[^>]*?href=(['""]?)(?<url>[^'""s>]+)1[^>]*>(?<text>(?:(?!</?ab).)*)</a>");
   MatchCollection Result_Match_List = Regex_Href.Matches(Menu_Content); //获取href链接和a标签 innerHTML 

   string Novel_Path = Directory.GetCurrentDirectory() + "Novel" + Novel_Name + ".txt";  //小说地址
   File.Create(Novel_Path).Close();
   StreamWriter Write_Content = new StreamWriter(Novel_Path);


   foreach (Match Result_Single in Result_Match_List)
   {
    string Url_Text = Result_Single.Groups["url"].Value;
    string Content_Text = Result_Single.Groups["text"].Value;

    string Content_Html = webHttpRepository.HttpGet(Url_Txt.Text + Url_Text, "");//获取内容页

    Regex Rege_Content = new Regex(@"(?is)(?<=<p class=""Book_Text"">).+?(?=</p>)");
    string Result_Content = Rege_Content.Match(Content_Html).Value; //获取文章内容


    Regex Regex_Main = new Regex(@"(    )(.*)");
    string Rsult_Main = Regex_Main.Match(Result_Content).Value; //正文   
    string Screen_Content = Rsult_Main.Replace(" ", "").Replace("<br />", "rn");

    Write_Content.WriteLine(Content_Text + "rn");//写入标题
    Write_Content.WriteLine(Screen_Content);//写入内容
   }


   Write_Content.Dispose();
   Write_Content.Close();
   MessageBox.Show(Novel_Name+".txt 创建成功!");
   System.Diagnostics.Process.Start(Directory.GetCurrentDirectory() + Novel);