用正则表达式抓取网页中的ul 和 li标签中最终的值！

获取你要抓取的页面

const string URL = "http://www.hn3ddf.gov.cn/price/GetList.html?pageno=1";
            string htmlStr = null;
            for (int i = 0; i < 10; i++)
            {
                try
                {
                    System.Net.HttpWebRequest request = (System.Net.HttpWebRequest)System.Net.WebRequest.Create(URL);
                    request.Headers.Set("Pragma", "no-cache");
                    request.Timeout = 10000 + (i * 5000);
                    System.Net.HttpWebResponse response = (System.Net.HttpWebResponse)request.GetResponse();
                    System.IO.Stream streamReceive = response.GetResponseStream();
                    System.IO.StreamReader streamReader = new System.IO.StreamReader(streamReceive, Encoding.GetEncoding("utf-8"));
                    htmlStr = streamReader.ReadToEnd();
                    break;
                }
                catch (Exception e)
                {
                    //----------------抓取异常！！
                }
            }

//抓取页面中的ul 标签中的特定一行属性

MatchCollection priceList = Regex.Matches(htmlStr, @"<ul style=""font-size:12px;width:320px; margin:0; padding:0;"">(.*?)</ul>", RegexOptions.Singleline);

            StringBuilder resultStr = new StringBuilder();
            for (int i = 0; i < priceList.Count; i++)
            {
                try
                {
                      //<ul style="font-size:12px;width:320px; margin:0; padding:0;">
                      // <li style="color:#555555; float:left; display:block; width:140px; height:22px; line-height:22px;" align="center">铔嬮浮閰嶅悎楗叉枡</li>
                      // <li align="center" style="color:#555555; float:left; display:block; width:100px; height:22px; line-height:22px;">2.83鍏?鍗冨厠</li>
                      // <li style="color:#555555; float:left; display:block; width:50px;text-align:center; height:22px; line-height:22px;">05-21</li>
                      //</ul>

//List<string> list = new List<string>();   //放结果的泛型集合
                    //string splitStr = "</li>";
                    //string[] strArray = priceList[i].Value.Split(splitStr.ToArray());    //一组一组的li标签
                    //foreach (string item in strArray)
                    //{
                    //    int first = item.IndexOf('>');
                    //    int last = item.IndexOf("</li>");
                    //    list.Add(item.Substring(first, last - first));
                    //    //list.add(item.substring(item.indexof(">")));
                    //}
                    //MatchCollection items = Regex.Matches(htmlStr, @"<li.*(?=>)(.|\n)*?</li>");

resultStr.Append("<tr>");

//<li style="color:#555555; float:left; display:block; width:140px; height:22px; line-height:22px;" align="center">蛋鸡配合饲料</li>

//<ul style="font-size:12px;width:320px; margin:0; padding:0;">
                    //    <li style="color:#555555; float:left; display:block; width:140px; height:22px; line-height:22px;" align="center">蛋鸡配合饲料</li>
                    //    <li align="center" style="color:#555555; float:left; display:block; width:100px; height:22px; line-height:22px;">2.83元/千克</li>
                    //    <li style="color:#555555; float:left; display:block; width:50px;text-align:center; height:22px; line-height:22px;">05-21</li>
                    //</ul>
                    string priceItem = priceList[i].Value;
                    //string name = Regex.Match(priceItem, @"<li style=""color:#555555; float:left; display:block; width:140px; height:22px; line-height:22px;"" align=""center"">(.*?)</li>").Value;

//配备<开头的在抓取的网页中的li标签中的所有属性进行配备为真的一行结果包含：样式和值
Match TitleMatch = Regex.Match(priceItem, @"<li style=""color:#555555; float:left; display:block; width:140px; height:22px; line-height:22px;"" align=""center"">([^<]*)</li>", RegexOptions.IgnoreCase | RegexOptions.Multiline);
　　　　　　　//取上面一行中的只有属性的值Value.Groups[1],1 代表Regex.Match方法得到的Groups的索引是从1开始的，而不是从0开始的
string name = TitleMatch.Groups[1].Value;

//"color:#555555; float:left; display:block; width:140px; height:22px; line-height:22px;" align="center">铔嬮浮閰嶅悎楗叉枡
//name = name.Substring(10, name.Length - 15);
//name = name.Substring(113, name.Length - 118);

//string price = Regex.Match(priceItem, @"<li align=""center"" style=""color:#555555; float:left; display:block; width:100px; height:22px; line-height:22px;"">(.*?)</li>").Value;
                    //price = price.Substring(13, price.Length - 18);
                    //price = price.Substring(115, price.Length -120);
                    Match priceMatch = Regex.Match(priceItem, @"<li align=""center"" style=""color:#555555; float:left; display:block; width:100px; height:22px; line-height:22px;"">([^<]*)</li>", RegexOptions.IgnoreCase | RegexOptions.Multiline);
                    string price = priceMatch.Groups[1].Value;
//                    string weeks = Regex.Match(priceItem, @"<li style=""color:#555555; float:left; display:block; width:50px;text-align:center; height:22px; line-height:22px;"">(.*?)</li>
//").Value;
//                    //weeks = weeks.Substring(9, weeks.Length - 16);
//                    weeks = weeks.Substring(116, weeks.Length - 122);

Match weeksMatch = Regex.Match(priceItem, @"<li style=""color:#555555; float:left; display:block; width:50px;text-align:center; height:22px; line-height:22px;"">([^<]*)</li>", RegexOptions.IgnoreCase | RegexOptions.Multiline);
                    string weeks = weeksMatch.Groups[1].Value;
                    resultStr.Append("<td width=\"195\" height=\"25\" align=\"left\">" + name + "</td><td width=\"70\" height=\"25\" align=\"center\" style=\"text-align:right;\">" + price + "</td><td height=\"25\" align=\"center\" style=\"color:#55a8ea;\">" + weeks + "</td>");
                    resultStr.Append("</tr>");
                    #region 原来的
                    //resultStr.Append("<tr>");
                    //string priceItem = priceList[i].Value;
                    //string name = Regex.Match(priceItem, "width=125>.*?</td>").Value;
                    //name = name.Substring(10, name.Length - 15);
                    //string price = Regex.Match(priceItem, "<td width=50.*?</td>").Value;
                    //price = price.Substring(13, price.Length - 18);
                    //string weeks = Regex.Match(priceItem, "class=en>.*?</font>").Value;
                    //weeks = weeks.Substring(9, weeks.Length - 16);
                    //resultStr.Append("<td width=\"195\" height=\"25\" align=\"left\">" + name + "</td><td width=\"70\" height=\"25\" align=\"center\">" + price + "</td><td height=\"25\" align=\"center\" style=\"color:#55a8ea;\">" + weeks + "</td>");
                    //resultStr.Append("</tr>");
                    #endregion
                }
                catch (Exception ex)
                {
                    //Common.Log4netUtil.Log().Error("获取跨域数据错误." + ex.Message);
                }
            }

return resultStr.ToString();

用正则表达式抓取网页中的ul 和 li标签中最终的值！的更多相关文章

正则表达式抓取文件内容中的http链接地址
import java.io.BufferedReader; import java.io.FileInputStream; import java.io.FileNotFoundException; ...
Java 抓取网页中的内容【持续更新】
背景:前几天复习Java的时候看到URL类,当时就想写个小程序试试,迫于考试没有动手,今天写了下,感觉还不错内容1. 抓取网页中的URL 知识点:Java URL+ 正则表达式 import jav ...
Python抓取网页中的图片到本地
今天在网上找了个从网页中通过图片URL,抓取图片并保存到本地的例子: #!/usr/bin/env python # -*- coding:utf- -*- # Author: xixihuang # ...
delphi 7中使用idhttp抓取网页解决假死现象
在delphi 7中使用idhttp抓取网页,造成窗口无反应的假死状态.通过搜索获得两种方法. 1.写在线程中,但是调用比较麻烦 2.使用delphi 提供的idantifreeze(必须安装indy ...
php抓取网页中的内容
以下就是几种常用的用php抓取网页中的内容的方法.1.file_get_contentsPHP代码代码如下:>>>>>>>>>>>&g ...
delphi 7中使用idhttp抓取网页解决假死现象（使用TIdAntiFreezeControl控件）
在delphi 7中使用idhttp抓取网页,造成窗口无反应的假死状态.通过搜索获得两种方法. 1.写在线程中,但是调用比较麻烦 2.使用delphi 提供的idantifreeze(必须安装indy ...
jmeter从上一个请求使用正则表达式抓取Set-Cookie值，在下一个请求中运用
工作中遇到的问题,登录请求,返回的Response Headers中有个参数Set-Cookie,需要抓取这个参数,运用到下一个请求中,见下图: 通过正则表达式抓取Set-Cookie的值,由于该值存 ...
python 解决抓取网页中的中文显示乱码问题
关于爬虫乱码有很多各式各样的问题,这里不仅是中文乱码,编码转换.还包括一些如日文.韩文 .俄文.藏文之类的乱码处理,因为解决方式是一致的,故在此统一说明. 网络爬虫出现乱码的原因源网页编码和爬取下来 ...
python抓取网页中图片并保存到本地
#-*-coding:utf-8-*- import os import uuid import urllib2 import cookielib '''获取文件后缀名''' def get_file ...

随机推荐

win7+IE11 中开发工具报错occurredJSLugin.3005解决办法
系统环境 win7+IE11 报错描述: Exception in window.onload: Error: An error has ocurredJSPlugin.3005 Stack Trac ...
Router和History (路由控制)-backbone
Router和History (路由控制) Backbone.Router担任了一部分Controller(控制器)的工作,它一般运行在单页应用中,能将特定的URL或锚点规则绑定到一个指定的方法(后文 ...
SharePoint 2013 实战碎嘴（ECMAScript客户端对象模型）：提示某个列表不存在
简单情景描述1:(在Sharepoint 2013 Solution 中) 在相应的.aspx页面引入一下两个.js文件: <script type="text/javascript ...
python核心编程-第三章-个人笔记
1.语句和语法 (1)反斜杠"\"表示语句继续.python良好的编程习惯是一行最后不超过80个字符,一行字符过多时便须用到反斜杠换行继续该语句. PS:在使用小括号.中括号.大括 ...
socket简单通信
粗糙简略的初版,后续多加点功能权当练手 /* ============================================================================ ...
U盘开发之安全U盘
普通型安全U盘,虚拟KEY和U盘两个设备,由主机软件分别对KEY和U盘进行操作,U盘与上位机采用usb mass storage接口,KEY采用HID接口,两者均无需驱动.也有虚拟成光盘和U盘两个设备 ...
李维作答《insideVCL》——李维实在很勤奋，而且勇于突破，从不以旧的内容充数
(编者按)<Inside VCL(VCL核心架构剖析)>一书出版以来,众多热心读者给李维先生.博文视点公司.CSDN写来信件,有更多朋友在各个论坛上发表关于该书的言论.读者们不但盛赞该书, ...
Unix/Linux环境C编程入门教程(30) 字符串操作那些事儿
函数介绍 rindex(查找字符串中最后一个出现的指定字符) 相关函数 index,memchr,strchr,strrchr 表头文件 #include<string.h> 定义函数 c ...
【FSFA 读书笔记】Ch 2 Computer Foundatinons（2）
Hard Disk Technology 1. 机械硬盘内部构造几个重要概念:Sector(扇区),Head(读写头),Track(磁道),Cylinder(柱面). 如果一个文件比较大,磁盘的写入 ...
SecureCRT按退格键出现^H问题解决
解决办法一: 解决办法二: ctrl+backspace.即是返回

用正则表达式抓取网页中的ul 和 li标签中最终的值！

用正则表达式抓取网页中的ul 和 li标签中最终的值！的更多相关文章

随机推荐

热门专题