xpath 参考
using System;
using System.Collections.Generic;
using System.Linq;
using System.Web;
using System.Text.RegularExpressions;
using System.Configuration; /// <summary>
////// </summary>
public static class SearchConst
{ public static readonly string ARG_CLIENT = "client"; public static readonly string ARG_WORD = "word"; public static readonly int DataColumnCount = ; public static readonly int ColumnOfUrl = ; public static readonly int ColumnOfTitle = ; public static readonly int ColumnOfInfo = ; public static readonly int ColumnOfAdUrl = ; public static readonly string FMT_Date = "yyyy/MM/dd"; public static readonly string FMT_TIME = "HH:mm:ss"; public static readonly string UserAgentPC = "Mozilla/5.0 (Windows NT 6.1; WOW64; rv:11.0) Gecko/20100101 Firefox/11.0"; public static readonly string UserAgentMobile = "Mozilla/5.0 (iPhone; CPU iPhone OS 6_0 like Mac OS X) AppleWebKit/536.26 (KHTML, like Gecko) Version/6.0 Mobile/10A403 Safari/8536.25"; public static readonly string SearchKeyWordPlace = "#{q}"; public static readonly string DefaultEncode = "UTF-8"; public static readonly string AttributeHref = "href"; public static readonly string FILEEXT_ZIP = ".zip"; public static readonly string FILE_TXT = "source.txt"; public static readonly string FILE_KEY = "SavePath"; public static readonly string BATCH_PARALLES_KEY = "BatchParalles"; public static readonly string FLG_ENABLED = ""; public static readonly string CLIENT_MONITOR = "BJMOR"; public static readonly string MSG_E_PAGE_STYLE_CHANGE = "fff"; public static class Google
{ public static readonly string UserAgent = UserAgentPC; public static readonly string[] XPATH_ROOT = { "mbEnd", "mbEnd" };
public static readonly string[] XPATH_CITE = { "//div[@id='mbEnd']//ol/li//cite", "//div[@id='mbEnd']//ol/li//cite" }; //获取url
public static readonly string[] XPATH_H3 = { "//div[@id='mbEnd']//ol/li//h3", "//div[@id='mbEnd']//ol/li/h3" }; // //获取标题
public static readonly string[] XPATH_ADURL = { "//div[@id='mbEnd']//ol/li//h3//a[1]", "//div[@id='mbEnd']//ol/li/h3//a[1]" };
public static readonly string[] XPATH_INFO = { "//div[@id='mbEnd']//ol/li//div[@class='ac ads-creative']", "//div[@id='mbEnd']//ol/li//div[@class='ads-creative']" };
// top info
public static readonly string[] XPATH_ROOT_TOP = { "taw", "taw" };
public static readonly string[] XPATH_CITE_TOP = { "//div[@id='tads']//ol/li//cite", "//div[@id='tads']//ol/li//cite" };
public static readonly string[] XPATH_H3_TOP = { "//div[@id='tads']//ol/li//h3", "//div[@id='tads']//ol/li/h3" };
public static readonly string[] XPATH_ADURL_TOP = { "//div[@id='tads']//ol/li//h3//a[1]", "//div[@id='tads']//ol/li/h3//a[1]" };
public static readonly string[] XPATH_INFO_TOP = { "//div[@id='tads']//ol/li//div[@class='ac ads-creative']", "//div[@id='tads']//ol/li//div[@class='ads-creative']" };
//
public static readonly Regex RegexAdUrl = new Regex(@"adurl=(http[\S]*$)");
//
public static readonly string BAITAI_ID = "";
} public static class GoogleM
{
public static readonly string UserAgent = UserAgentMobile; //info
public static readonly string[] XPATH_ROOT = { "bottomads", "bottomads" };
public static readonly string[] XPATH_CITE = { "//div[@id='tadsb']/ol/li//cite", "//div[@id='tadsb']/ol/li//cite" };
public static readonly string[] XPATH_H3 = { "//div[@id='tadsb']/ol/li//h3", "//div[@id='tadsb']/ol/li//h3" };
public static readonly string[] XPATH_ADURL = { "//div[@id='tadsb']/ol/li//h3//a", "//div[@id='tadsb']/ol/li//h3//a" };
public static readonly string[] XPATH_INFO = { "//div[@id='tadsb']/ol/li//div[@class='ac ads-creative']", "//div[@id='tadsb']/ol/li//div[@class='ads-creative']" }; // top info
public static readonly string[] XPATH_ROOT_TOP = { "tads", "tads" };
public static readonly string[] XPATH_CITE_TOP = { "//div[@id='tads']/ol/li//cite", "//div[@id='tads']/ol/li//cite" };
public static readonly string[] XPATH_H3_TOP = { "//div[@id='tads']/ol/li//h3", "//div[@id='tads']/ol/li//h3" };
public static readonly string[] XPATH_ADURL_TOP = { "//div[@id='tads']/ol/li//h3//a", "//div[@id='tads']/ol/li//h3//a" };
public static readonly string[] XPATH_INFO_TOP = { "//div[@id='tads']/ol/li//div[@class='ac ads-creative']", "//div[@id='tads']/ol/li//div[@class='ads-creative']" };
//
public static readonly Regex RegexAdUrl = new Regex(@"adurl=(http[\S]*$)");
//
public static readonly string BAITAI_ID = "";
} public static class MSN
{
public static readonly string UserAgent = UserAgentPC;
//b_context/b_ad
public static readonly string[] XPATH_ROOT = { "sidebar", "b_context" };
public static readonly string[] XPATH_CITE = { "//div[@class='sb_adsNv2']//li//cite", "//ol[@id='b_context']//li[@class='b_ad']//li//cite" };
public static readonly string[] XPATH_H3 = { "//div[@class='sb_adsNv2']//li//h3", "//ol[@id='b_context']//li[@class='b_ad']//li//h2" };
public static readonly string[] XPATH_ADURL = { "//div[@class='sb_adsNv2']//li//a", "//ol[@id='b_context']//li[@class='b_ad']//li//a" };
public static readonly string[] XPATH_INFO = { "//div[@class='sb_adsNv2']//li//p", "//ol[@id='b_context']//li[@class='b_ad']//li//p" };
//b_results/b_ad
public static readonly string[] XPATH_ROOT_TOP = { "results_container", "b_results" };
public static readonly string[] XPATH_CITE_TOP = { "//div[@class='sb_adsWv2']//li//cite", "//ol[@id='b_results']//li[@class='b_ad']//li//cite" };
public static readonly string[] XPATH_H3_TOP = { "//div[@class='sb_adsWv2']//li//h3", "//ol[@id='b_results']//li[@class='b_ad']//li//h2" };
public static readonly string[] XPATH_ADURL_TOP = { "//div[@class='sb_adsWv2']//li//a", "//ol[@id='b_results']//li[@class='b_ad']//li//a" };
public static readonly string[] XPATH_INFO_TOP = { "//div[@class='sb_adsWv2']//li//p", "//ol[@id='b_results']//li[@class='b_ad']//li//p" };
//
public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)");
//
public static readonly string BAITAI_ID = "";
} public static class Yahoo
{
public static readonly string UserAgent = UserAgentPC; public static readonly string XPATH_ROOT = "sIn";
public static readonly string XPATH_CITE1 = "//div[@id='So3']/div[@class='bd']/div[@class='w']/div[@class='a cf']";
public static readonly string XPATH_H31 = "//div[@id='So3']/div[@class='bd']/div[@class='w']/h3";
public static readonly string XPATH_ADURL1 = "//div[@id='So3']/div[@class='bd']/div[@class='w']/h3/a";
public static readonly string XPATH_INFO1 = "//div[@id='So3']/div[@class='bd']/div[@class='w']/p";
//
public static readonly string XPATH_ROOT_TOP = "So1";
public static readonly string XPATH_CITE_TOP = "//div[@id='So1']/div[@class='bd']/div[@class='w']/div[@class='a cf']";
public static readonly string XPATH_H3_TOP = "//div[@id='So1']/div[@class='bd']/div[@class='w']/h3";
public static readonly string XPATH_ADURL_TOP = "//div[@id='So1']/div[@class='bd']/div[@class='w']/h3/a";
public static readonly string XPATH_INFO_TOP = "//div[@id='So1']/div[@class='bd']/div[@class='w']/p";
//
public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)");
public static readonly string NullUrl = ">";
//
public static readonly string BAITAI_ID = "";
} public static class Yahoo2
{
public static readonly string UserAgent = UserAgentPC; public static readonly string XPATH_ROOT_TOP = "contents";
public static readonly string XPATH_CITE_TOP = "//div[@id='contents']/div[@class='cWrap']/div[@class='listWrap cf']/ul/li/cite";
public static readonly string XPATH_H3_TOP = "//div[@id='contents']/div[@class='cWrap']/div[@class='listWrap cf']/ul/li/h2/a";
public static readonly string XPATH_ADURL_TOP = "//div[@id='contents']/div[@class='cWrap']/div[@class='listWrap cf']/ul/li/h2/a";
public static readonly string XPATH_INFO_TOP = "//div[@id='contents']/div[@class='cWrap']/div[@class='listWrap cf']/ul/li/p[@class='smr']";
//
public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)");
public static readonly string NullUrl = ">";
//
public static readonly string BAITAI_ID = "";
} public static class YahooM
{
public static readonly string UserAgent = UserAgentMobile; public static readonly string XPATH_ROOT = "contentsInner";
public static readonly string XPATH_CITE = "//div[@id='contentsInner']//aside[@class='So']/div[@class='bd']/ul/li/cite";
public static readonly string XPATH_H3 = "//div[@id='contentsInner']//aside[@class='So']/div[@class='bd']/ul/li/h3";
public static readonly string XPATH_ADURL = "//div[@id='contentsInner']//aside[@class='So']/div[@class='bd']/ul/li/h3/a";
public static readonly string XPATH_INFO = "//div[@id='contentsInner']//aside[@class='So']/div[@class='bd']/ul/li/p[@class='dtl']"; public static readonly string XPATH_ROOT_TOP = "contentsInner";
public static readonly string XPATH_CITE_TOP = "//div[@id='contentsInner']/aside[@class='So next-cmm']/div[@class='bd']/ul/li/cite";
public static readonly string XPATH_H3_TOP = "//div[@id='contentsInner']/aside[@class='So next-cmm']/div[@class='bd']/ul/li/h3";
public static readonly string XPATH_ADURL_TOP = "//div[@id='contentsInner']/aside[@class='So next-cmm']/div[@class='bd']/ul/li/h3/a";
public static readonly string XPATH_INFO_TOP = "//div[@id='contentsInner']/aside[@class='So next-cmm']/div[@class='bd']/ul/li/p[@class='dtl']";
//
public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)");
public static readonly string NullUrl = ">";
//
public static readonly string BAITAI_ID = "";
} public static class BaiDu { public static readonly string UserAgent = "Mozilla/5.0 (Windows NT 6.1; WOW64; Trident/7.0; rv:11.0) like Gecko"; public static readonly string[] XPATH_ROOT = { "ec_im_container", "ec_im_container" }; //第一种情况 。
public static readonly string[] XPATH_CITE = { "//a/font[@size='-1' and @class]","//a/font[@size='-1' and @class]" }; //第一种情况
public static readonly string[] XPATH_H3 = { "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-is-main-url]", "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-is-main-url]" };//第一种情况
public static readonly string[] XPATH_ADURL = { "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-is-main-url]", "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-is-main-url]" };
public static readonly string[] XPATH_INFO = { "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-click]/font[1]", "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-click]/font[1]" };
// top info
public static readonly string[] XPATH_ROOT_TOP = { "content_left", "content_left" };
public static readonly string[] XPATH_CITE_TOP = { "//table[@data-click]/tbody/tr/td//a[not(@data-is-main-url) and not(contains(@href,'tool'))]/span", "//div[@class and @style]/div/div/a/span[1]|//div/table/tbody/tr/td[2]/div//a/span[1]" }; //前下后上
public static readonly string[] XPATH_H3_TOP = { "//table/tbody/tr/td/a[ @data-is-main-url]", "//div[@class and @style]/div/div/h3" }; //前下后上
public static readonly string[] XPATH_ADURL_TOP = { "//table/tbody/tr/td/a[ @data-is-main-url]", "//table/tbody/tr/td/a[ @data-is-main-url]" }; //前下后上
public static readonly string[] XPATH_INFO_TOP = { "//table[@data-click and @class]/tbody/tr[3]/td/a[not(./span)]|//table[@data-click and @class]/tbody/tr/td/table/tbody/tr/td/div/font/a", "//div[@class and @style]/div/div[not(./span)]/a|//div/table/tbody/tr/td/div/font/a[not(./span)]" }; //前
//
//public static readonly Regex RegexAdUrl = new Regex(@"http[\S]*$");
//
public static readonly string BAITAI_ID = "";
} public static class CnBing { public static readonly string UserAgent = UserAgentPC; public static readonly string[] XPATH_ROOT = { "b_context", "b_context" };
public static readonly string[] XPATH_CITE = { "//div[@class='sb_add sb_adTA']//cite", "//div[@class='sb_add sb_adTA']//cite" };
public static readonly string[] XPATH_H3 = { "//div[@class='sb_add sb_adTA']//h2/a", "//div[@class='sb_add sb_adTA']//h2/a" };//第一种情况
public static readonly string[] XPATH_ADURL = { "//div[@class='sb_add sb_adTA']//h2/a", "//div[@class='sb_add sb_adTA']//h2/a" };
public static readonly string[] XPATH_INFO = { "//div[@class='sb_add sb_adTA']//div[@class='b_caption']/p", "//div[@class='sb_add sb_adTA']//div[@class='b_caption']/p" };
// top info
public static readonly string[] XPATH_ROOT_TOP = { "gg", "gg" };
public static readonly string[] XPATH_CITE_TOP = { "", "" }; //前下后上
public static readonly string[] XPATH_H3_TOP = { "", "" }; //前下后上
public static readonly string[] XPATH_ADURL_TOP = { "", "" }; //前下后上
public static readonly string[] XPATH_INFO_TOP = { "", "" }; //前下部分广告后上
//
public static readonly Regex RegexAdUrl = new Regex(@"rturl=(http[\S]*$)");
//
public static readonly string BAITAI_ID = "";
} public static class HaoSou { public static readonly string UserAgent = UserAgentPC;
// 右边的广告
public static readonly string[] XPATH_ROOT = { "side", "side" }; //获取范围
public static readonly string[] XPATH_CITE = { "//ul[@id='rightbox']/li/p/cite[not(contains(text(),' http://e.360.cn'))]|//div[@id='m-spread-left']//cite", "//ul[@id='rightbox']/li/p/cite[not(contains(text(),' http://e.360.cn'))]|//div[@id='m-spread-left']//cite" }; //第一种情况
public static readonly string[] XPATH_H3 = { "//ul[@id='rightbox']/li/h3/a[not(contains(text(),'ss'))]|//div[@id='m-spread-left']//h3/a", "//ul[@id='rightbox']/li/h3/a[not(contains(text(),'ss'))]|//div[@id='m-spread-left']//h3/a" };//第一种情况
public static readonly string[] XPATH_ADURL = { "//ul[@id='rightbox']/li/h3/a[not(contains(text(),'ss'))]|//div[@id='m-spread-left']//h3/a", "//ul[@id='rightbox']/li/h3/a[not(contains(text(),'ss'))]|//div[@id='m-spread-left']//h3/a" };
public static readonly string[] XPATH_INFO = { "//ul[@id='e_idea_pp']/li//p|//ul[@id='rightbox']/li/p[not(contains(text(),'4000-360-360'))]", "//ul[@id='e_idea_pp']/li//p|//ul[@id='rightbox']/li/p[not(contains(text(),'4000-360-360'))]" };
// top info
public static readonly string[] XPATH_ROOT_TOP = {"ss", "sss" };
public static readonly string[] XPATH_CITE_TOP = { "", "" }; //前下后上
public static readonly string[] XPATH_H3_TOP = { "", "" }; //前下后上
public static readonly string[] XPATH_ADURL_TOP = { "", "" }; //前下后上
public static readonly string[] XPATH_INFO_TOP = { "", "" }; //前下部分广告后上
//
//public static readonly Regex RegexAdUrl = new Regex(@"http[\S]*$");
//
public static readonly string BAITAI_ID = "";
} public static class Sogou {
public static readonly string UserAgent = UserAgentPC;
//right 部分
public static readonly string[] XPATH_ROOT = { "right" };
public static readonly string[] XPATH_CITE = { "//div[@class='bizr_fb']" };//绿色的url
public static readonly string[] XPATH_H3 = { "//h3[@class='bizr_title']" };//#ad_leftresult_0 > h3:nth-child(1)
public static readonly string[] XPATH_ADURL = { "//h3[@class='bizr_title']/a" };//.h3的url
public static readonly string[] XPATH_INFO = { "//div[@class='bizr_ft']" };
//top 部分
public static readonly string[] XPATH_ROOT_TOP = { "promotion_adv_container" };//*[@id="promotion_adv_container"]/div/div
public static readonly string[] XPATH_CITE_TOP = { "//div[contains(@class,'biz_rb')and @id]/div//cite" };
public static readonly string[] XPATH_H3_TOP = { "//h3[@class='biz_title']" };
public static readonly string[] XPATH_ADURL_TOP = { "//h3[@class='biz_title']/a" };
public static readonly string[] XPATH_INFO_TOP = { "//div[@class='crown_info_box' or @class='biz_ft']|//div[contains(@id,'box_id')]/table" };// "" //
//public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)");
//
public static readonly string BAITAI_ID = "";
public static readonly string NullUrl = ">";
} }
using System;using System.Collections.Generic;using System.Linq;using System.Web;using System.Text.RegularExpressions;using System.Configuration;
/// <summary>/// SearchHelper の概要の説明です/// </summary>public static class SearchConst{
public static readonly string ARG_CLIENT = "client";
public static readonly string ARG_WORD = "word";
public static readonly int DataColumnCount = 4;
public static readonly int ColumnOfUrl = 0;
public static readonly int ColumnOfTitle = 1;
public static readonly int ColumnOfInfo = 2;
public static readonly int ColumnOfAdUrl = 3;
public static readonly string FMT_Date = "yyyy/MM/dd";
public static readonly string FMT_TIME = "HH:mm:ss";
public static readonly string UserAgentPC = "Mozilla/5.0 (Windows NT 6.1; WOW64; rv:11.0) Gecko/20100101 Firefox/11.0";
public static readonly string UserAgentMobile = "Mozilla/5.0 (iPhone; CPU iPhone OS 6_0 like Mac OS X) AppleWebKit/536.26 (KHTML, like Gecko) Version/6.0 Mobile/10A403 Safari/8536.25";
public static readonly string SearchKeyWordPlace = "#{q}";
public static readonly string DefaultEncode = "UTF-8";
public static readonly string AttributeHref = "href";
public static readonly string FILEEXT_ZIP = ".zip";
public static readonly string FILE_TXT = "source.txt";
public static readonly string FILE_KEY = "SavePath";
public static readonly string BATCH_PARALLES_KEY = "BatchParalles";
public static readonly string FLG_ENABLED = "1";
public static readonly string CLIENT_MONITOR = "BJMOR";
public static readonly string MSG_E_PAGE_STYLE_CHANGE = "スポンサーチェックの検索媒体レイアウト変更";
public static class Google {
public static readonly string UserAgent = UserAgentPC;
public static readonly string[] XPATH_ROOT = { "mbEnd", "mbEnd" }; public static readonly string[] XPATH_CITE = { "//div[@id='mbEnd']//ol/li//cite", "//div[@id='mbEnd']//ol/li//cite" }; //获取url public static readonly string[] XPATH_H3 = { "//div[@id='mbEnd']//ol/li//h3", "//div[@id='mbEnd']//ol/li/h3" }; // //获取标题 public static readonly string[] XPATH_ADURL = { "//div[@id='mbEnd']//ol/li//h3//a[1]", "//div[@id='mbEnd']//ol/li/h3//a[1]" }; public static readonly string[] XPATH_INFO = { "//div[@id='mbEnd']//ol/li//div[@class='ac ads-creative']", "//div[@id='mbEnd']//ol/li//div[@class='ads-creative']" }; // top info public static readonly string[] XPATH_ROOT_TOP = { "taw", "taw" }; public static readonly string[] XPATH_CITE_TOP = { "//div[@id='tads']//ol/li//cite", "//div[@id='tads']//ol/li//cite" }; public static readonly string[] XPATH_H3_TOP = { "//div[@id='tads']//ol/li//h3", "//div[@id='tads']//ol/li/h3" }; public static readonly string[] XPATH_ADURL_TOP = { "//div[@id='tads']//ol/li//h3//a[1]", "//div[@id='tads']//ol/li/h3//a[1]" }; public static readonly string[] XPATH_INFO_TOP = { "//div[@id='tads']//ol/li//div[@class='ac ads-creative']", "//div[@id='tads']//ol/li//div[@class='ads-creative']" }; // public static readonly Regex RegexAdUrl = new Regex(@"adurl=(http[\S]*$)"); // public static readonly string BAITAI_ID = "001"; }
public static class GoogleM { public static readonly string UserAgent = UserAgentMobile;
//info public static readonly string[] XPATH_ROOT = { "bottomads", "bottomads" }; public static readonly string[] XPATH_CITE = { "//div[@id='tadsb']/ol/li//cite", "//div[@id='tadsb']/ol/li//cite" }; public static readonly string[] XPATH_H3 = { "//div[@id='tadsb']/ol/li//h3", "//div[@id='tadsb']/ol/li//h3" }; public static readonly string[] XPATH_ADURL = { "//div[@id='tadsb']/ol/li//h3//a", "//div[@id='tadsb']/ol/li//h3//a" }; public static readonly string[] XPATH_INFO = { "//div[@id='tadsb']/ol/li//div[@class='ac ads-creative']", "//div[@id='tadsb']/ol/li//div[@class='ads-creative']" };
// top info public static readonly string[] XPATH_ROOT_TOP = { "tads", "tads" }; public static readonly string[] XPATH_CITE_TOP = { "//div[@id='tads']/ol/li//cite", "//div[@id='tads']/ol/li//cite" }; public static readonly string[] XPATH_H3_TOP = { "//div[@id='tads']/ol/li//h3", "//div[@id='tads']/ol/li//h3" }; public static readonly string[] XPATH_ADURL_TOP = { "//div[@id='tads']/ol/li//h3//a", "//div[@id='tads']/ol/li//h3//a" }; public static readonly string[] XPATH_INFO_TOP = { "//div[@id='tads']/ol/li//div[@class='ac ads-creative']", "//div[@id='tads']/ol/li//div[@class='ads-creative']" }; // public static readonly Regex RegexAdUrl = new Regex(@"adurl=(http[\S]*$)"); // public static readonly string BAITAI_ID = "005"; }
public static class MSN { public static readonly string UserAgent = UserAgentPC; //b_context/b_ad public static readonly string[] XPATH_ROOT = { "sidebar", "b_context" }; public static readonly string[] XPATH_CITE = { "//div[@class='sb_adsNv2']//li//cite", "//ol[@id='b_context']//li[@class='b_ad']//li//cite" }; public static readonly string[] XPATH_H3 = { "//div[@class='sb_adsNv2']//li//h3", "//ol[@id='b_context']//li[@class='b_ad']//li//h2" }; public static readonly string[] XPATH_ADURL = { "//div[@class='sb_adsNv2']//li//a", "//ol[@id='b_context']//li[@class='b_ad']//li//a" }; public static readonly string[] XPATH_INFO = { "//div[@class='sb_adsNv2']//li//p", "//ol[@id='b_context']//li[@class='b_ad']//li//p" }; //b_results/b_ad public static readonly string[] XPATH_ROOT_TOP = { "results_container", "b_results" }; public static readonly string[] XPATH_CITE_TOP = { "//div[@class='sb_adsWv2']//li//cite", "//ol[@id='b_results']//li[@class='b_ad']//li//cite" }; public static readonly string[] XPATH_H3_TOP = { "//div[@class='sb_adsWv2']//li//h3", "//ol[@id='b_results']//li[@class='b_ad']//li//h2" }; public static readonly string[] XPATH_ADURL_TOP = { "//div[@class='sb_adsWv2']//li//a", "//ol[@id='b_results']//li[@class='b_ad']//li//a" }; public static readonly string[] XPATH_INFO_TOP = { "//div[@class='sb_adsWv2']//li//p", "//ol[@id='b_results']//li[@class='b_ad']//li//p" }; // public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)"); // public static readonly string BAITAI_ID = "003"; }
public static class Yahoo { public static readonly string UserAgent = UserAgentPC;
public static readonly string XPATH_ROOT = "sIn"; public static readonly string XPATH_CITE1 = "//div[@id='So3']/div[@class='bd']/div[@class='w']/div[@class='a cf']"; public static readonly string XPATH_H31 = "//div[@id='So3']/div[@class='bd']/div[@class='w']/h3"; public static readonly string XPATH_ADURL1 = "//div[@id='So3']/div[@class='bd']/div[@class='w']/h3/a"; public static readonly string XPATH_INFO1 = "//div[@id='So3']/div[@class='bd']/div[@class='w']/p"; // public static readonly string XPATH_ROOT_TOP = "So1"; public static readonly string XPATH_CITE_TOP = "//div[@id='So1']/div[@class='bd']/div[@class='w']/div[@class='a cf']"; public static readonly string XPATH_H3_TOP = "//div[@id='So1']/div[@class='bd']/div[@class='w']/h3"; public static readonly string XPATH_ADURL_TOP = "//div[@id='So1']/div[@class='bd']/div[@class='w']/h3/a"; public static readonly string XPATH_INFO_TOP = "//div[@id='So1']/div[@class='bd']/div[@class='w']/p"; // public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)"); public static readonly string NullUrl = ">"; // public static readonly string BAITAI_ID = "002"; }
public static class Yahoo2 { public static readonly string UserAgent = UserAgentPC;
public static readonly string XPATH_ROOT_TOP = "contents"; public static readonly string XPATH_CITE_TOP = "//div[@id='contents']/div[@class='cWrap']/div[@class='listWrap cf']/ul/li/cite"; public static readonly string XPATH_H3_TOP = "//div[@id='contents']/div[@class='cWrap']/div[@class='listWrap cf']/ul/li/h2/a"; public static readonly string XPATH_ADURL_TOP = "//div[@id='contents']/div[@class='cWrap']/div[@class='listWrap cf']/ul/li/h2/a"; public static readonly string XPATH_INFO_TOP = "//div[@id='contents']/div[@class='cWrap']/div[@class='listWrap cf']/ul/li/p[@class='smr']"; // public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)"); public static readonly string NullUrl = ">"; // public static readonly string BAITAI_ID = "004"; }
public static class YahooM { public static readonly string UserAgent = UserAgentMobile;
public static readonly string XPATH_ROOT = "contentsInner"; public static readonly string XPATH_CITE = "//div[@id='contentsInner']//aside[@class='So']/div[@class='bd']/ul/li/cite"; public static readonly string XPATH_H3 = "//div[@id='contentsInner']//aside[@class='So']/div[@class='bd']/ul/li/h3"; public static readonly string XPATH_ADURL = "//div[@id='contentsInner']//aside[@class='So']/div[@class='bd']/ul/li/h3/a"; public static readonly string XPATH_INFO = "//div[@id='contentsInner']//aside[@class='So']/div[@class='bd']/ul/li/p[@class='dtl']";
public static readonly string XPATH_ROOT_TOP = "contentsInner"; public static readonly string XPATH_CITE_TOP = "//div[@id='contentsInner']/aside[@class='So next-cmm']/div[@class='bd']/ul/li/cite"; public static readonly string XPATH_H3_TOP = "//div[@id='contentsInner']/aside[@class='So next-cmm']/div[@class='bd']/ul/li/h3"; public static readonly string XPATH_ADURL_TOP = "//div[@id='contentsInner']/aside[@class='So next-cmm']/div[@class='bd']/ul/li/h3/a"; public static readonly string XPATH_INFO_TOP = "//div[@id='contentsInner']/aside[@class='So next-cmm']/div[@class='bd']/ul/li/p[@class='dtl']"; // public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)"); public static readonly string NullUrl = ">"; // public static readonly string BAITAI_ID = "006"; }
public static class BaiDu {
public static readonly string UserAgent = "Mozilla/5.0 (Windows NT 6.1; WOW64; Trident/7.0; rv:11.0) like Gecko";
public static readonly string[] XPATH_ROOT = { "ec_im_container", "ec_im_container" }; //第一种情况 好像就一种情况。 public static readonly string[] XPATH_CITE = { "//a/font[@size='-1' and @class]","//a/font[@size='-1' and @class]" }; //第一种情况 public static readonly string[] XPATH_H3 = { "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-is-main-url]", "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-is-main-url]" };//第一种情况 public static readonly string[] XPATH_ADURL = { "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-is-main-url]", "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-is-main-url]" }; public static readonly string[] XPATH_INFO = { "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-click]/font[1]", "//a[contains(@class,'EC_BL')and contains(@id,'dfs')and @data-click]/font[1]" }; // top info public static readonly string[] XPATH_ROOT_TOP = { "content_left", "content_left" }; public static readonly string[] XPATH_CITE_TOP = { "//table[@data-click]/tbody/tr/td//a[not(@data-is-main-url) and not(contains(@href,'tool'))]/span", "//div[@class and @style]/div/div/a/span[1]|//div/table/tbody/tr/td[2]/div//a/span[1]" }; //前下后上 public static readonly string[] XPATH_H3_TOP = { "//table/tbody/tr/td/a[ @data-is-main-url]", "//div[@class and @style]/div/div/h3" }; //前下后上 public static readonly string[] XPATH_ADURL_TOP = { "//table/tbody/tr/td/a[ @data-is-main-url]", "//table/tbody/tr/td/a[ @data-is-main-url]" }; //前下后上 public static readonly string[] XPATH_INFO_TOP = { "//table[@data-click and @class]/tbody/tr[3]/td/a[not(./span)]|//table[@data-click and @class]/tbody/tr/td/table/tbody/tr/td/div/font/a", "//div[@class and @style]/div/div[not(./span)]/a|//div/table/tbody/tr/td/div/font/a[not(./span)]" }; //前下部分广告后上 // //public static readonly Regex RegexAdUrl = new Regex(@"http[\S]*$"); // public static readonly string BAITAI_ID = "007"; }
public static class CnBing {
public static readonly string UserAgent = UserAgentPC;
public static readonly string[] XPATH_ROOT = { "b_context", "b_context" }; public static readonly string[] XPATH_CITE = { "//div[@class='sb_add sb_adTA']//cite", "//div[@class='sb_add sb_adTA']//cite" }; public static readonly string[] XPATH_H3 = { "//div[@class='sb_add sb_adTA']//h2/a", "//div[@class='sb_add sb_adTA']//h2/a" };//第一种情况 public static readonly string[] XPATH_ADURL = { "//div[@class='sb_add sb_adTA']//h2/a", "//div[@class='sb_add sb_adTA']//h2/a" }; public static readonly string[] XPATH_INFO = { "//div[@class='sb_add sb_adTA']//div[@class='b_caption']/p", "//div[@class='sb_add sb_adTA']//div[@class='b_caption']/p" }; // top info public static readonly string[] XPATH_ROOT_TOP = { "なし", "なし" }; public static readonly string[] XPATH_CITE_TOP = { "", "" }; //前下后上 public static readonly string[] XPATH_H3_TOP = { "", "" }; //前下后上 public static readonly string[] XPATH_ADURL_TOP = { "", "" }; //前下后上 public static readonly string[] XPATH_INFO_TOP = { "", "" }; //前下部分广告后上 // public static readonly Regex RegexAdUrl = new Regex(@"rturl=(http[\S]*$)"); // public static readonly string BAITAI_ID = "008"; }
public static class HaoSou {
public static readonly string UserAgent = UserAgentPC; // 右边的广告 public static readonly string[] XPATH_ROOT = { "side", "side" }; //获取范围 public static readonly string[] XPATH_CITE = { "//ul[@id='rightbox']/li/p/cite[not(contains(text(),' http://e.360.cn'))]|//div[@id='m-spread-left']//cite", "//ul[@id='rightbox']/li/p/cite[not(contains(text(),' http://e.360.cn'))]|//div[@id='m-spread-left']//cite" }; //第一种情况 public static readonly string[] XPATH_H3 = { "//ul[@id='rightbox']/li/h3/a[not(contains(text(),'好搜推广'))]|//div[@id='m-spread-left']//h3/a", "//ul[@id='rightbox']/li/h3/a[not(contains(text(),'好搜推广'))]|//div[@id='m-spread-left']//h3/a" };//第一种情况 public static readonly string[] XPATH_ADURL = { "//ul[@id='rightbox']/li/h3/a[not(contains(text(),'好搜推广'))]|//div[@id='m-spread-left']//h3/a", "//ul[@id='rightbox']/li/h3/a[not(contains(text(),'好搜推广'))]|//div[@id='m-spread-left']//h3/a" }; public static readonly string[] XPATH_INFO = { "//ul[@id='e_idea_pp']/li//p|//ul[@id='rightbox']/li/p[not(contains(text(),'4000-360-360'))]", "//ul[@id='e_idea_pp']/li//p|//ul[@id='rightbox']/li/p[not(contains(text(),'4000-360-360'))]" }; // top info public static readonly string[] XPATH_ROOT_TOP = {"なし", "なし" }; public static readonly string[] XPATH_CITE_TOP = { "", "" }; //前下后上 public static readonly string[] XPATH_H3_TOP = { "", "" }; //前下后上 public static readonly string[] XPATH_ADURL_TOP = { "", "" }; //前下后上 public static readonly string[] XPATH_INFO_TOP = { "", "" }; //前下部分广告后上 // //public static readonly Regex RegexAdUrl = new Regex(@"http[\S]*$"); // public static readonly string BAITAI_ID = "009"; }
public static class Sogou { public static readonly string UserAgent = UserAgentPC; //right 部分 public static readonly string[] XPATH_ROOT = { "right" }; public static readonly string[] XPATH_CITE = { "//div[@class='bizr_fb']" };//绿色的url public static readonly string[] XPATH_H3 = { "//h3[@class='bizr_title']" };//#ad_leftresult_0 > h3:nth-child(1) public static readonly string[] XPATH_ADURL = { "//h3[@class='bizr_title']/a" };//.h3的url public static readonly string[] XPATH_INFO = { "//div[@class='bizr_ft']" }; //top 部分 public static readonly string[] XPATH_ROOT_TOP = { "promotion_adv_container" };//*[@id="promotion_adv_container"]/div/div public static readonly string[] XPATH_CITE_TOP = { "//div[contains(@class,'biz_rb')and @id]/div//cite" }; public static readonly string[] XPATH_H3_TOP = { "//h3[@class='biz_title']" }; public static readonly string[] XPATH_ADURL_TOP = { "//h3[@class='biz_title']/a" }; public static readonly string[] XPATH_INFO_TOP = { "//div[@class='crown_info_box' or @class='biz_ft']|//div[contains(@id,'box_id')]/table" };// ""
// //public static readonly Regex RegexAdUrl = new Regex(@"\*\*(http[\S]*$)"); //0 public static readonly string BAITAI_ID = "010"; public static readonly string NullUrl = ">"; }
}
xpath 参考的更多相关文章
- 【转】XPath 示例
XPath 示例 其他版本 本主题回顾整个 XPath 参考中出现的语法示例. 所有示例均基于 XPath 语法的示例 XML 文件 (inventory.xml). 有关在测试文件中使用 X ...
- XPATH 带命名空间数据的读取
在XML中,很多情况下有命名空间,如果直接使用XPATH 读取是会读到空节点. 解决办法如下: InputStream is=loader.getResourceAsStream("com/ ...
- HtmlCleanner结合xpath用法(转载)
HtmlCleaner cleaner = new HtmlCleaner(); TagNode node = cleaner.clean(new URL("http://finance.s ...
- scrapy2_初窥Scrapy
递归知识:oop,xpath,jsp,items,pipline等专业网络知识,初级水平并不是很scrapy,可以从简单模块自己写. 初窥Scrapy Scrapy是一个为了爬取网站数据,提取结构性数 ...
- 较全的IT方面帮助文档
http://www.shouce.ren/post/d/id/108632 XSLT参考手册-新.CHMhttp://www.shouce.ren/post/d/id/108633 XSL-FO参考 ...
- selenium java 浏览器操作
环境搭建 selenium 2.53 selenium-java-2.53.0.jar selenium-java-2.53.0-srcs.jar 原代码包 拷贝的工程lib下,做build path ...
- JDOM 操作XML
http://www.cnblogs.com/hoojo/archive/2011/08/11/2134638.html 可扩展标记语言——eXtensible Markup Language 用户可 ...
- python爬虫 scrapy2_初窥Scrapy
sklearn实战-乳腺癌细胞数据挖掘 https://study.163.com/course/introduction.htm?courseId=1005269003&utm_campai ...
- [开发笔记]-Linq to xml学习笔记
最近需要用到操作xml文档的方法,学习了一下linq to xml,特此记录. 测试代码: class Program { //参考: LINQ to XML 编程基础 - luckdv - 博客园 ...
随机推荐
- Android 短信的还原
上篇文章讲到<Android 短信的备份>,本文主要实现Android 短信的还原,即是将一条 布局文件: <RelativeLayout xmlns:android="h ...
- [读书笔记] CSS权威指南1: 选择器
通配选择器 可以与任何元素匹配,就像是一个通配符 /*每一个元素的字体都设置为红色*/ * { color: red; } 元素选择器 指示文档元素的选择器. /*为body的字体设置为红色*/ bo ...
- [在线] html 转 pdf
http://www.htm2pdf.co.uk/
- Grunt安装配置教程:前端自动化工作流
Grunt这货是啥? Grunt 是一个基于任务的 JavaScript 项目命令行构建工具. 最近很火的前端自动化小工具,基于任务的命令行构建工具 http://gruntjs.com Grunt能 ...
- java环境变量 windows centos 安装jdk
windows: 1.安装jdk,注意不是jre 2. 计算机→属性→高级系统设置→高级→环境变量,选择下面的那个系统环境变量 3. 系统变量→新建 JAVA_HOME 变量 . 变量值填写jdk的安 ...
- 双十一来了,别让你的mongodb宕机了
好久没过来吹牛了,前段时间一直赶项目,没有时间来更新博客,项目也终于赶完了,接下来就要面临双十一这场惊心动魄的处女秀考验, 我们项目中会有一个wcf集群,而集群地址则放在mongodb中,所以mong ...
- Java并发之ThreadPoolExecutor 线程执行服务
package com.thread.test.thread; import java.util.concurrent.ExecutorService; import java.util.concur ...
- 烂泥:使用snmpwalk采集设备的OID信息
本文由秀依林枫提供友情赞助,首发于烂泥行天下. 打算开始学习有关监控方面的知识,但是现在很多监控系统都是根据SNMP进行的.而SNMP监控的性能指标很多都是通过snmpwalk采集设备的OID信息得到 ...
- springmvc+log4j操作日志记录,详细配置
没有接触过的,先了解一下:log4j教程 部分内容来:log4j教程 感谢! 需要导入包: log包:log4j-12.17.jar 第一步:web.xml配置 <!-- log4j配置,文件路 ...
- 高性能MySQL笔记 第4章 Schema与数据类型优化
4.1 选择优化的数据类型 通用原则 更小的通常更好 前提是要确保没有低估需要存储的值范围:因为它占用更少的磁盘.内存.CPU缓存,并且处理时需要的CPU周期也更少. 简单就好 简 ...