源码网商城,靠谱的源码在线交易网站 我的订单 购物车 帮助

源码网商城

提取HTML代码中文字的C#函数

  • 时间:2020-09-03 14:23 编辑: 来源: 阅读:
  • 扫一扫,手机访问
摘要:提取HTML代码中文字的C#函数
/// <summary>   /// 去除HTML标记   /// </summary>   /// <param name="strHtml">包括HTML的源码 </param>   /// <returns>已经去除后的文字</returns>   public static string StripHTML(string strHtml)   {    string [] aryReg ={           @"<script[^>]*?>.*?</script>",           @"<(\/\s*)?!?((\w+:)?\w+)(\w+(\s*=?\s*(([""'])([url=file://\\[""]\\[""'tbnr]|[^\7])*?\7|\w+)|.{0})|\s)*?(\/\s[/url]*)?>",           @"([\r\n])[\s]+",           @"&(quot|#34);",           @"&(amp|#38);",           @"&(lt|#60);",           @"&(gt|#62);",           @"&(nbsp|#160);",           @"&(iexcl|#161);",           @"&(cent|#162);",           @"&(pound|#163);",           @"&(copy|#169);",           @"&#(\d+);",           @"-->",           @"<!--.*\n"          };    string [] aryRep = {            "",            "",            "",            "\"",            "&",            "<",            ">",            " ",            "\xa1",//chr(161),            "\xa2",//chr(162),            "\xa3",//chr(163),            "\xa9",//chr(169),            "",            "\r\n",            ""           };    string newReg =aryReg[0];    string strOutput=strHtml;    for(int i = 0;i<aryReg.Length;i++)    {     Regex regex = new Regex(aryReg[i],RegexOptions.IgnoreCase );     strOutput = regex.Replace(strOutput,aryRep[i]);    }    strOutput.Replace("<","");    strOutput.Replace(">","");    strOutput.Replace("\r\n","");    return strOutput;   }
  • 全部评论(0)
联系客服
客服电话:
400-000-3129
微信版

扫一扫进微信版
返回顶部