public string cutStr(string html)
  {
   
   
   System.Text.RegularExpressions.Regex regex1 = new System.Text.RegularExpressions.Regex(@"<script[\s\S]+</script *>",System.Text.RegularExpressions.RegexOptions.IgnoreCase);  
   
   //System.Text.RegularExpressions.Regex regex2 = new System.Text.RegularExpressions.Regex(@" href *= *[\s\S]*script *:",System.Text.RegularExpressions.RegexOptions.IgnoreCase);  
   
   //System.Text.RegularExpressions.Regex regex2 = new System.Text.RegularExpressions.Regex(@"<a[\s\S]+</a *>",System.Text.RegularExpressions.RegexOptions.IgnoreCase);  
    
   //System.Text.RegularExpressions.Regex regex3 = new System.Text.RegularExpressions.Regex(@" on[\s\S]*=",System.Text.RegularExpressions.RegexOptions.IgnoreCase);  
   System.Text.RegularExpressions.Regex regex4 = new System.Text.RegularExpressions.Regex(@"<iframe[\s\S]+</iframe *>",System.Text.RegularExpressions.RegexOptions.IgnoreCase);  
   System.Text.RegularExpressions.Regex regex5 = new System.Text.RegularExpressions.Regex(@"<frameset[\s\S]+</frameset *>",System.Text.RegularExpressions.RegexOptions.IgnoreCase);  
   System.Text.RegularExpressions.Regex regex6 = new System.Text.RegularExpressions.Regex(@"<(.[^>]*)>",System.Text.RegularExpressions.RegexOptions.IgnoreCase);  
   
   html = regex1.Replace(html, ""); //过滤<script>标记
   html = regex4.Replace(html, ""); //过滤iframe
   html = regex5.Replace(html, ""); //过滤frameset   
   html = regex6.Replace(html, ""); //过滤其他标记 
   html=html.Replace("&nbsp;","");
   html=html.Replace("\n","");
   html=html.Replace("\r","");
   html=html.Replace(" ","");

   return html;
   
  }

posted on 2006-10-27 09:50  Eric Yao  阅读(244)  评论(0)    收藏  举报