C # में स्ट्रिंग सहित HTML टैग निकालें

Question 1

मैं C # में regex का उपयोग करके & nbsp सहित सभी HTML टैग कैसे निकाल सकता हूं। मेरा स्ट्रिंग दिखता है

  "<div>hello</div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div>&nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp; &nbsp;&nbsp;</div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div><div><br></div>"

Question 2

यदि आप टैग्स को फ़िल्टर करने के लिए HTML पार्सर उन्मुख समाधान का उपयोग नहीं कर सकते हैं, तो इसके लिए एक सरल रेगेक्स है।

string noHTML = Regex.Replace(inputHTML, @"<[^>]+>|&nbsp;", "").Trim();

आपको आदर्श रूप से regex फ़िल्टर के माध्यम से एक और पास बनाना चाहिए जो कई स्थानों का ध्यान रखता है

string noHTMLNormalised = Regex.Replace(noHTML, @"\s{2,}", " ");

Question 3

मैंने @ रावी थपलियाल का कोड लिया और एक विधि बनाई: यह सरल है और शायद सब कुछ साफ नहीं हो सकता है, लेकिन अभी तक यह वही कर रहा है जो मुझे करने की आवश्यकता है।

public static string ScrubHtml(string value) {
    var step1 = Regex.Replace(value, @"<[^>]+>|&nbsp;", "").Trim();
    var step2 = Regex.Replace(step1, @"\s{2,}", " ");
    return step2;
}

Question 4

मैं कुछ समय से इस फ़ंक्शन का उपयोग कर रहा हूं। बहुत अधिक किसी भी गंदे HTML को हटाता है जिसे आप इस पर फेंक सकते हैं और पाठ को बरकरार रख सकते हैं।

        private static readonly Regex _tags_ = new Regex(@"<[^>]+?>", RegexOptions.Multiline | RegexOptions.Compiled);

        //add characters that are should not be removed to this regex
        private static readonly Regex _notOkCharacter_ = new Regex(@"[^\w;&#@.:/\\?=|%!() -]", RegexOptions.Compiled);

        public static String UnHtml(String html)
        {
            html = HttpUtility.UrlDecode(html);
            html = HttpUtility.HtmlDecode(html);

            html = RemoveTag(html, "<!--", "-->");
            html = RemoveTag(html, "<script", "</script>");
            html = RemoveTag(html, "<style", "</style>");

            //replace matches of these regexes with space
            html = _tags_.Replace(html, " ");
            html = _notOkCharacter_.Replace(html, " ");
            html = SingleSpacedTrim(html);

            return html;
        }

        private static String RemoveTag(String html, String startTag, String endTag)
        {
            Boolean bAgain;
            do
            {
                bAgain = false;
                Int32 startTagPos = html.IndexOf(startTag, 0, StringComparison.CurrentCultureIgnoreCase);
                if (startTagPos < 0)
                    continue;
                Int32 endTagPos = html.IndexOf(endTag, startTagPos + 1, StringComparison.CurrentCultureIgnoreCase);
                if (endTagPos <= startTagPos)
                    continue;
                html = html.Remove(startTagPos, endTagPos - startTagPos + endTag.Length);
                bAgain = true;
            } while (bAgain);
            return html;
        }

        private static String SingleSpacedTrim(String inString)
        {
            StringBuilder sb = new StringBuilder();
            Boolean inBlanks = false;
            foreach (Char c in inString)
            {
                switch (c)
                {
                    case '\r':
                    case '\n':
                    case '\t':
                    case ' ':
                        if (!inBlanks)
                        {
                            inBlanks = true;
                            sb.Append(' ');
                        }   
                        continue;
                    default:
                        inBlanks = false;
                        sb.Append(c);
                        break;
                }
            }
            return sb.ToString().Trim();
        }

Question 5

var noHtml = Regex.Replace(inputHTML, @"<[^>]*(>|$)|&nbsp;|&zwnj;|&raquo;|&laquo;", string.Empty).Trim();

Question 6

मैंने @RaviThapliyal & @Don Rolling के कोड का उपयोग किया है लेकिन थोड़ा संशोधन किया है। चूंकि हम खाली स्ट्रिंग के साथ & nbsp की जगह ले रहे हैं, लेकिन इसकी जगह & nbsp को स्थान से बदलना चाहिए, इसलिए एक अतिरिक्त चरण जोड़ा गया। इसने मेरे लिए एक आकर्षण की तरह काम किया।

public static string FormatString(string value) {
    var step1 = Regex.Replace(value, @"<[^>]+>", "").Trim();
    var step2 = Regex.Replace(step1, @"&nbsp;", " ");
    var step3 = Regex.Replace(step2, @"\s{2,}", " ");
    return step3;
}

सेमीकॉलन के बिना प्रयुक्त और nbps क्योंकि यह स्टैक ओवरफ्लो द्वारा स्वरूपित हो रहा था।

Question 7

यह:

(<.+?> | &nbsp;)

किसी भी टैग से मेल खाएगा या  

string regex = @"(<.+?>|&nbsp;)";
var x = Regex.Replace(originalString, regex, "").Trim();

तो x = hello

Question 8

Html डॉक्यूमेंट को सैनिटाइज करने में बहुत सारी पेचीदा चीजें शामिल हैं। यह पैकेज शायद मदद के लिए: https://github.com/mganss/HtmlSanitizer

Question 9

HTML अपने मूल रूप में सिर्फ XML में है। आप XmlDocument ऑब्जेक्ट में अपने पाठ को पार्स कर सकते हैं, और मूल तत्व पर इनर टेक्स्ट को टेक्स्ट निकालने के लिए कॉल कर सकते हैं। यह किसी भी रूप में सभी HTML tages को छीन लेगा और & lt; & Nbsp; सभी एक बार में

Question 10

(<([^>]+)>|&nbsp;)

आप इसे यहाँ देख सकते हैं: https://regex101.com/r/kB0rQ4/1