OpenXML replace text in all document

前端 未结 3 634
一整个雨季
一整个雨季 2021-02-08 15:23

I have the piece of code below. I\'d like replace the text \"Text1\" by \"NewText\", that\'s work. But when I place the text \"Text1\" in a table that\'s not work anymore for th

3条回答
  •  执念已碎
    2021-02-08 15:49

    Borrowing on some other answers in various places, and with the fact that four main obstacles must be overcome:

    1. Delete any high level Unicode chars from your replace string that cannot be read from Word (from bad user input)
    2. Ability to search for your find result across multiple runs or text elements within a paragraph (Word will often break up a single sentence into several text runs)
    3. Ability to include a line break in your replace text so as to insert multi-line text into the document.
    4. Ability to pass in any node as the starting point for your search so as to restrict the search to that part of the document (such as the body, the header, the footer, a specific table, table row, or tablecell).

    I am sure advanced scenarios such as bookmarks, complex nesting will need more modification on this, but it is working for the types of basic word documents I have run into so far, and is much more helpful to me than disregarding runs altogether or using a RegEx on the entire file with no ability to target a specific TableCell or Document part (for advanced scenarios).

    Example Usage:

     var body = document.MainDocumentPart.Document.Body;
     ReplaceText(body, replace, with);
    

    The code:

    using System;
    using System.Collections.Generic;
    using System.Linq;
    using DocumentFormat.OpenXml;
    using DocumentFormat.OpenXml.Packaging;
    using DocumentFormat.OpenXml.Wordprocessing;
    
    namespace My.Web.Api.OpenXml
    {
        public static class WordTools
        {
    
    
    /// 
            /// Find/replace within the specified paragraph.
            /// 
            /// 
            /// 
            /// 
            public static void ReplaceText(Paragraph paragraph, string find, string replaceWith)
            {
                var texts = paragraph.Descendants();
                for (int t = 0; t < texts.Count(); t++)
                {   // figure out which Text element within the paragraph contains the starting point of the search string
                    Text txt = texts.ElementAt(t);
                    for (int c = 0; c < txt.Text.Length; c++)
                    {
                        var match = IsMatch(texts, t, c, find);
                        if (match != null)
                        {   // now replace the text
                            string[] lines = replaceWith.Replace(Environment.NewLine, "\r").Split('\n', '\r'); // handle any lone n/r returns, plus newline.
    
                            int skip = lines[lines.Length - 1].Length - 1; // will jump to end of the replacement text, it has been processed.
    
                            if (c > 0)
                                lines[0] = txt.Text.Substring(0, c) + lines[0];  // has a prefix
                            if (match.EndCharIndex + 1 < texts.ElementAt(match.EndElementIndex).Text.Length)
                                lines[lines.Length - 1] = lines[lines.Length - 1] + texts.ElementAt(match.EndElementIndex).Text.Substring(match.EndCharIndex + 1);
    
                            txt.Space = new EnumValue(SpaceProcessingModeValues.Preserve); // in case your value starts/ends with whitespace
                            txt.Text = lines[0];
    
                            // remove any extra texts.
                            for (int i = t + 1; i <= match.EndElementIndex; i++)
                            {
                                texts.ElementAt(i).Text = string.Empty; // clear the text
                            }
    
                            // if 'with' contained line breaks we need to add breaks back...
                            if (lines.Count() > 1)
                            {
                                OpenXmlElement currEl = txt;
                                Break br;
    
                                // append more lines
                                var run = txt.Parent as Run;
                                for (int i = 1; i < lines.Count(); i++)
                                {
                                    br = new Break();
                                    run.InsertAfter(br, currEl);
                                    currEl = br;
                                    txt = new Text(lines[i]);
                                    run.InsertAfter(txt, currEl);
                                    t++; // skip to this next text element
                                    currEl = txt;
                                }
                                c = skip; // new line
                            }
                            else
                            {   // continue to process same line
                                c += skip;
                            }
                        }
                    }
                }
            }
    
    
    
            /// 
            /// Determine if the texts (starting at element t, char c) exactly contain the find text
            /// 
            /// 
            /// 
            /// 
            /// 
            /// null or the result info
            static Match IsMatch(IEnumerable texts, int t, int c, string find)
            {
                int ix = 0;
                for (int i = t; i < texts.Count(); i++)
                {
                    for (int j = c; j < texts.ElementAt(i).Text.Length; j++)
                    {
                        if (find[ix] != texts.ElementAt(i).Text[j])
                        {
                            return null; // element mismatch
                        }
                        ix++; // match; go to next character
                        if (ix == find.Length)
                            return new Match() { EndElementIndex = i, EndCharIndex = j }; // full match with no issues
                    }
                    c = 0; // reset char index for next text element
                }
                return null; // ran out of text, not a string match
            }
    
            /// 
            /// Defines a match result
            /// 
            class Match
            {
                /// 
                /// Last matching element index containing part of the search text
                /// 
                public int EndElementIndex { get; set; }
                /// 
                /// Last matching char index of the search text in last matching element
                /// 
                public int EndCharIndex { get; set; }
            }
    
         }   // class
    }  // namespace
    
    
    public static class OpenXmlTools
        {
            // filters control characters but allows only properly-formed surrogate sequences
            private static Regex _invalidXMLChars = new Regex(
                @"(?
            /// removes any unusual unicode characters that can't be encoded into XML which give exception on save
            /// 
            public static string RemoveInvalidXMLChars(string text)
            {
                if (string.IsNullOrEmpty(text)) return "";
                return _invalidXMLChars.Replace(text, "");
            }
        }
    

提交回复
热议问题