StringTokenizer สำหรับ. NET Core

Sep 18 2020

การแยกสตริงในโทเค็นเป็นหัวข้อที่ซับซ้อนกว่าString.Split ()ต้องการทำให้เราเชื่อ มีนโยบายทั่วไปอย่างน้อยสามนโยบายที่อาจตีความและแยกสตริงในโทเค็น

นโยบาย 1: เทียบเท่ากับ String.Split ()

ไม่ค่อยมีใครพูดถึงเกี่ยวกับนโยบายนี้ กำหนดสตริงsและตัวคั่นdแบ่งออกเป็นส่วนคั่นด้วยs dข้อเสียเปรียบหลักคือหากตัวคั่นเป็นส่วนหนึ่งของโทเค็นอย่างน้อยหนึ่งโทเค็นการสร้างโทเค็นที่ต้องการใหม่อาจมีค่าใช้จ่ายสูง

นโยบาย 2: หลีกเลี่ยงอักขระพิเศษ

อักขระถูกประกาศเป็นอักขระหลีก e (โดยทั่วไปคือเครื่องหมายแบ็กสแลช\) ส่งผลให้อักขระที่ตามหลังสูญเสียความหมายพิเศษ สตริงโทเค็นอาจมีลักษณะดังนี้:

token_1 token_2 very\ long \ token

ซึ่งจะเทียบเท่ากับ

{ "token_1", "token_2", "very long token" }

นโยบาย 3: วางโทเค็นในเครื่องหมายคำพูด

วิธีนี้เป็นตัวอย่างที่ใช้ในไฟล์ CSV ที่สร้างใน MSExcel ทุกสิ่งที่อยู่ระหว่างเครื่องหมายคำพูดถือเป็นโทเค็น หากเครื่องหมายคำพูดเป็นส่วนหนึ่งของโทเค็นที่พวกเขาเป็นสองเท่า" ""สตริงโทเค็นอาจมีลักษณะดังนี้:

token_1,token_2,"token2,5"

ซึ่งจะเทียบเท่ากับ

{ "token_1", "token_2", "token2,5" }

รหัส

using System;
using System.Text;
using System.Text.RegularExpressions;
using System.Collections.Generic;

namespace Pillepalle1.ConsoleTelegramBot.Model.Misc
{
    public sealed class StringTokenizer
    {
        private string _sourceString = null;                            // Provided data to split

        #region Constructors
        /// <summary>
        /// Creates a new StringTokenizer
        /// </summary>
        /// <param name="dataInput">Data to be split into tokens</param>
        public StringTokenizer(string dataInput)
        {
            _sourceString = dataInput ?? string.Empty;
        }
        #endregion

        #region Interface
        /// <summary>
        /// Access tokens by index
        /// </summary>
        public string this[int index]
        {
            get
            {
                if (index >= this.Count)
                {
                    return String.Empty;
                }

                return _Tokens[index];
            }
        }

        /// <summary>
        /// How many tokens does the command consist of
        /// </summary>
        public int Count
        {
            get
            {
                return _Tokens.Count;
            }
        }

        /// <summary>
        /// Which strategy is used to split the string into tokens
        /// </summary>
        public StringTokenizerStrategy Strategy
        {
            get
            {
                return _strategy;
            }
            set
            {
                if (value != _strategy)
                {
                    _strategy = value;
                    _tokens = null;
                }
            }
        }
        private StringTokenizerStrategy _strategy = StringTokenizerStrategy.Split;

        /// <summary>
        /// Character used to delimit tokens
        /// </summary>
        public char Delimiter
        {
            get
            {
                return _delimiter;
            }
            set
            {
                if (value != _delimiter)
                {
                    _delimiter = value;
                    _tokens = null;
                }
            }
        }
        private char _delimiter = ' ';

        /// <summary>
        /// Character used to escape the following character
        /// </summary>
        public char Escape
        {
            get
            {
                return _escape;
            }
            set
            {
                if (value != _escape)
                {
                    _escape = value;

                    if (Strategy == StringTokenizerStrategy.Escaping)
                    {
                        _tokens = null;
                    }
                }
            }
        }
        private char _escape = '\\';

        /// <summary>
        /// Character used to surround tokens
        /// </summary>
        public char Quotes
        {
            get
            {
                return _quotes;
            }
            set
            {
                if (value != _quotes)
                {
                    _quotes = value;

                    if (Strategy == StringTokenizerStrategy.Quotation)
                    {
                        _tokens = null;
                    }
                }
            }
        }
        private char _quotes = '"';
        #endregion

        #region Predefined Regex
        private Regex Whitespaces
        {
            get
            {
                return new Regex("\\s+");
            }
        }
        #endregion

        #region Implementation Details
        /// <summary>
        /// Formats and splits the tokens by delimiter allowing to add delimiters by quoting
        /// </summary>
        private List<string> _SplitRespectingQuotation()
        {
            string data = _sourceString;

            // Doing some basic transformations
            data = Whitespaces.Replace(data, " ");

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Initialisation
            List<string> l = new List<string>();
            char[] record = data.ToCharArray();

            StringBuilder property = new StringBuilder();
            char c;

            bool quoting = false;

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Scan character by character
            for (int i = 0; i < record.Length; i++)
            {
                c = record[i];

                // Quotation-Character: Single -> Quote; Double -> Append
                if (c == Quotes)
                {
                    if (i == record.Length - 1)
                    {
                        quoting = !quoting;
                    }
                    else if (Quotes == record[1 + i])
                    {
                        property.Append(c);
                        i++;
                    }
                    else
                    {
                        quoting = !quoting;
                    }
                }

                // Delimiter: Escaping -> Append; Otherwise append
                else if (c == Delimiter)
                {
                    if (quoting)
                    {
                        property.Append(c);
                    }
                    else
                    {
                        l.Add(property.ToString());
                        property.Clear();
                    }
                }

                // Any other character: Append
                else
                {
                    property.Append(c);
                }
            }

            l.Add(property.ToString());                         // Add last token

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Checking consistency
            if (quoting) throw new FormatException();          // All open quotation marks closed

            return l;
        }

        /// <summary>
        /// Splits the string by declaring one character as escape
        /// </summary>
        private List<string> _SplitRespectingEscapes()
        {
            string data = _sourceString;

            // Doing some basic transformations
            data = Whitespaces.Replace(data, " ");

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Initialisation
            List<string> l = new List<string>();
            char[] record = data.ToCharArray();

            StringBuilder property = new StringBuilder();
            char c;

            bool escaping = false;

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Scan character by character
            for (int i = 0; i < record.Length; i++)
            {
                c = record[i];

                if (escaping)
                {
                    property.Append(c);
                    escaping = false;
                    continue;
                }

                if (c == Escape)
                {
                    escaping = true;
                }
                else if (c == Delimiter)
                {
                    l.Add(property.ToString());
                    property.Clear();
                }
                else
                {
                    property.Append(c);
                }
            }

            return l;
        }

        /// <summary>
        /// Splits the string by calling a simple String.Split
        /// </summary>
        private List<string> _SplitPlain()
        {
            return new List<string>(Whitespaces.Replace(_sourceString, " ").Split(Delimiter));
        }

        /// <summary>
        /// Backer for tokens
        /// </summary>
        private List<string> _Tokens
        {
            get
            {
                if (null == _tokens)
                {
                    switch (Strategy)
                    {
                        case (StringTokenizerStrategy.Quotation): _tokens = _SplitRespectingQuotation(); break;
                        case (StringTokenizerStrategy.Escaping): _tokens = _SplitRespectingEscapes(); break;

                        default: _tokens = _SplitPlain(); break;
                    }
                }

                return _tokens;
            }
        }
        private List<string> _tokens = null;
        #endregion
    }

    public enum StringTokenizerStrategy
    {
        Split,
        Quotation,
        Escaping
    }
}

คำตอบ

7 TheRubberDuck Sep 19 2020 at 01:42

ฉันไม่แน่ใจว่าสิ่งนี้อยู่ในชั้นเรียน - อย่างน้อยก็ไม่ใช่คนเดียว!

ย้อนกลับไปดูว่าอะไรที่รวมเข้าด้วยกันและอะไรที่แยก "กลยุทธ์" แต่ละอย่างออกจากกัน พวกเขาทั้งหมดต้องแปลงสตริงอินพุตเป็นรายการโทเค็นตามตัวคั่นตัวแปร อย่างไรก็ตามมีคุณสมบัติที่ใช้โดยหนึ่งในสามตัวเลือกเท่านั้นและตรรกะการแยกส่วนใหญ่เป็นลักษณะเฉพาะของกลยุทธ์

คำแนะนำ 1: ฟังก์ชัน "สแตนด์อโลน" สามฟังก์ชัน

คุณจะต้องนำพวกเขาไปอยู่ในคลาสคงที่หรือทำอะไรเป็นพิเศษกับผู้ร่วมประชุม / แลมบ์ดาส แต่ในที่สุดก็มีเพียงเล็กน้อยที่จะได้รับจากการมีคลาสใหญ่ ๆ

  public static IList<string> SplitRespectingQuotation(string sourceString, char delimiter = ' ', char quote = '"') { ... }
  public static IList<string> SplitRespectingEscapes(string sourceString, char delimiter = ' ', char escape = '\') { ... }
  public static IList<string> SplitPlain(string sourceString, char delimiter = ' ') { ... }

หากคุณต้องการให้เอาต์พุตสื่อสารพารามิเตอร์อินพุตคุณสามารถสร้างคลาสที่มีน้ำหนักเบากว่านี้ได้มาก คุณสมบัติของมันจะเป็นreadonly; หากคุณต้องการเปลี่ยนและคำนวณใหม่เพียงแค่เรียกใช้ฟังก์ชันอีกครั้ง ท้ายที่สุดนั่นคือสิ่งที่คุณกำลังทำอยู่ในชั้นเรียนปัจจุบันของคุณ!

ข้อดีอีกอย่าง: ถ้าคุณคิดกลยุทธ์ใหม่ในการแยกคุณสามารถสร้างฟังก์ชันใหม่ได้โดยไม่ส่งผลกระทบต่อผู้อื่น ทั้งหมดนี้สามารถทดสอบแก้ไขและลบได้อย่างอิสระ

ข้อเสนอแนะที่ 2: คลาสคอนกรีตสามชั้นที่ขยายคลาสฐานนามธรรม

ฉันชอบสิ่งที่คุณทำกับ_Tokensทรัพย์สิน: ช่วยให้คุณเลื่อนการคำนวณไปจนกว่าคุณจะต้องการจริงๆซึ่งจะเป็นประโยชน์ในกรณีที่คุณไม่ต้องการ นอกจากนี้กรณีการใช้งานหนึ่งที่รองรับ (ซึ่งไม่รองรับโดยฟังก์ชัน "สแตนด์อโลน") คือการเปลี่ยนเช่นอักขระหลีกและให้ผลลัพธ์เป็น "ไม่ถูกต้อง" โดยอัตโนมัติ

ในการรักษาพฤติกรรมดังกล่าวคุณสามารถดึงองค์ประกอบทั่วไปเข้าสู่คลาสฐานนามธรรมได้ดังต่อไปนี้:

public abstract class StringTokenizer
{
  public string SourceString { get; }

  public StringTokenizer(string dataInput)
  {
    SourceString = dataInput ?? string.Empty;
  }

  public string this[int index] => index >= this.Count ? String.Empty : Tokens[index];

  public int Count => Tokens.Count;

  public char Delimiter
  {
    get { return _delimiter; }
    set
    {
      if (value != _delimiter)
      {
         _delimiter = value;
         InvalidateResult();
      }
    }
  }
  private char _delimiter = ' ';

  public IEnumerable<string> Tokens
  {
    get
    {
      if (_tokens is null)
      {
        _tokens = ComputeTokens();
      }
      return _tokens;
    }
  }
  private List<string> _tokens = null;

  protected abstract List<string> ComputeTokens();

  protected void InvalidateResult()
  {
    _tokens = null;
  }
}

การเปลี่ยนแปลงที่สำคัญ:

  1. ไม่มีตรรกะการแยกที่แท้จริง แต่ละกลยุทธ์จะให้ของตัวเอง
  2. ไม่มีคุณสมบัติเฉพาะกลยุทธ์ ไม่จำเป็นต้องมีกลยุทธ์หลบหนีเพื่อให้มีคุณสมบัติสำหรับอักขระอัญประกาศและในทางกลับกัน
  3. แทนการตั้งค่าโดยตรงคุณสมบัติควรจะเรียก_tokens = null InvalidateResultสิ่งนี้อนุญาตให้_tokensสร้างขึ้นprivateซึ่งทำให้ตรรกะมีอยู่ในคลาสฐาน
  4. Tokensเป็นสาธารณะและเป็นIEnumerableไฟล์. ทำให้ผู้บริโภคใช้งานforeachได้ แต่ไม่สนับสนุนการดัดแปลงโดยตรง

ชั้นฐานตอนนี้มีอีกหนึ่งงาน: ComputeTokensดำเนินการ หากจำเป็นต้องสร้างคุณสมบัติเพื่อทำเช่นนั้นก็อาจทำได้โดยอาศัยตรรกะเฉพาะกลยุทธ์ของตัวเอง หากคุณสมบัติเหล่านั้นจำเป็นต้องทำให้โทเค็นที่คำนวณก่อนหน้านี้เป็นโมฆะเมื่อมีการเปลี่ยนแปลงคุณสมบัติเหล่านั้นอาจเรียกInvalidateResultใช้

นี่คือตัวอย่างคร่าวๆว่าคลาสย่อยของกลยุทธ์จะมีลักษณะอย่างไร:

public sealed class EscapeStringTokenizer : StringTokenizer
{
  public EscapeStringTokenizer (string dataInput) : base(dataInput) { }

  public char Escape
  {
    get { return _escape; }
    set
    {
      if (value != _escape)
      {
         _escape = value;
         InvalidateResult();
      }
    }
  }

  protected override List<string> ComputeTokens()
  {
    // Actual logic omitted
  }
}

ข้อสังเกตอื่น ๆ

  1. คุณอนุญาตให้ระบุตัวคั่น แต่คุณจะย่อช่องว่างเสมอ ถ้าผมแยก"a,a and b,b"กับตัวคั่นของ","ผมจะคาดหวังที่จะได้รับ{"a", "a and b", "b"}กลับ - แต่จริง ๆ {"a", "a and b", "b"}แล้วจะได้รับ
  2. หากตัวคั่น ฯลฯ สามารถอ่านแบบสาธารณะได้ทำไมไม่เปิดเผยสตริงต้นทางด้วย ดูSourceStringคลาสนามธรรมของฉันด้านบน
  3. ฉันพบว่าตัวเข้าถึงคุณสมบัติที่เป็นนิพจน์ (ค่อนข้างใหม่) นั้นดีกว่าสำหรับคุณสมบัติทั่วไป ดูCountในชั้นเรียนนามธรรมของฉันด้านบน
  4. ฉันไม่คิดว่าเป็นไปได้ที่จะกำหนดnullให้ตัวแปรเป็นเงื่อนไขของคำสั่ง if โดยไม่ได้ตั้งใจ เนื่องจากการx = nullประเมินเป็นประเภทเดียวกับxซึ่งจำเป็นต้องเป็น a bool(ดังนั้นจึงไม่เป็นโมฆะ) เพื่อให้เป็นเงื่อนไขที่ถูกต้อง หากคุณยังต้องการหลีกเลี่ยงx == nullคุณสามารถพูดx is nullได้
  5. เป็นที่กล่าวถึงโดยคนอื่น ๆ _ที่คุณไม่ควรนำหน้าด้วยคุณสมบัติ มันไม่ได้มีไว้เพื่อแยกความแตกต่างระหว่างสาธารณะและส่วนตัว แต่เป็นระหว่างตัวแปรโลคัลและฟิลด์คลาส โดยส่วนตัวแล้วฉันไม่ได้ใช้_ในกรณีนั้น แต่จะชอบมากกว่าthis.ถ้าจำเป็น แต่โดยรวมแล้วคุณจะต้องมีความยืดหยุ่นเกี่ยวกับเรื่องนี้และตรวจสอบให้แน่ใจว่าได้ทำตามรูปแบบที่กำหนดไว้แล้วในทีมหรือโครงการที่มีอยู่
  6. เช่นเดียวกับที่คนอื่น ๆ กล่าวถึงให้ใช้varเมื่อประกาศตัวแปรทุกครั้งที่ทำได้ IDE ที่ดีใด ๆ จะสามารถบอกประเภทของคุณได้เมื่อคุณวางเมาส์เหนือตัวแปรและชื่อของมันควรจะบอกคุณว่ามันมีไว้เพื่ออะไรแม้ว่าจะไม่มีประเภทก็ตาม
  7. เมื่อทราบว่าชื่อหลีกเลี่ยงชอบและc เป็นเรื่องปกติเพราะสำนวนเป็นตัวแปรลูป / ดัชนี แต่ตัวแปรอื่น ๆ ต้องการบริบทเพิ่มเติมเพื่อทำความเข้าใจ ตัวละครที่มารหัสที่มีราคาถูกเพื่อให้การจ่ายเงินบางส่วนให้สามารถอ่านได้พิเศษโดยการใช้และlicurrentCharfinishedTokens
  8. คุณไม่จำเป็นต้องแปลแหล่งที่มาstringเป็นchar[]; คุณสามารถเข้าถึงอักขระในstringดัชนีโดย
9 Heslacher Sep 18 2020 at 16:53
  • คุณไม่ควรมีWhitespacesเป็นคุณสมบัติรับอย่างเดียว แต่เป็นreadonlyฟิลด์ส่วนตัวและคุณควรรวบรวม regex นั้นไว้เนื่องจากคุณใช้งานบ่อยมาก

  • การใช้regionถือเป็นการป้องกันโรค

  • ใช้เครื่องหมายขีดล่าง - นำหน้าสำหรับฟิลด์ส่วนตัวเท่านั้น อย่าใช้สำหรับวิธีการหรือคุณสมบัติ

  • หากประเภทของตัวแปรชัดเจนจากด้านขวามือของงานคุณควรใช้varแทนประเภทคอนกรีต

  • รหัสจะทำ althought มาก_sourceStringอาจจะเป็นstring.Emptyเพราะที่ผ่าน ctor อาร์กิวเมนต์dataInputอาจจะเป็นหรือnull string.Emptyฉันต้องการโยนข้อยกเว้นในไฟล์ctor.

  • แทนที่จะกำหนดตัวแปรให้กับตัวแปรอื่นแล้วจัดการตัวแปรที่เป็นผลลัพธ์คุณสามารถทำได้ในบรรทัดเดียวเช่นเช่น

    string data = Whitespaces.Replace(_sourceString, " ");  
    

    แทน

    string data = _sourceString;
    
    // Doing some basic transformations
    data = Whitespaces.Replace(data, " ");  
    
  • หากคุณต้องการเข้าถึงเพียงรายการเดียวของอาร์เรย์และไม่จำเป็นต้องมองไปข้างหน้าคุณควรเลือกแบบforeachโอเวอร์forลูป

5 AlexanderPetrov Sep 18 2020 at 19:51
  • lชื่อที่เป็นตัวอักษรตัวเดียวดูเหมือนจะไม่ดีสำหรับฉัน

  • ฉันคิดว่าคุณควรเพิ่มข้อความในข้อยกเว้นที่อธิบายสาเหตุของข้อผิดพลาด

  • โดยค่าเริ่มต้นคุณจะลบช่องว่างทั้งหมดออกจากข้อมูล แต่อาจจำเป็นต้องใช้ภายในโทเค็น คุณสามารถสร้างตัวเลือกเพิ่มเติมเพื่อระบุสิ่งนี้

1 Benj Sep 20 2020 at 00:13

ขอบคุณทุกคนสำหรับข้อเสนอแนะที่ดี ฉันได้นำการเปลี่ยนแปลงส่วนใหญ่มาใช้กับโค้ดของฉันซึ่งโฮสต์เป็น FOS บนhttps://github.com/pillepalle1/dotnet-pillepalle1 ซึ่งจะได้รับการบำรุงรักษาเพิ่มเติม

สำหรับตอนนี้ฉันได้บรรจุตรรกะการแยกออกเป็นวิธีการขยายแบบคงที่สามวิธี นอกจากนี้ฉันได้สร้างเครื่องห่อหุ้มตามคำแนะนำของtherubberduckเพื่อเป็นทางเลือกที่จะรักษาความสะดวกสบายของการยกเลิกโทเค็นอัตโนมัติ

ข้อเสนอแนะที่ฉันได้ดำเนินการ

  • การตั้งชื่อตัวแปรชื่อตัวแปรเช่นlถูกแทนที่ด้วยชื่อที่สื่อความหมายได้มากขึ้น

  • มีการเพิ่มข้อความยกเว้น

  • การแก้ไขเนื้อหาโทเค็นถูกลบออกจากวิธีการส่วนขยายอย่างสมบูรณ์และทำให้สามารถเลือกใช้ได้ใน Wrapper

  • ภูมิภาคต่างๆถูกลบออกทั้งหมด

  • ใช้ varเมื่อใดก็ตามที่สมเหตุสมผล / เป็นไปได้

  • ลูปชอบforeachมากกว่าforลูปและวนซ้ำsourceStringแทนการแปลงเป็นchar[]ครั้งแรก

  • Inputstring Throwing ArgumentNullExceptionแทนที่จะแปลงnullเป็นString.Empty

  • CSV Splittingตาม RFC4180

ฉันจะนำการเปลี่ยนแปลงมาใช้มากขึ้น แต่คำแนะนำบางอย่าง (เช่นเกี่ยวกับWhitespacesและคุณสมบัติของร่างกายที่แสดงออก) ล้าสมัยในการใช้งานใหม่

ข้อเสนอแนะฉันยังไม่ได้ดำเนินการ

  • การตั้งชื่อขีดล่างสำหรับทุกสิ่งที่เป็นส่วนตัว / ได้รับการป้องกันดูเหมือนจะสมเหตุสมผลสำหรับฉันมากกว่าการแยกแยะระหว่างตัวแปรสมาชิกและท้องถิ่นตั้งแต่เมื่อใช้โครงสร้างข้อมูลที่มีประสิทธิภาพพร้อมกัน (ซึ่งกลายเป็นเรื่องใหญ่นับตั้งแต่Tasksมีการใช้งาน) มันมีคุณค่าอย่างเหลือเชื่อที่จะเห็นได้ในแวบแรกว่า a วิธีดำเนินการตรวจสอบการทำงานพร้อมกัน (สาธารณะ) หรือไม่ (ส่วนตัว)

รหัส

วิธี Tokenizer แบบคงที่

using System;
using System.Text;
using System.Collections.Immutable;

namespace pillepalle1.Text
{
    public static class StringTokenizer
    {
        private static FormatException _nonQuotedTokenMayNotContainQuotes =
            new FormatException("[RFC4180] If fields are not enclosed with double quotes, then double quotes may not appear inside the fields.");

        private static FormatException _quotesMustBeEscapedException =
            new FormatException("[RFC4180] If double-quotes are used to enclose fields, then a double-quote appearing inside a field must be escaped by preceding it with another double quote.");

        private static FormatException _tokenNotFullyEnclosed =
            new FormatException("[RFC4180] \"Each field may or may not be enclosed in double quotes\". However, for the final field the closing quotes are missing.");


        /// <summary>
        /// <para>
        /// Formats and splits the tokens by delimiter allowing to add delimiters by quoting 
        /// similar to https://tools.ietf.org/html/rfc4180
        /// </para>
        /// 
        /// <para>
        /// Each field may or may not be enclosed in double quotes (however some programs, such as 
        /// Microsoft Excel, do not use double quotes at all). If fields are not enclosed with 
        /// double quotes, then double quotes may not appear inside the fields.
        /// </para>
        /// 
        /// <para>
        /// Fields containing line breaks (CRLF), double quotes, and commas should be enclosed in 
        /// double-quotes.
        /// </para>
        /// 
        /// <para>
        /// If double-quotes are used to enclose fields, then a double-quote appearing inside a 
        /// field must be escaped by preceding it with another double quote.
        /// </para>
        /// 
        /// <para>
        /// The ABNF defines 
        /// 
        /// [field = (escaped / non-escaped)] ||  
        /// [non-escaped = *TEXTDATA] || 
        /// [TEXTDATA =  %x20-21 / %x23-2B / %x2D-7E]
        /// 
        /// specifically forbidding to include quotes in non-escaped fields, hardening the *SHOULD*
        /// requirement above.
        /// </para>
        /// </summary>
        public static ImmutableList<string> SplitRespectingQuotation(this string sourceString, char delimiter = ' ', char quotes = '"')
        {
            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Initialisation
            var tokenList = ImmutableList<string>.Empty;
            var tokenBuilder = new StringBuilder();

            var expectingDelimiterOrQuotes = false;     // Next char must be Delimiter or Quotes
            var hasReadTokenChar = false;               // We are not between tokens (=> No quoting)
            var isQuoting = false;

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Scan character by character
            foreach (char c in sourceString)
            {
                if (expectingDelimiterOrQuotes)
                {
                    expectingDelimiterOrQuotes = false;

                    if (c == delimiter)
                    {
                        isQuoting = false;
                    }

                    else if (c == quotes)
                    {
                        tokenBuilder.Append(c);
                        hasReadTokenChar = true;
                        continue;
                    }

                    else
                    {
                        throw _quotesMustBeEscapedException;
                    }
                }

                // -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- -- --

                if (c == quotes)
                {
                    if (isQuoting)
                    {
                        expectingDelimiterOrQuotes = true;
                    }

                    else
                    {
                        if (hasReadTokenChar)
                        {
                            throw _nonQuotedTokenMayNotContainQuotes;
                        }

                        isQuoting = true;
                    }
                }

                else if (c == delimiter)
                {
                    if (isQuoting)
                    {
                        tokenBuilder.Append(c);
                        hasReadTokenChar = true;
                    }
                    else
                    {
                        tokenList = tokenList.Add(tokenBuilder.ToString());
                        tokenBuilder.Clear();
                        hasReadTokenChar = false;
                    }
                }

                // Any other character is just being appended to
                else
                {
                    tokenBuilder.Append(c);
                    hasReadTokenChar = true;
                }
            }

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Tidy up open flags and checking consistency

            tokenList = tokenList.Add(tokenBuilder.ToString());

            if (isQuoting && !expectingDelimiterOrQuotes)
            {
                throw _tokenNotFullyEnclosed;
            }

            return tokenList;
        }

        /// <summary>
        /// Splits the string by declaring one character as escape
        /// </summary>
        public static ImmutableList<string> SplitRespectingEscapes(this string sourceString, char delimiter = ' ', char escapeChar = '\\')
        {
            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Initialisation
            var tokenList = ImmutableList<string>.Empty;
            var tokenBuilder = new StringBuilder();

            var escapeNext = false;

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Scan character by character
            foreach (char c in sourceString)
            {
                if (escapeNext)
                {
                    tokenBuilder.Append(c);
                    escapeNext = false;
                    continue;
                }

                if (c == escapeChar)
                {
                    escapeNext = true;
                }
                else if (c == delimiter)
                {
                    tokenList = tokenList.Add(tokenBuilder.ToString());
                    tokenBuilder.Clear();
                }
                else
                {
                    tokenBuilder.Append(c);
                }
            }

            // - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 
            // Tidy up open flags and checking consistency
            tokenList = tokenList.Add(tokenBuilder.ToString());

            if (escapeNext) throw new FormatException();            // Expecting additional char


            return tokenList;
        }

        /// <summary>
        /// Splits the string by calling a simple String.Split
        /// </summary>
        public static ImmutableList<string> SplitPlain(this string sourceString, char delimiter = ' ')
        {
            return ImmutableList<string>.Empty.AddRange(sourceString.Split(delimiter));
        }
    }
}

คลาสฐานกระดาษห่อนามธรรม

using System;
using System.Collections.Immutable;

namespace pillepalle1.Text
{
    public abstract class AStringTokenizer
    {
        public AStringTokenizer()
        {

        }

        public AStringTokenizer(string sourceString)
        {
            SourceString = sourceString;
        }

        /// <summary>
        /// String that is supposed to be split in tokens
        /// </summary>
        public string SourceString
        {
            get
            {
                return _sourceString;
            }
            set
            {
                if (null == value)
                {
                    throw new ArgumentNullException("Cannot split null in tokens");
                }
                else if (_sourceString.Equals(value))
                {
                    // nop
                }
                else
                {
                    _sourceString = value;
                    _InvalidateTokens();
                }
            }
        }
        private string _sourceString = String.Empty;

        /// <summary>
        /// Character indicating how the source string is supposed to be split
        /// </summary>
        public char Delimiter
        {
            get
            {
                return _delimiter;
            }
            set
            {
                if (value != _delimiter)
                {
                    _delimiter = value;
                    _InvalidateTokens();
                }
            }
        }
        private char _delimiter = ' ';

        /// <summary>
        /// Flag indicating whether whitespaces should be removed from start and end of each token
        /// </summary>
        public bool TrimTokens
        {
            get
            {
                return _trimTokens;
            }
            set
            {
                if (value != _trimTokens)
                {
                    _trimTokens = value;
                    _InvalidateTokens();
                }
            }
        }
        private bool _trimTokens = false;

        /// <summary>
        /// Result of tokenization
        /// </summary>
        public ImmutableList<string> Tokens
        {
            get
            {
                if (null == _tokens)
                {
                    _tokens = Tokenize();

                    if (TrimTokens)
                    {
                        _tokens = _TrimTokens(_tokens);
                    }
                }

                return _tokens;
            }
        }
        private ImmutableList<string> _tokens = null;

        /// <summary>
        /// Split SourceString into tokens
        /// </summary>
        protected abstract ImmutableList<string> Tokenize();

        /// <summary>
        /// Trims whitespaces from tokens
        /// </summary>
        /// <param name="candidates">List of tokens</param>
        private ImmutableList<string> _TrimTokens(ImmutableList<string> candidates)
        {
            var trimmedTokens = ImmutableList<string>.Empty;

            foreach (var token in candidates)
            {
                trimmedTokens = trimmedTokens.Add(token.Trim());
            }

            return trimmedTokens;
        }

        /// <summary>
        /// Invalidate and recompute tokens if necessary
        /// </summary>
        protected void _InvalidateTokens()
        {
            _tokens = null;
        }
    }
}

Wrapper สำหรับโทเค็นธรรมดา

using System.Collections.Immutable;

namespace pillepalle1.Text
{
    public class PlainStringTokenizer : AStringTokenizer
    {
        protected override ImmutableList<string> Tokenize()
        {
            return SourceString.SplitPlain(Delimiter);
        }
    }
}

Wrapper สำหรับโทเค็นใบเสนอราคา

using System.Collections.Immutable;

namespace pillepalle1.Text
{
    public class QuotationStringTokenizer : AStringTokenizer
    {
        /// <summary>
        /// Indicates which character is used to encapsulate tokens
        /// </summary>
        public char Quotes
        {
            get
            {
                return _quotes;
            }
            set
            {
                if (value != _quotes)
                {
                    _quotes = value;
                    _InvalidateTokens();
                }
            }
        }
        private char _quotes = '"';

        protected override ImmutableList<string> Tokenize()
        {
            return SourceString.SplitRespectingQuotation(Delimiter, Quotes);
        }
    }
}

Wrapper สำหรับ Escape tokenization

using System.Collections.Immutable;

namespace pillepalle1.Text
{
    public class EscapedStringTokenizer : AStringTokenizer
    {
        /// <summary>
        /// Indicates which character is used to escape characters
        /// </summary>
        public char Escape
        {
            get
            {
                return _escape;
            }
            set
            {
                if (value != _escape)
                {
                    _escape = value;
                    _InvalidateTokens();
                }
            }
        }
        private char _escape = '"';

        protected override ImmutableList<string> Tokenize()
        {
            return SourceString.SplitRespectingEscapes(Delimiter, Escape);
        }
    }
}