Back to home page

EIC code displayed by LXR

 
 

    


File indexing completed on 2026-09-28 09:21:02

0001 // Created on: 2013-01-28
0002 // Created by: Kirill GAVRILOV
0003 // Copyright (c) 2013-2014 OPEN CASCADE SAS
0004 //
0005 // This file is part of Open CASCADE Technology software library.
0006 //
0007 // This library is free software; you can redistribute it and/or modify it under
0008 // the terms of the GNU Lesser General Public License version 2.1 as published
0009 // by the Free Software Foundation, with special exception defined in the file
0010 // OCCT_LGPL_EXCEPTION.txt. Consult the file LICENSE_LGPL_21.txt included in OCCT
0011 // distribution for complete text of the license and disclaimer of any warranty.
0012 //
0013 // Alternatively, this file may be used under the terms of Open CASCADE
0014 // commercial license or contractual agreement.
0015 
0016 #ifndef NCollection_UtfString_HeaderFile
0017 #define NCollection_UtfString_HeaderFile
0018 
0019 #include <NCollection_UtfIterator.hxx>
0020 
0021 #include <cstring>
0022 #include <cstdlib>
0023 
0024 //! This template class represent constant UTF-* string.
0025 //! String stored in memory continuously, always NULL-terminated
0026 //! and can be used as standard C-string using ToCString() method.
0027 //!
0028 //! Notice that changing the string is not allowed
0029 //! and any modifications should produce new string.
0030 //!
0031 //! In comments to this class, terms "Unicode symbol" is used as
0032 //! synonym of "Unicode code point".
0033 template <typename Type>
0034 class NCollection_UtfString
0035 {
0036 
0037 public:
0038   NCollection_UtfIterator<Type> Iterator() const { return NCollection_UtfIterator<Type>(myString); }
0039 
0040   //! @return the size of the buffer in bytes, excluding NULL-termination symbol
0041   int Size() const noexcept { return mySize; }
0042 
0043   //! @return the length of the string in Unicode symbols
0044   int Length() const noexcept { return myLength; }
0045 
0046   //! Retrieve Unicode symbol at specified position.
0047   //! Warning! This is a slow access. Iterator should be used for consecutive parsing.
0048   //! @param theCharIndex the index of the symbol, should be lesser than Length()
0049   //! @return the Unicode symbol value
0050   char32_t GetChar(const int theCharIndex) const;
0051 
0052   //! Retrieve string buffer at specified position.
0053   //! Warning! This is a slow access. Iterator should be used for consecutive parsing.
0054   //! @param theCharIndex the index of the symbol, should be less than Length()
0055   //!        (first symbol of the string has index 0)
0056   //! @return the pointer to the symbol
0057   const Type* GetCharBuffer(const int theCharIndex) const;
0058 
0059   //! Retrieve Unicode symbol at specified position.
0060   //! Warning! This is a slow access. Iterator should be used for consecutive parsing.
0061   char32_t operator[](const int theCharIndex) const { return GetChar(theCharIndex); }
0062 
0063   //! Initialize empty string.
0064   NCollection_UtfString();
0065 
0066   //! Copy constructor.
0067   //! @param theCopy string to copy.
0068   NCollection_UtfString(const NCollection_UtfString& theCopy);
0069 
0070   //! Move constructor
0071   NCollection_UtfString(NCollection_UtfString&& theOther) noexcept;
0072 
0073   //! Copy constructor from UTF-8 string.
0074   //! @param theCopyUtf8 UTF-8 string to copy
0075   //! @param theLength   optional length limit in Unicode symbols (NOT bytes!)
0076   //! The string is copied till NULL symbol or, if theLength >0,
0077   //! till either NULL or theLength-th symbol (which comes first).
0078   NCollection_UtfString(const char* theCopyUtf8, const int theLength = -1);
0079 
0080   //! Copy constructor from UTF-16 string.
0081   //! @param theCopyUtf16 UTF-16 string to copy
0082   //! @param theLength    the length limit in Unicode symbols (NOT bytes!)
0083   //! The string is copied till NULL symbol or, if theLength >0,
0084   //! till either NULL or theLength-th symbol (which comes first).
0085   NCollection_UtfString(const char16_t* theCopyUtf16, const int theLength = -1);
0086 
0087   //! Copy constructor from UTF-32 string.
0088   //! @param theCopyUtf32 UTF-32 string to copy
0089   //! @param theLength    the length limit in Unicode symbols (NOT bytes!)
0090   //! The string is copied till NULL symbol or, if theLength >0,
0091   //! till either NULL or theLength-th symbol (which comes first).
0092   NCollection_UtfString(const char32_t* theCopyUtf32, const int theLength = -1);
0093 
0094 #if !defined(_MSC_VER) || defined(_NATIVE_WCHAR_T_DEFINED)                                         \
0095   || (defined(_MSC_VER) && _MSC_VER >= 1900)
0096   //! Copy constructor from wide UTF string.
0097   //! @param theCopyUtfWide wide UTF string to copy
0098   //! @param theLength      the length limit in Unicode symbols (NOT bytes!)
0099   //! The string is copied till NULL symbol or, if theLength >0,
0100   //! till either NULL or theLength-th symbol (which comes first).
0101   //!
0102   //! This constructor is undefined if wchar_t is the same type as char16_t.
0103   NCollection_UtfString(const wchar_t* theCopyUtfWide, const int theLength = -1);
0104 #endif
0105 
0106   //! Copy from Unicode string in UTF-8, UTF-16, or UTF-32 encoding,
0107   //! determined by size of TypeFrom character type.
0108   //! @param theStringUtf Unicode string
0109   //! @param theLength    the length limit in Unicode symbols
0110   //! The string is copied till NULL symbol or, if theLength >0,
0111   //! till either NULL or theLength-th symbol (which comes first).
0112   template <typename TypeFrom>
0113   inline void FromUnicode(const TypeFrom* theStringUtf, const int theLength = -1)
0114   {
0115     NCollection_UtfIterator<TypeFrom> anIterRead(theStringUtf);
0116     if (*anIterRead == 0)
0117     {
0118       // special case
0119       Clear();
0120       return;
0121     }
0122     fromUnicodeImpl(theStringUtf, theLength, anIterRead);
0123   }
0124 
0125   //! Copy from multibyte string in current system locale.
0126   //! @param theString multibyte string
0127   //! @param theLength the length limit in Unicode symbols
0128   //! The string is copied till NULL symbol or, if theLength >0,
0129   //! till either NULL or theLength-th symbol (which comes first).
0130   void FromLocale(const char* theString, const int theLength = -1);
0131 
0132   //! Destructor.
0133   ~NCollection_UtfString();
0134 
0135   //! Compares this string with another one.
0136   bool IsEqual(const NCollection_UtfString& theCompare) const noexcept;
0137 
0138   //! Returns the substring.
0139   //! @param theStart start index (inclusive) of subString
0140   //! @param theEnd   end index   (exclusive) of subString
0141   //! @return the substring
0142   NCollection_UtfString SubString(const int theStart, const int theEnd) const;
0143 
0144   //! Returns NULL-terminated Unicode string.
0145   //! Should not be modified or deleted!
0146   //! @return (const Type* ) pointer to string
0147   const Type* ToCString() const noexcept { return myString; }
0148 
0149   //! @return copy in UTF-8 format
0150   const NCollection_UtfString<char> ToUtf8() const;
0151 
0152   //! @return copy in UTF-16 format
0153   const NCollection_UtfString<char16_t> ToUtf16() const;
0154 
0155   //! @return copy in UTF-32 format
0156   const NCollection_UtfString<char32_t> ToUtf32() const;
0157 
0158   //! @return copy in wide format (UTF-16 on Windows and UTF-32 on Linux)
0159   const NCollection_UtfString<wchar_t> ToUtfWide() const;
0160 
0161   //! Converts the string into string in the current system locale.
0162   //! @param theBuffer    output buffer
0163   //! @param theSizeBytes buffer size in bytes
0164   //! @return true on success
0165   bool ToLocale(char* theBuffer, const int theSizeBytes) const;
0166 
0167   //! @return true if string is empty
0168   bool IsEmpty() const noexcept { return myString[0] == Type(0); }
0169 
0170   //! Zero string.
0171   void Clear();
0172 
0173 public: //! @name assign operators
0174   //! Copy from another string.
0175   const NCollection_UtfString& Assign(const NCollection_UtfString& theOther);
0176 
0177   //! Exchange the data of two strings (without reallocating memory).
0178   void Swap(NCollection_UtfString& theOther) noexcept;
0179 
0180   //! Copy from another string.
0181   const NCollection_UtfString& operator=(const NCollection_UtfString& theOther)
0182   {
0183     return Assign(theOther);
0184   }
0185 
0186   //! Move assignment operator.
0187   NCollection_UtfString& operator=(NCollection_UtfString&& theOther)
0188   {
0189     Swap(theOther);
0190     return *this;
0191   }
0192 
0193   //! Copy from UTF-8 NULL-terminated string.
0194   const NCollection_UtfString& operator=(const char* theStringUtf8);
0195 
0196   //! Copy from wchar_t UTF NULL-terminated string.
0197   const NCollection_UtfString& operator=(const wchar_t* theStringUtfWide);
0198 
0199   //! Join strings.
0200   NCollection_UtfString& operator+=(const NCollection_UtfString& theAppend);
0201 
0202   //! Join two strings.
0203   friend NCollection_UtfString operator+(const NCollection_UtfString& theLeft,
0204                                          const NCollection_UtfString& theRight)
0205   {
0206     NCollection_UtfString aSumm;
0207     strFree(aSumm.myString);
0208     aSumm.mySize   = theLeft.mySize + theRight.mySize;
0209     aSumm.myLength = theLeft.myLength + theRight.myLength;
0210     aSumm.myString = strAlloc(aSumm.mySize);
0211 
0212     // copy bytes
0213     strCopy((uint8_t*)aSumm.myString, (const uint8_t*)theLeft.myString, theLeft.mySize);
0214     strCopy((uint8_t*)aSumm.myString + theLeft.mySize,
0215             (const uint8_t*)theRight.myString,
0216             theRight.mySize);
0217     return aSumm;
0218   }
0219 
0220 public: //! @name compare operators
0221   bool operator==(const NCollection_UtfString& theCompare) const noexcept
0222   {
0223     return IsEqual(theCompare);
0224   }
0225 
0226   bool operator!=(const NCollection_UtfString& theCompare) const noexcept;
0227 
0228 private: //! @name low-level methods
0229   //! Implementation of copy routine for string of the same type
0230   void fromUnicodeImpl(const Type*                    theStringUtf,
0231                        const int                      theLength,
0232                        NCollection_UtfIterator<Type>& theIterator)
0233   {
0234     Type* anOldBuffer = myString; // necessary in case of self-copying
0235 
0236     // advance to the end
0237     const int aLengthMax = (theLength > 0) ? theLength : IntegerLast();
0238     for (; *theIterator != 0 && theIterator.Index() < aLengthMax; ++theIterator)
0239     {
0240     }
0241 
0242     mySize   = int((uint8_t*)theIterator.BufferHere() - (uint8_t*)theStringUtf);
0243     myLength = theIterator.Index();
0244     myString = strAlloc(mySize);
0245     strCopy((uint8_t*)myString, (const uint8_t*)theStringUtf, mySize);
0246 
0247     strFree(anOldBuffer);
0248   }
0249 
0250   //! Implementation of copy routine for string of other types
0251   template <typename TypeFrom>
0252   void fromUnicodeImpl(
0253     typename opencascade::std::enable_if<!opencascade::std::is_same<Type, TypeFrom>::value,
0254                                          const TypeFrom*>::type theStringUtf,
0255     const int                                                   theLength,
0256     NCollection_UtfIterator<TypeFrom>&                          theIterator)
0257   {
0258     Type* anOldBuffer = myString; // necessary in case of self-copying
0259 
0260     mySize               = 0;
0261     const int aLengthMax = (theLength > 0) ? theLength : IntegerLast();
0262     for (; *theIterator != 0 && theIterator.Index() < aLengthMax; ++theIterator)
0263     {
0264       mySize += theIterator.template AdvanceBytesUtf<Type>();
0265     }
0266     myLength = theIterator.Index();
0267 
0268     myString = strAlloc(mySize);
0269 
0270     // copy string
0271     theIterator.Init(theStringUtf);
0272     Type* anIterWrite = myString;
0273     for (; *theIterator != 0 && theIterator.Index() < myLength; ++theIterator)
0274     {
0275       anIterWrite = theIterator.GetUtf(anIterWrite);
0276     }
0277 
0278     strFree(anOldBuffer);
0279   }
0280 
0281   //! Allocate NULL-terminated string buffer.
0282   static Type* strAlloc(const size_t theSizeBytes)
0283   {
0284     Type* aPtr = (Type*)Standard::Allocate(theSizeBytes + sizeof(Type));
0285     if (aPtr != nullptr)
0286     {
0287       // always NULL-terminate the string
0288       aPtr[theSizeBytes / sizeof(Type)] = Type(0);
0289     }
0290     return aPtr;
0291   }
0292 
0293   //! Release string buffer and nullify the pointer.
0294   static void strFree(Type*& thePtr) { Standard::Free(thePtr); }
0295 
0296   //! Provides bytes interface to avoid incorrect pointer arithmetics.
0297   static void strCopy(uint8_t* theStrDst, const uint8_t* theStrSrc, const int theSizeBytes) noexcept
0298   {
0299     std::memcpy(theStrDst, theStrSrc, (size_t)theSizeBytes);
0300   }
0301 
0302   //! Compare two Unicode strings per-byte.
0303   static bool strAreEqual(const Type* theString1,
0304                           const int   theSizeBytes1,
0305                           const Type* theString2,
0306                           const int   theSizeBytes2) noexcept
0307   {
0308     return (theSizeBytes1 == theSizeBytes2)
0309            && (std::memcmp(theString1, theString2, (size_t)theSizeBytes1) == 0);
0310   }
0311 
0312 private:          //! @name private fields
0313   Type* myString; //!< string buffer
0314   int   mySize;   //!< buffer size in bytes, excluding NULL-termination symbol
0315   // clang-format off
0316   int myLength; //!< length of the string in Unicode symbols (cached value, excluding NULL-termination symbol)
0317   // clang-format on
0318 };
0319 
0320 // template implementation (inline methods)
0321 #include <NCollection_UtfString.lxx>
0322 
0323 #endif // _NCollection_UtfString_H__