// MarkupSTL.cpp: implementation of the CMarkupSTL class. // // Markup Release 7.3 // Copyright (C) 1999-2004 First Objective Software, Inc. All rights reserved // Go to www.firstobject.com for the latest CMarkup and EDOM documentation // Use in commercial applications requires written permission // This software is provided "as is", with no warranty. #include #include #include #include "MarkupSTL.h" using namespace std; // Customization #define x_EOL "\r\n" // can be \r\n or \n or empty #define x_EOLLEN (sizeof(x_EOL)-1) // string length of x_EOL #define x_ATTRIBQUOTE "\"" // can be double or single quote void CMarkupSTL::operator=( const CMarkupSTL& markup ) { m_iPosParent = markup.m_iPosParent; m_iPos = markup.m_iPos; m_iPosChild = markup.m_iPosChild; m_iPosFree = markup.m_iPosFree; m_iPosDeleted = markup.m_iPosDeleted; m_nNodeType = markup.m_nNodeType; m_nNodeOffset = markup.m_nNodeOffset; m_nNodeLength = markup.m_nNodeLength; m_strDoc = markup.m_strDoc; m_strError = markup.m_strError; m_nFlags = markup.m_nFlags; // Copy used part of the index array m_aPos.RemoveAll(); m_aPos.nSize = m_iPosFree; if ( m_aPos.nSize < 8 ) m_aPos.nSize = 8; m_aPos.nSegs = m_aPos.SegsUsed(); if ( m_aPos.nSegs ) { m_aPos.pSegs = (ElemPos**)(new char[m_aPos.nSegs*sizeof(char*)]); int nSegSize = 1 << m_aPos.PA_SEGBITS; for ( int nSeg=0; nSeg < m_aPos.nSegs; ++nSeg ) { if ( nSeg + 1 == m_aPos.nSegs ) nSegSize = m_aPos.GetSize() - (nSeg << m_aPos.PA_SEGBITS); m_aPos.pSegs[nSeg] = (ElemPos*)(new char[nSegSize*sizeof(ElemPos)]); memcpy( m_aPos.pSegs[nSeg], markup.m_aPos.pSegs[nSeg], nSegSize*sizeof(ElemPos) ); } } // Copy SavedPos map m_mapSavedPos.RemoveAll(); if ( markup.m_mapSavedPos.pTable ) { m_mapSavedPos.AllocMapTable(); for ( int nSlot=0; nSlot < SavedPosMap::SPM_SIZE; ++nSlot ) { SavedPos* pCopySavedPos = markup.m_mapSavedPos.pTable[nSlot]; if ( pCopySavedPos ) { int nCount = 0; while ( pCopySavedPos[nCount].nSavedPosFlags & SavedPosMap::SPM_USED ) { ++nCount; if ( pCopySavedPos[nCount-1].nSavedPosFlags & SavedPosMap::SPM_LAST ) break; } SavedPos* pNewSavedPos = new SavedPos[nCount]; for ( int nCopy=0; nCopynSavedPosFlags & SavedPosMap::SPM_USED ) { ++nCount; if ( pSavedPos->nSavedPosFlags & SavedPosMap::SPM_LAST ) break; ++pSavedPos; } sprintf( szSlot, "%d ", nCount ); strBalance += szSlot; } */ return true; } return false; } bool CMarkupSTL::RestorePos( const char* szPosName ) { // Restore element position if found in saved position map if ( szPosName ) { int nSlot = m_mapSavedPos.Hash( szPosName ); SavedPos* pSavedPos = m_mapSavedPos.pTable[nSlot]; if ( pSavedPos ) { int nOffset = 0; while ( pSavedPos[nOffset].nSavedPosFlags & SavedPosMap::SPM_USED ) { if ( pSavedPos[nOffset].strName == szPosName ) { int i = pSavedPos[nOffset].iPos; if ( pSavedPos[nOffset].nSavedPosFlags & SavedPosMap::SPM_CHILD ) x_SetPos( m_aPos[m_aPos[i].iElemParent].iElemParent, m_aPos[i].iElemParent, i ); else if ( pSavedPos[nOffset].nSavedPosFlags & SavedPosMap::SPM_MAIN ) x_SetPos( m_aPos[i].iElemParent, i, 0 ); else x_SetPos( i, 0, 0 ); return true; } if ( pSavedPos[nOffset].nSavedPosFlags & SavedPosMap::SPM_LAST ) break; ++nOffset; } } } return false; } bool CMarkupSTL::RemoveElem() { // Remove current main position element if ( m_iPos && m_nNodeType == MNT_ELEMENT ) { int iPos = x_RemoveElem( m_iPos ); x_SetPos( m_iPosParent, iPos, 0 ); return true; } return false; } bool CMarkupSTL::RemoveChildElem() { // Remove current child position element if ( m_iPosChild ) { int iPosChild = x_RemoveElem( m_iPosChild ); x_SetPos( m_iPosParent, m_iPos, iPosChild ); return true; } return false; } ////////////////////////////////////////////////////////////////////// // Private Methods ////////////////////////////////////////////////////////////////////// bool CMarkupSTL::x_AllocPosArray( int nNewSize /*=0*/ ) { // Resize m_aPos when the document is created or the array is filled // The PosArray class is implemented using segments to reduce contiguous memory requirements // It reduces reallocations (copying of memory) since this only occurs within one segment // The "Grow By" algorithm ensures there are no reallocations after 2 segments // if ( ! nNewSize ) nNewSize = m_iPosFree + (m_iPosFree>>1); // Grow By: multiply size by 1.5 if ( m_aPos.GetSize() < nNewSize ) { // Grow By: new size can be at most one more complete segment int nSeg = (m_aPos.GetSize()?m_aPos.GetSize()-1:0) >> m_aPos.PA_SEGBITS; int nNewSeg = (nNewSize-1) >> m_aPos.PA_SEGBITS; if ( nNewSeg > nSeg + 1 ) { nNewSeg = nSeg + 1; nNewSize = (nNewSeg+1) << m_aPos.PA_SEGBITS; } // Allocate array of segments if ( m_aPos.nSegs <= nNewSeg ) { int nNewSegments = 4 + nNewSeg * 2; char* pNewSegments = new char[nNewSegments*sizeof(char*)]; if ( m_aPos.SegsUsed() ) memcpy( pNewSegments, m_aPos.pSegs, m_aPos.SegsUsed()*sizeof(char*) ); if ( m_aPos.pSegs ) delete[] (char*)m_aPos.pSegs; m_aPos.pSegs = (ElemPos**)pNewSegments; m_aPos.nSegs = nNewSegments; } // Calculate segment sizes int nSegSize = m_aPos.GetSize() - (nSeg << m_aPos.PA_SEGBITS); int nNewSegSize = nNewSize - (nNewSeg << m_aPos.PA_SEGBITS); // Complete first segment int nFullSegSize = 1 << m_aPos.PA_SEGBITS; if ( nSeg < nNewSeg && nSegSize < nFullSegSize ) { char* pNewFirstSeg = new char[ nFullSegSize * sizeof(ElemPos) ]; if ( nSegSize ) { // Reallocate memcpy( pNewFirstSeg, m_aPos.pSegs[nSeg], nSegSize * sizeof(ElemPos) ); delete[] (char*)m_aPos.pSegs[nSeg]; } m_aPos.pSegs[nSeg] = (ElemPos*)pNewFirstSeg; } // New segment char* pNewSeg = new char[ nNewSegSize * sizeof(ElemPos) ]; if ( nNewSeg == nSeg && nSegSize ) { // Reallocate memcpy( pNewSeg, m_aPos.pSegs[nSeg], nSegSize * sizeof(ElemPos) ); delete[] (char*)m_aPos.pSegs[nSeg]; } m_aPos.pSegs[nNewSeg] = (ElemPos*)pNewSeg; m_aPos.nSize = nNewSize; } return true; } bool CMarkupSTL::x_ParseDoc() { // Preserve pre-parse result string strResult = m_strError; // Reset indexes ResetPos(); m_mapSavedPos.RemoveAll(); // Starting size of position array: 1 element per 64 bytes of document // Tight fit when parsing small doc, only 0 to 2 reallocs when parsing large doc // Start at 8 when creating new document m_iPosFree = 1; x_AllocPosArray( (int)m_strDoc.size() / 64 + 8 ); m_iPosDeleted = 0; // Parse document bool bWellFormed = false; m_strError.erase(); if ( m_strDoc.size() ) { TokenPos token( m_strDoc.c_str() ); m_aPos[0].ClearVirtualParent(); int iPos = x_ParseElem( 0, token ); if ( iPos > 0 ) { m_aPos[0].iElemChild = iPos; m_aPos[0].nLength = (int)m_strDoc.size(); bWellFormed = true; } } else m_strError = "Empty document"; // Clear indexes if parse failed or empty document if ( ! bWellFormed ) { m_aPos[0].ClearVirtualParent(); m_iPosFree = 1; } ResetPos(); // Combine preserved result with parse error if ( ! strResult.empty() ) { if ( m_strError.empty() ) m_strError = strResult; else m_strError = strResult + ", " + m_strError; } return bWellFormed; }; int CMarkupSTL::x_ParseElem( int iPosParent, TokenPos& token ) { // This is either called by x_ParseDoc or x_AddSubDoc // This returns the new position if a tag is found, otherwise zero // In all cases we need to get a new ElemPos, but release it if unused // int iElemRoot = 0; int iPos = iPosParent; int nRootDepth = m_aPos[iPos].Level(); token.nNext = 0; // Loop through the nodes of the document NodeStack aNodes; aNodes.Add(); int nDepth = 0; int nTypeFound = 0; ElemPos* pElem; int iElemFirst, iElemLast; while ( nTypeFound >= 0 ) { nTypeFound = x_ParseNode( token, aNodes.Top() ); if ( nTypeFound == MNT_ELEMENT ) // start tag { iPos = x_GetFreePos(); if ( ! iElemRoot ) iElemRoot = iPos; else if ( nDepth == 0 ) { char* szError = new char[aNodes.Top().strName.size()+100]; sprintf( szError, "Element '%s' at offset %d is sibling to root", aNodes.Top().strName.c_str(), aNodes.Top().nStart ); m_strError = szError; delete [] szError; return -1; } pElem = &m_aPos[iPos]; pElem->iElemParent = iPosParent; pElem->iElemNext = 0; if ( m_aPos[iPosParent].iElemChild ) { iElemFirst = m_aPos[iPosParent].iElemChild; iElemLast = m_aPos[iElemFirst].iElemPrev; m_aPos[iElemLast].iElemNext = iPos; pElem->iElemPrev = iElemLast; m_aPos[iElemFirst].iElemPrev = iPos; pElem->nFlags = 0; } else { m_aPos[iPosParent].iElemChild = iPos; pElem->iElemPrev = iPos; pElem->nFlags = MNF_FIRST; } pElem->SetLevel( nRootDepth + nDepth ); pElem->iElemChild = 0; pElem->nStart = aNodes.Top().nStart; pElem->SetStartTagLen( aNodes.Top().nLength ); if ( aNodes.Top().nFlags & MNF_EMPTY ) { iPos = iPosParent; pElem->SetEndTagLen( 0 ); pElem->nLength = aNodes.Top().nLength; } else { iPosParent = iPos; ++nDepth; aNodes.Add(); } } else if ( nTypeFound == 0 ) // end tag { if ( aNodes.TopIndex() == 0 ) { char* szError = new char[token.Length()+100]; sprintf( szError, "No start tag for end tag '%s' at offset %d", x_GetToken(token).c_str(), aNodes.Top().nStart ); m_strError = szError; delete [] szError; return -1; } pElem = &m_aPos[iPos]; pElem->nLength = aNodes.Top().nStart - pElem->nStart + aNodes.Top().nLength; pElem->SetEndTagLen( aNodes.Top().nLength ); aNodes.Remove(); if ( ! token.Match(aNodes.Top().strName.c_str()) ) { char* szError = new char[aNodes.Top().strName.size()+token.Length()+100]; sprintf( szError, "End tag '%s' at offset %d does not match start tag '%s' at offset %d", x_GetToken(token).c_str(), token.nL-1, aNodes.Top().strName.c_str(), pElem->nStart ); m_strError = szError; delete [] szError; return -1; } --nDepth; iPosParent = pElem->iElemParent; iPos = iPosParent; } } if ( nTypeFound == -1 ) { m_strError = aNodes.Top().strName; return -1; } else if ( nDepth > 0 ) { aNodes.Remove(); char* szError = new char[aNodes.Top().strName.size()+100]; sprintf( szError, "Element '%s' at offset %d not ended", aNodes.Top().strName.c_str(), aNodes.Top().nStart ); m_strError = szError; delete [] szError; return -1; } else if ( ! iElemRoot ) { m_strError = "Root element not found"; return 0; } // Successfully parsed element (and contained elements) return iElemRoot; } bool CMarkupSTL::x_FindChar( const char* szDoc, int& nChar, char c ) { // static function const char* pChar = &szDoc[nChar]; while ( *pChar && *pChar != c ) ++pChar; nChar = (int)(pChar - szDoc); if ( ! *pChar ) return false; return true; } bool CMarkupSTL::x_FindAny( const char* szDoc, int& nChar ) { // Starting at nChar, find a non-whitespace char // return false if no non-whitespace before end of document, nChar points to end // otherwise return true and nChar points to non-whitespace char while ( szDoc[nChar] && strchr(" \t\n\r",szDoc[nChar]) ) ++nChar; return szDoc[nChar] != '\0'; } bool CMarkupSTL::x_FindToken( CMarkupSTL::TokenPos& token ) { // Starting at token.nNext, bypass whitespace and find the next token // returns true on success, members of token point to token // returns false on end of document, members point to end of document const char* szDoc = token.szDoc; int nChar = token.nNext; token.bIsString = false; // By-pass leading whitespace if ( ! x_FindAny(szDoc,nChar) ) { // No token was found before end of document token.nL = nChar; token.nR = nChar - 1; token.nNext = nChar; return false; } // Is it an opening quote? char cFirstChar = szDoc[nChar]; if ( cFirstChar == '\"' || cFirstChar == '\'' ) { token.bIsString = true; // Move past opening quote ++nChar; token.nL = nChar; // Look for closing quote x_FindChar( token.szDoc, nChar, cFirstChar ); // Set right to before closing quote token.nR = nChar - 1; // Set nChar past closing quote unless at end of document if ( szDoc[nChar] ) ++nChar; } else { // Go until special char or whitespace token.nL = nChar; while ( szDoc[nChar] && ! strchr(" \t\n\r<>=\\/?!",szDoc[nChar]) ) ++nChar; // Adjust end position if it is one special char if ( nChar == token.nL ) ++nChar; // it is a special char token.nR = nChar - 1; } // nNext points to one past last char of token token.nNext = nChar; return true; } string CMarkupSTL::x_GetToken( const CMarkupSTL::TokenPos& token ) { // The token contains indexes into the document identifying a small substring // Build the substring from those indexes and return it if ( token.nL > token.nR ) return ""; string strToken( &token.szDoc[token.nL], token.Length() ); return strToken; } int CMarkupSTL::x_FindElem( int iPosParent, int iPos, const char* szPath ) { // If szPath is NULL or empty, go to next sibling element // Otherwise go to next sibling element with matching path // if ( iPos ) iPos = m_aPos[iPos].iElemNext; else iPos = m_aPos[iPosParent].iElemChild; // Finished here if szPath not specified if ( szPath == NULL || !szPath[0] ) return iPos; // Search TokenPos token( m_strDoc.c_str() ); while ( iPos ) { // Compare tag name token.nNext = m_aPos[iPos].nStart + 1; x_FindToken( token ); // Locate tag name if ( token.Match(szPath) ) return iPos; iPos = m_aPos[iPos].iElemNext; } return 0; } int CMarkupSTL::x_ParseNode( CMarkupSTL::TokenPos& token, CMarkupSTL::NodePos& node ) { // Call this with token.nNext set to the start of the node or tag // Upon return token.nNext points to the char after the node or tag // // comment // dtd // processing instruction // cdata section // element start tag // element end tag // // returns the nodetype, or 0 for end tag, -1 for end of document, -2 for error // enum ParseBits { PD_OPENTAG = 1, PD_BANG = 2, PD_DASH = 4, PD_BRACKET = 8, PD_TEXTORWS = 16, PD_DOCTYPE = 32, PD_INQUOTE_S = 64, PD_INQUOTE_D = 128, }; int nParseFlags = 0; const char* szFindEnd = NULL; int nNodeType = -1; int nEndLen = 0; int nName = 0; #define FINDNODETYPE(e,t,n) { szFindEnd=e; nEndLen=(sizeof(e)-1); nNodeType=t; if(n) nName=(int)(pDoc-token.szDoc)+n-1; } #define ISNODEERROR(e) { char szE[100]; nNodeType=-1; sprintf(szE,"Incorrect %s at offset %d",e,nR); node.strName=szE; break; } node.nStart = token.nNext; node.nFlags = 0; int nR = token.nNext; const char* pDoc = &token.szDoc[nR]; if ( ! *pDoc ) { node.nLength = 0; node.nNodeType = 0; return -2; // end of document } while ( 1 ) { if ( ! *pDoc ) { nR = (int)(pDoc - token.szDoc) - 1; if ( nNodeType != MNT_WHITESPACE && nNodeType != MNT_TEXT ) { const char* szType = "tag"; if ( (nParseFlags & PD_DOCTYPE) || nNodeType == MNT_DOCUMENT_TYPE ) szType = "Doctype"; else if ( nNodeType == MNT_ELEMENT ) szType = "Element tag"; else if ( nNodeType == 0 ) szType = "Element end tag"; else if ( nNodeType == MNT_CDATA_SECTION ) szType = "CDATA Section"; else if ( nNodeType == MNT_PROCESSING_INSTRUCTION ) szType = "Processing instruction"; else if ( nNodeType == MNT_COMMENT ) szType = "Comment"; nNodeType = -1; char szError[100]; sprintf( szError, "%s at offset %d unterminated", szType, node.nStart ); node.strName = szError; } break; } if ( nName ) { if ( strchr(" \t\n\r/>",*pDoc) ) { int nNameLen = (int)(pDoc - token.szDoc) - nName; if ( nNodeType == 0 ) { token.nL = nName; token.nR = nName + nNameLen - 1; } else { node.strName.assign( &token.szDoc[nName], nNameLen ); } nName = 0; } else { ++pDoc; continue; } } if ( szFindEnd ) { if ( *pDoc == '>' && ! (nParseFlags & (PD_INQUOTE_S|PD_INQUOTE_D)) ) { nR = (int)(pDoc - token.szDoc); if ( nEndLen == 1 ) { szFindEnd = NULL; if ( nNodeType == MNT_ELEMENT && *(pDoc-1) == '/' ) node.nFlags |= MNF_EMPTY; } else if ( nR > nEndLen ) { // Test for end of PI or comment const char* pEnd = pDoc - nEndLen + 1; const char* pFindEnd = szFindEnd; int nLen = nEndLen; while ( --nLen && *pEnd++ == *pFindEnd++ ); if ( nLen == 0 ) szFindEnd = NULL; } if ( ! szFindEnd && ! (nParseFlags & PD_DOCTYPE) ) break; } else if ( *pDoc == '<' && nNodeType == MNT_TEXT ) { nR = (int)(pDoc - token.szDoc) - 1; break; } else if ( nNodeType == MNT_ELEMENT ) { if ( *pDoc == '\"' && ! (nParseFlags&PD_INQUOTE_S) ) nParseFlags ^= PD_INQUOTE_D; else if ( *pDoc == '\'' && ! (nParseFlags&PD_INQUOTE_D) ) nParseFlags ^= PD_INQUOTE_S; } } else if ( nParseFlags ) { if ( nParseFlags & PD_TEXTORWS ) { if ( *pDoc == '<' ) { nR = (int)(pDoc - token.szDoc) - 1; nNodeType = MNT_WHITESPACE; break; } else if ( ! strchr(" \t\n\r",*pDoc) ) { nParseFlags ^= PD_TEXTORWS; FINDNODETYPE( "<", MNT_TEXT, 0 ) } } else if ( nParseFlags & PD_OPENTAG ) { nParseFlags ^= PD_OPENTAG; if ( *pDoc > 0x60 || ( *pDoc > 0x40 && *pDoc < 0x5b ) || *pDoc == 0x5f || *pDoc == 0x3a ) FINDNODETYPE( ">", MNT_ELEMENT, 1 ) else if ( *pDoc == '/' ) FINDNODETYPE( ">", 0, 2 ) else if ( *pDoc == '!' ) nParseFlags |= PD_BANG; else if ( *pDoc == '?' ) FINDNODETYPE( "?>", MNT_PROCESSING_INSTRUCTION, 2 ) else ISNODEERROR( "tag name character" ) } else if ( nParseFlags & PD_BANG ) { nParseFlags ^= PD_BANG; if ( *pDoc == '-' ) nParseFlags |= PD_DASH; else if ( *pDoc == '[' && !(nParseFlags & PD_DOCTYPE) ) nParseFlags |= PD_BRACKET; else if ( *pDoc == 'D' && !(nParseFlags & PD_DOCTYPE) ) nParseFlags |= PD_DOCTYPE; else if ( strchr("EAN",*pDoc) ) // ", -1, 0 ) else ISNODEERROR( "! tag" ) } else if ( nParseFlags & PD_DASH ) { nParseFlags ^= PD_DASH; if ( *pDoc == '-' ) FINDNODETYPE( "-->", MNT_COMMENT, 0 ) else ISNODEERROR( "comment tag" ) } else if ( nParseFlags & PD_BRACKET ) { nParseFlags ^= PD_BRACKET; if ( *pDoc == 'C' ) FINDNODETYPE( "]]>", MNT_CDATA_SECTION, 0 ) else ISNODEERROR( "tag" ) } else if ( nParseFlags & PD_DOCTYPE ) { if ( *pDoc == '<' ) nParseFlags |= PD_OPENTAG; else if ( *pDoc == '>' ) { nR = (int)(pDoc - token.szDoc); nNodeType = MNT_DOCUMENT_TYPE; break; } } } else if ( *pDoc == '<' ) { nParseFlags |= PD_OPENTAG; } else { nNodeType = MNT_WHITESPACE; if ( strchr(" \t\n\r",*pDoc) ) nParseFlags |= PD_TEXTORWS; else FINDNODETYPE( "<", MNT_TEXT, 0 ) } ++pDoc; } token.nNext = nR + 1; node.nLength = token.nNext - node.nStart; node.nNodeType = nNodeType; return nNodeType; } string CMarkupSTL::x_GetTagName( int iPos ) const { // Return the tag name at specified element TokenPos token( m_strDoc.c_str() ); token.nNext = m_aPos[iPos].nStart + 1; if ( ! iPos || ! x_FindToken( token ) ) return ""; // Return substring of document return x_GetToken( token ); } bool CMarkupSTL::x_FindAttrib( CMarkupSTL::TokenPos& token, const char* szAttrib ) { // If szAttrib is NULL find next attrib, otherwise find named attrib // Return true if found int nAttrib = 0; for ( int nCount = 0; x_FindToken(token); ++nCount ) { if ( ! token.bIsString ) { // Is it the right angle bracket? char cChar = token.szDoc[token.nL]; if ( cChar == '>' || cChar == '/' || cChar == '?' ) break; // attrib not found // Equal sign if ( cChar == '=' ) continue; // Potential attribute if ( ! nAttrib && nCount ) { // Attribute name search? if ( ! szAttrib || ! szAttrib[0] ) return true; // return with token at attrib name // Compare szAttrib if ( token.Match(szAttrib) ) nAttrib = nCount; } } else if ( nAttrib && nCount == nAttrib + 2 ) { return true; } } // Not found return false; } string CMarkupSTL::x_GetAttrib( int iPos, const char* szAttrib ) const { // Return the value of the attrib TokenPos token( m_strDoc.c_str() ); if ( iPos && m_nNodeType == MNT_ELEMENT ) token.nNext = m_aPos[iPos].nStart + 1; else if ( iPos == m_iPos && m_nNodeLength && m_nNodeType == MNT_PROCESSING_INSTRUCTION ) token.nNext = m_nNodeOffset + 2; else return ""; if ( szAttrib && x_FindAttrib( token, szAttrib ) ) return x_TextFromDoc( token.nL, token.Length() ); return ""; } bool CMarkupSTL::x_SetAttrib( int iPos, const char* szAttrib, int nValue ) { // Convert integer to string and call SetChildAttrib char szVal[25]; sprintf( szVal, "%d", nValue ); return x_SetAttrib( iPos, szAttrib, szVal ); } bool CMarkupSTL::x_SetAttrib( int iPos, const char* szAttrib, const char* szValue ) { // Set attribute in iPos element TokenPos token( m_strDoc.c_str() ); int nInsertAt; if ( iPos && m_nNodeType == MNT_ELEMENT ) { token.nNext = m_aPos[iPos].nStart + 1; nInsertAt = m_aPos[iPos].StartContent() - (m_aPos[iPos].IsEmptyElement()?2:1); } else if ( iPos == m_iPos && m_nNodeLength && m_nNodeType == MNT_PROCESSING_INSTRUCTION ) { token.nNext = m_nNodeOffset + 2; nInsertAt = m_nNodeOffset + m_nNodeLength - 2; } else return false; // Create insertion text depending on whether attribute already exists int nReplace = 0; string strInsert; if ( x_FindAttrib( token, szAttrib ) ) { // Replace value only // Decision: for empty value leaving attrib="" instead of removing attrib strInsert = x_TextToDoc( szValue, true ); nInsertAt = token.nL; nReplace = token.Length(); } else { // Insert string name value pair string strFormat; strFormat = " "; strFormat += szAttrib; strFormat += "=" x_ATTRIBQUOTE; strFormat += x_TextToDoc( szValue, true ); strFormat += x_ATTRIBQUOTE; strInsert = strFormat; } x_DocChange( nInsertAt, nReplace, strInsert ); int nAdjust = (int)strInsert.size() - nReplace; m_aPos[iPos].AdjustStartTagLen( nAdjust ); m_aPos[iPos].nLength += nAdjust; x_Adjust( iPos, nAdjust ); MARKUP_SETDEBUGSTATE; return true; } bool CMarkupSTL::x_CreateNode( string& strNode, int nNodeType, const char* szText ) { // Set strNode based on nNodeType and szData // Return false if szData would jeopardize well-formed document // switch ( nNodeType ) { case MNT_CDATA_SECTION: if ( strstr(szText,"]]>") != NULL ) return false; strNode = ""; break; } return true; } bool CMarkupSTL::x_SetData( int iPos, const char* szData, int nCDATA ) { // Set data at specified position // if nCDATA==1, set content of element to a CDATA Section string strInsert; // Set data in iPos element if ( ! iPos || m_aPos[iPos].iElemChild ) return false; // Build strInsert from szData based on nCDATA if ( nCDATA != 0 ) { // Split CDATA Sections if there are any end delimiters const char* pszNextStart = szData; strInsert += "" ); while ( pszEnd ) { strInsert.append( pszNextStart, (int)(pszEnd - pszNextStart) ); strInsert += "]]]]>"; pszNextStart = pszEnd + 3; pszEnd = strstr( pszNextStart, "]]>" ); } strInsert += pszNextStart; strInsert += "]]>"; } else strInsert = x_TextToDoc( szData ); // Decide where to insert int nInsertAt, nReplace; if ( m_aPos[iPos].IsEmptyElement() ) { string strTagName = x_GetTagName( iPos ); string strFormat; strFormat = ">"; strFormat += strInsert; strFormat += "7" to "6>7" // // < less than // & ampersand // > greater than // // and for attributes: // // ' apostrophe or single quote // " double quote // static const char* szaReplace[] = { "<","&",">","'",""" }; const char* pFind = bAttrib?"<&>\'\"":"<&>"; string strText; const char* pSource = szText; int nDestSize = (int)strlen(pSource); nDestSize += nDestSize / 10 + 7; strText.reserve( nDestSize ); char cSource = *pSource; const char* pFound; while ( cSource ) { if ( (pFound=strchr(pFind,cSource)) != NULL ) { pFound = szaReplace[pFound-pFind]; strText.append( pFound ); } else { strText += cSource; } ++pSource; cSource = *pSource; } return strText; } string CMarkupSTL::x_TextFromDoc( int nLeft, int nCopy ) const { // Convert XML friendly text to text as seen outside XML document // ampersand escape codes replaced with special characters e.g. convert "6>7" to "6>7" // ampersand numeric codes replaced with character e.g. convert < to < // Conveniently the result is always the same or shorter in byte length // static const char* szaCode[] = { "lt;","amp;","gt;","apos;","quot;" }; static int anCodeLen[] = { 3,4,3,5,5 }; static const char* szSymbol = "<&>\'\""; string strText; const char* pSource = m_strDoc.c_str(); int nEnd = nLeft + nCopy; strText.reserve( nCopy ); int nChar = nLeft; while ( nChar < nEnd ) { if ( pSource[nChar] == '&' ) { bool bCodeConverted = false; // Is it a numeric character reference? if ( pSource[nChar+1] == '#' ) { // Is it a hex number? int nBase = 10; int nNumericChar = nChar + 2; char cChar = pSource[nNumericChar]; if ( cChar == 'x' ) { ++nNumericChar; cChar = pSource[nNumericChar]; nBase = 16; } // Look for terminating semi-colon within 7 characters int nCodeLen = 0; while ( nCodeLen < 7 && cChar && cChar != ';' ) { // only ASCII digits 0-9, A-F, a-f expected ++nCodeLen; cChar = pSource[nNumericChar + nCodeLen]; } // Process unicode if ( cChar == ';' ) { int nUnicode = strtol( &pSource[nNumericChar], NULL, nBase ); /* MBCS int nMBLen = wctomb( &pDest[nLen], (wchar_t)nUnicode ); if ( nMBLen > 0 ) nLen += nMBLen; else nUnicode = 0; */ if ( nUnicode < 0x80 ) strText += (char)nUnicode; else if ( nUnicode < 0x800 ) { // Convert to 2-byte UTF-8 strText += (char)(((nUnicode&0x7c0)>>6) | 0xc0); strText += (char)((nUnicode&0x3f) | 0x80); } else { // Convert to 3-byte UTF-8 strText += (char)(((nUnicode&0xf000)>>12) | 0xe0); strText += (char)(((nUnicode&0xfc0)>>6) | 0x80); strText += (char)((nUnicode&0x3f) | 0x80); } if ( nUnicode ) { // Increment index past ampersand semi-colon nChar = nNumericChar + nCodeLen + 1; bCodeConverted = true; } } } else // does not start with # { // Look for matching &code; for ( int nMatch = 0; nMatch < 5; ++nMatch ) { if ( nChar < nEnd - anCodeLen[nMatch] && strncmp(szaCode[nMatch],&pSource[nChar+1],anCodeLen[nMatch]) == 0 ) { // Insert symbol and increment index past ampersand semi-colon strText += szSymbol[nMatch]; nChar += anCodeLen[nMatch] + 1; bCodeConverted = true; break; } } } // If the code is not converted, leave it as is if ( ! bCodeConverted ) { strText += '&'; ++nChar; } } else // not & { strText += pSource[nChar]; ++nChar; } } return strText; } void CMarkupSTL::x_DocChange( int nLeft, int nReplace, const string& strInsert ) { // Insert strInsert int m_strDoc at nLeft replacing nReplace chars // Do this with only one buffer reallocation if it grows // int nDocLength = (int)m_strDoc.size(); int nInsLength = (int)strInsert.size(); int nNewLength = nInsLength + nDocLength - nReplace; // When creating a document, reduce reallocs by reserving string space // Allow for 1.5 times the current allocation int nBufferLen = nNewLength; int nAllocLen = (int)m_strDoc.capacity(); if ( nNewLength > nAllocLen ) { nBufferLen += nBufferLen/2 + 128; if ( nBufferLen < nNewLength ) nBufferLen = nNewLength; m_strDoc.reserve( nBufferLen ); } m_strDoc.replace( nLeft, nReplace, strInsert ); } void CMarkupSTL::x_Adjust( int iPos, int nShift, bool bAfterPos ) { // Loop through affected elements and adjust indexes // Algorithm: // 1. update children unless bAfterPos // (if no children or bAfterPos is true, length of iPos not affected) // 2. update starts of next siblings and their children // 3. go up until there is a next sibling of a parent and update starts // 4. step 2 int iPosTop = m_aPos[iPos].iElemParent; bool bPosFirst = bAfterPos; // mark as first to skip its children while ( iPos ) { // Were we at containing parent of affected position? bool bPosTop = false; if ( iPos == iPosTop ) { // Move iPosTop up one towards root iPosTop = m_aPos[iPos].iElemParent; bPosTop = true; } // Traverse to the next update position if ( ! bPosTop && ! bPosFirst && m_aPos[iPos].iElemChild ) { // Depth first iPos = m_aPos[iPos].iElemChild; } else if ( m_aPos[iPos].iElemNext ) { iPos = m_aPos[iPos].iElemNext; } else { // Look for next sibling of a parent of iPos // When going back up, parents have already been done except iPosTop while ( (iPos=m_aPos[iPos].iElemParent) != 0 && iPos != iPosTop ) if ( m_aPos[iPos].iElemNext ) { iPos = m_aPos[iPos].iElemNext; break; } } bPosFirst = false; // Shift indexes at iPos if ( iPos != iPosTop ) m_aPos[iPos].nStart += nShift; else m_aPos[iPos].nLength += nShift; } } void CMarkupSTL::x_LocateNew( int iPosParent, int& iPosRel, int& nOffset, int nLength, int nFlags ) { // Determine where to insert new element or node // bool bInsert = (nFlags&1)?true:false; bool bHonorWhitespace = (nFlags&2)?true:false; int nStartAt; if ( nLength ) { // Located at a non-element node if ( bInsert ) nStartAt = nOffset; else nStartAt = nOffset + nLength; } else if ( iPosRel ) { // Located at an element nStartAt = m_aPos[iPosRel].nStart; if ( ! bInsert ) // follow iPosRel nStartAt += m_aPos[iPosRel].nLength; } else if ( ! iPosParent ) { // Outside of all elements if ( bInsert ) nStartAt = 0; else nStartAt = (int)m_strDoc.size(); } else if ( m_aPos[iPosParent].IsEmptyElement() ) { // Parent has no separate end tag, so split empty element nStartAt = m_aPos[iPosParent].StartContent() - 1; } else { if ( bInsert ) // after start tag nStartAt = m_aPos[iPosParent].StartContent(); else // before end tag nStartAt = m_aPos[iPosParent].StartAfter() - m_aPos[iPosParent].EndTagLen(); } // Go up to start of next node, unless its splitting an empty element if ( ! bHonorWhitespace && ! m_aPos[iPosParent].IsEmptyElement() ) { const char* szDoc = m_strDoc.c_str(); int nChar = nStartAt; if ( ! x_FindAny(szDoc,nChar) || szDoc[nChar] == '<' ) nStartAt = nChar; } // Determine iPosBefore int iPosBefore = 0; if ( iPosRel ) { if ( bInsert ) { if ( ! (m_aPos[iPosRel].nFlags & MNF_FIRST) ) iPosBefore = m_aPos[iPosRel].iElemPrev; } else iPosBefore = iPosRel; } else if ( m_aPos[iPosParent].iElemChild ) { if ( bInsert ) iPosBefore = iPosRel; else iPosBefore = m_aPos[m_aPos[iPosParent].iElemChild].iElemPrev; } nOffset = nStartAt; iPosRel = iPosBefore; } bool CMarkupSTL::x_AddElem( const char* szName, int nValue, bool bInsert, bool bAddChild ) { // Convert integer to string char szVal[25]; sprintf( szVal, "%d", nValue ); return x_AddElem( szName, szVal, bInsert, bAddChild ); } bool CMarkupSTL::x_AddElem( const char* szName, const char* szValue, bool bInsert, bool bAddChild ) { if ( bAddChild ) { // Adding a child element under main position if ( ! m_iPos ) return false; } else if ( m_iPosParent == 0 ) { // Adding root element if ( IsWellFormed() ) return false; // Locate after any version and DTD m_aPos[0].nLength = (int)m_strDoc.size(); } // Locate where to add element relative to current node int iPosParent, iPosBefore, nOffset = 0, nLength = 0; if ( bAddChild ) { iPosParent = m_iPos; iPosBefore = m_iPosChild; } else { iPosParent = m_iPosParent; iPosBefore = m_iPos; nOffset = m_nNodeOffset; nLength = m_nNodeLength; } int nFlags = bInsert?1:0; x_LocateNew( iPosParent, iPosBefore, nOffset, nLength, nFlags ); bool bEmptyParent = m_aPos[iPosParent].IsEmptyElement(); bool bNoContentParent = (iPosParent && m_aPos[iPosParent].ContentLen() == 0)?true:false; if ( bEmptyParent || bNoContentParent ) nOffset += x_EOLLEN; // Create element and modify positions of affected elements // If no szValue is specified, an empty element is created // i.e. either value or // int iPos = x_GetFreePos(); ElemPos* pElem = &m_aPos[iPos]; pElem->nStart = nOffset; pElem->iElemChild = 0; x_LinkElem( iPosParent, iPosBefore, iPos ); // Create string for insert string strInsert; int nLenName = (int)strlen(szName); int nLenValue = szValue? (int)strlen(szValue) : 0; if ( ! nLenValue ) { // empty element strInsert = "<"; strInsert += szName; strInsert += "/>" x_EOL; pElem->SetStartTagLen( nLenName + 3 ); pElem->SetEndTagLen( 0 ); pElem->nLength = nLenName + 3; } else { // value string strValue = x_TextToDoc( szValue ); nLenValue = (int)strValue.size(); strInsert = "<"; strInsert += szName; strInsert += ">"; strInsert += strValue; strInsert += "" x_EOL; pElem->SetStartTagLen( nLenName + 2 ); pElem->SetEndTagLen( nLenName + 3 ); pElem->nLength = nLenName * 2 + nLenValue + 5; } // Insert int nReplace = 0, nInsertAt = pElem->nStart; if ( bEmptyParent ) { string strParentTagName = x_GetTagName(iPosParent); string strFormat; strFormat = ">" x_EOL; strFormat += strInsert; strFormat += "" x_EOL; strFormat += strInsert; strFormat += "StartContent() - 2; nReplace = 1; pParent->AdjustStartTagLen( -1 ); pParent->SetEndTagLen( 3 + (int)strParentTagName.size() ); } x_DocChange( nInsertAt, nReplace, strInsert ); x_Adjust( iPos, (int)strInsert.size() - nReplace, true ); // Set position to top element of subdocument if ( bAddChild ) x_SetPos( m_iPosParent, iPosParent, iPos ); else // Main x_SetPos( m_iPosParent, iPos, 0 ); return true; } int CMarkupSTL::x_RemoveElem( int iPos ) { // Remove element and all contained elements // Return new position // if ( ! iPos ) return 0; // Determine whether any whitespace up to next tag int nAfterEnd = m_aPos[iPos].StartAfter(); const char* szDoc = m_strDoc.c_str(); int nChar = nAfterEnd; if ( ! x_FindAny(szDoc,nChar) || szDoc[nChar] == '<' ) nAfterEnd = nChar; // Remove from document, adjust affected indexes, and unlink int nLen = nAfterEnd - m_aPos[iPos].nStart; x_DocChange( m_aPos[iPos].nStart, nLen, string() ); x_Adjust( iPos, - nLen, true ); return x_UnlinkElem( iPos ); } void CMarkupSTL::x_LinkElem( int iPosParent, int iPosBefore, int iPos ) { // Link in element, and initialize nFlags, and iElem indexes ElemPos* pElem = &m_aPos[iPos]; pElem->iElemParent = iPosParent; if ( iPosBefore ) { // Link in after iPosBefore pElem->nFlags = 0; pElem->iElemNext = m_aPos[iPosBefore].iElemNext; if ( pElem->iElemNext ) m_aPos[pElem->iElemNext].iElemPrev = iPos; else m_aPos[m_aPos[iPosParent].iElemChild].iElemPrev = iPos; m_aPos[iPosBefore].iElemNext = iPos; pElem->iElemPrev = iPosBefore; } else { // Link in as first child pElem->nFlags = MNF_FIRST; if ( m_aPos[iPosParent].iElemChild ) { pElem->iElemNext = m_aPos[iPosParent].iElemChild; pElem->iElemPrev = m_aPos[pElem->iElemNext].iElemPrev; m_aPos[pElem->iElemNext].iElemPrev = iPos; m_aPos[pElem->iElemNext].nFlags ^= MNF_FIRST; } else { pElem->iElemNext = 0; pElem->iElemPrev = iPos; } m_aPos[iPosParent].iElemChild = iPos; } if ( iPosParent ) pElem->SetLevel( m_aPos[iPosParent].Level() + 1 ); } int CMarkupSTL::x_UnlinkElem( int iPos ) { // Fix links to remove element and mark as deleted // return previous position or zero if none ElemPos* pElem = &m_aPos[iPos]; // Find previous sibling and bypass removed element int iPosPrev = 0; if ( pElem->nFlags & MNF_FIRST ) { if ( pElem->iElemNext ) // set next as first child { m_aPos[pElem->iElemParent].iElemChild = pElem->iElemNext; m_aPos[pElem->iElemNext].iElemPrev = pElem->iElemPrev; m_aPos[pElem->iElemNext].nFlags |= MNF_FIRST; } else // no children remaining m_aPos[pElem->iElemParent].iElemChild = 0; } else { iPosPrev = pElem->iElemPrev; m_aPos[iPosPrev].iElemNext = pElem->iElemNext; if ( pElem->iElemNext ) m_aPos[pElem->iElemNext].iElemPrev = iPosPrev; else m_aPos[m_aPos[pElem->iElemParent].iElemChild].iElemPrev = iPosPrev; } pElem->nFlags = MNF_DELETED; pElem->iElemNext = m_iPosDeleted; m_iPosDeleted = iPos; return iPosPrev; }