src/common/tokenzr.cpp

/////////////////////////////////////////////////////////////////////////////
// Name:        tokenzr.cpp
// Purpose:     String tokenizer
// Author:      Guilhem Lavaux
// Modified by: Vadim Zeitlin
// Created:     04/22/98
// RCS-ID:      $Id$
// Copyright:   (c) Guilhem Lavaux
// Licence:     wxWindows licence
/////////////////////////////////////////////////////////////////////////////

// ============================================================================
// declarations
// ============================================================================

// ----------------------------------------------------------------------------
// headers
// ----------------------------------------------------------------------------

#ifdef __GNUG__
    #pragma implementation "tokenzr.h"
#endif

// For compilers that support precompilation, includes "wx.h".
#include "wx/wxprec.h"

#ifdef __BORLANDC__
    #pragma hdrstop
#endif

#include "wx/tokenzr.h"

// Required for wxIs... functions
#include <ctype.h>

// ============================================================================
// implementation
// ============================================================================

// ----------------------------------------------------------------------------
// wxStringTokenizer construction
// ----------------------------------------------------------------------------

wxStringTokenizer::wxStringTokenizer(const wxString& str,
                                     const wxString& delims,
                                     wxStringTokenizerMode mode)
{
    SetString(str, delims, mode);
}

void wxStringTokenizer::SetString(const wxString& str,
                                  const wxString& delims,
                                  wxStringTokenizerMode mode)
{
    if ( mode == wxTOKEN_DEFAULT )
    {
        // by default, we behave like strtok() if the delimiters are only
        // whitespace characters and as wxTOKEN_RET_EMPTY otherwise (for
        // whitespace delimiters, strtok() behaviour is better because we want
        // to count consecutive spaces as one delimiter)
        const wxChar *p;
        for ( p = delims.c_str(); *p; p++ )
        {
            if ( !wxIsspace(*p) )
                break;
        }

        if ( *p )
        {
            // not whitespace char in delims
            mode = wxTOKEN_RET_EMPTY;
        }
        else
        {
            // only whitespaces
            mode = wxTOKEN_STRTOK;
        }
    }

    m_delims = delims;
    m_mode = mode;

    Reinit(str);
}

void wxStringTokenizer::Reinit(const wxString& str)
{
    wxASSERT_MSG( IsOk(), _T("you should call SetString() first") );

    m_string = str;
    m_pos = 0;

    // empty string doesn't have any tokens
    m_hasMore = !m_string.empty();
}

// ----------------------------------------------------------------------------
// access to the tokens
// ----------------------------------------------------------------------------

// do we have more of them?
bool wxStringTokenizer::HasMoreTokens() const
{
    wxCHECK_MSG( IsOk(), FALSE, _T("you should call SetString() first") );

    if ( m_string.find_first_not_of(m_delims) == wxString::npos )
    {
        // no non empty tokens left, but in wxTOKEN_RET_EMPTY_ALL mode we
        // still may return TRUE if GetNextToken() wasn't called yet for the
        // last trailing empty token
        return m_mode == wxTOKEN_RET_EMPTY_ALL ? m_hasMore : FALSE;
    }
    else
    {
        // there are non delimiter characters left, hence we do have more
        // tokens
        return TRUE;
    }
}

// count the number of tokens in the string
size_t wxStringTokenizer::CountTokens() const
{
    wxCHECK_MSG( IsOk(), 0, _T("you should call SetString() first") );

    // VZ: this function is IMHO not very useful, so it's probably not very
    //     important if it's implementation here is not as efficient as it
    //     could be - but OTOH like this we're sure to get the correct answer
    //     in all modes
    wxStringTokenizer *self = (wxStringTokenizer *)this;    // const_cast
    wxString stringInitial = m_string;

    size_t count = 0;
    while ( self->HasMoreTokens() )
    {
        count++;

        (void)self->GetNextToken();
    }

    self->Reinit(stringInitial);

    return count;
}

// ----------------------------------------------------------------------------
// token extraction
// ----------------------------------------------------------------------------

wxString wxStringTokenizer::GetNextToken()
{
    // strtok() doesn't return empty tokens, all other modes do
    bool allowEmpty = m_mode != wxTOKEN_STRTOK;

    wxString token;
    do
    {
        if ( !HasMoreTokens() )
        {
            break;
        }
        // find the end of this token
        size_t pos = m_string.find_first_of(m_delims);

        // and the start of the next one
        if ( pos == wxString::npos )
        {
            // no more delimiters, the token is everything till the end of
            // string
            token = m_string;

            m_pos += m_string.length();
            m_string.clear();

            // no more tokens in this string, even in wxTOKEN_RET_EMPTY_ALL
            // mode (we will return the trailing one right now in this case)
            m_hasMore = FALSE;
        }
        else
        {
            size_t pos2 = pos + 1;

            // in wxTOKEN_RET_DELIMS mode we return the delimiter character
            // with token
            token = wxString(m_string, m_mode == wxTOKEN_RET_DELIMS ? pos2
                                                                    : pos);

            // remove token with the following it delimiter from string
            m_string.erase(0, pos2);

            // keep track of the position in the original string too
            m_pos += pos2;
        }
    }
    while ( !allowEmpty && token.empty() );

    return token;
}
Commit	Line	Data
	1	/////////////////////////////////////////////////////////////////////////////
	2	// Name: tokenzr.cpp
	3	// Purpose: String tokenizer
	4	// Author: Guilhem Lavaux
	5	// Modified by: Vadim Zeitlin
	6	// Created: 04/22/98
	7	// RCS-ID: $Id$
	8	// Copyright: (c) Guilhem Lavaux
	9	// Licence: wxWindows licence
	10	/////////////////////////////////////////////////////////////////////////////
	11
	12	// ============================================================================
	13	// declarations
	14	// ============================================================================
	15
	16	// ----------------------------------------------------------------------------
	17	// headers
	18	// ----------------------------------------------------------------------------
	19
	20	#ifdef __GNUG__
	21	#pragma implementation "tokenzr.h"
	22	#endif
	23
	24	// For compilers that support precompilation, includes "wx.h".
	25	#include "wx/wxprec.h"
	26
	27	#ifdef __BORLANDC__
	28	#pragma hdrstop
	29	#endif
	30
	31	#include "wx/tokenzr.h"
	32
	33	// Required for wxIs... functions
	34	#include <ctype.h>
	35
	36	// ============================================================================
	37	// implementation
	38	// ============================================================================
	39
	40	// ----------------------------------------------------------------------------
	41	// wxStringTokenizer construction
	42	// ----------------------------------------------------------------------------
	43
	44	wxStringTokenizer::wxStringTokenizer(const wxString& str,
	45	const wxString& delims,
	46	wxStringTokenizerMode mode)
	47	{
	48	SetString(str, delims, mode);
	49	}
	50
	51	void wxStringTokenizer::SetString(const wxString& str,
	52	const wxString& delims,
	53	wxStringTokenizerMode mode)
	54	{
	55	if ( mode == wxTOKEN_DEFAULT )
	56	{
	57	// by default, we behave like strtok() if the delimiters are only
	58	// whitespace characters and as wxTOKEN_RET_EMPTY otherwise (for
	59	// whitespace delimiters, strtok() behaviour is better because we want
	60	// to count consecutive spaces as one delimiter)
	61	const wxChar *p;
	62	for ( p = delims.c_str(); *p; p++ )
	63	{
	64	if ( !wxIsspace(*p) )
	65	break;
	66	}
	67
	68	if ( *p )
	69	{
	70	// not whitespace char in delims
	71	mode = wxTOKEN_RET_EMPTY;
	72	}
	73	else
	74	{
	75	// only whitespaces
	76	mode = wxTOKEN_STRTOK;
	77	}
	78	}
	79
	80	m_delims = delims;
	81	m_mode = mode;
	82
	83	Reinit(str);
	84	}
	85
	86	void wxStringTokenizer::Reinit(const wxString& str)
	87	{
	88	wxASSERT_MSG( IsOk(), _T("you should call SetString() first") );
	89
	90	m_string = str;
	91	m_pos = 0;
	92
	93	// empty string doesn't have any tokens
	94	m_hasMore = !m_string.empty();
	95	}
	96
	97	// ----------------------------------------------------------------------------
	98	// access to the tokens
	99	// ----------------------------------------------------------------------------
	100
	101	// do we have more of them?
	102	bool wxStringTokenizer::HasMoreTokens() const
	103	{
	104	wxCHECK_MSG( IsOk(), FALSE, _T("you should call SetString() first") );
	105
	106	if ( m_string.find_first_not_of(m_delims) == wxString::npos )
	107	{
	108	// no non empty tokens left, but in wxTOKEN_RET_EMPTY_ALL mode we
	109	// still may return TRUE if GetNextToken() wasn't called yet for the
	110	// last trailing empty token
	111	return m_mode == wxTOKEN_RET_EMPTY_ALL ? m_hasMore : FALSE;
	112	}
	113	else
	114	{
	115	// there are non delimiter characters left, hence we do have more
	116	// tokens
	117	return TRUE;
	118	}
	119	}
	120
	121	// count the number of tokens in the string
	122	size_t wxStringTokenizer::CountTokens() const
	123	{
	124	wxCHECK_MSG( IsOk(), 0, _T("you should call SetString() first") );
	125
	126	// VZ: this function is IMHO not very useful, so it's probably not very
	127	// important if it's implementation here is not as efficient as it
	128	// could be - but OTOH like this we're sure to get the correct answer
	129	// in all modes
	130	wxStringTokenizer self = (wxStringTokenizer )this; // const_cast
	131	wxString stringInitial = m_string;
	132
	133	size_t count = 0;
	134	while ( self->HasMoreTokens() )
	135	{
	136	count++;
	137
	138	(void)self->GetNextToken();
	139	}
	140
	141	self->Reinit(stringInitial);
	142
	143	return count;
	144	}
	145
	146	// ----------------------------------------------------------------------------
	147	// token extraction
	148	// ----------------------------------------------------------------------------
	149
	150	wxString wxStringTokenizer::GetNextToken()
	151	{
	152	// strtok() doesn't return empty tokens, all other modes do
	153	bool allowEmpty = m_mode != wxTOKEN_STRTOK;
	154
	155	wxString token;
	156	do
	157	{
	158	if ( !HasMoreTokens() )
	159	{
	160	break;
	161	}
	162	// find the end of this token
	163	size_t pos = m_string.find_first_of(m_delims);
	164
	165	// and the start of the next one
	166	if ( pos == wxString::npos )
	167	{
	168	// no more delimiters, the token is everything till the end of
	169	// string
	170	token = m_string;
	171
	172	m_pos += m_string.length();
	173	m_string.clear();
	174
	175	// no more tokens in this string, even in wxTOKEN_RET_EMPTY_ALL
	176	// mode (we will return the trailing one right now in this case)
	177	m_hasMore = FALSE;
	178	}
	179	else
	180	{
	181	size_t pos2 = pos + 1;
	182
	183	// in wxTOKEN_RET_DELIMS mode we return the delimiter character
	184	// with token
	185	token = wxString(m_string, m_mode == wxTOKEN_RET_DELIMS ? pos2
	186	: pos);
	187
	188	// remove token with the following it delimiter from string
	189	m_string.erase(0, pos2);
	190
	191	// keep track of the position in the original string too
	192	m_pos += pos2;
	193	}
	194	}
	195	while ( !allowEmpty && token.empty() );
	196
	197	return token;
	198	}