[wxWidgets.git] / src / common / tokenzr.cpp

/////////////////////////////////////////////////////////////////////////////
// Name:        tokenzr.cpp
// Purpose:     String tokenizer
// Author:      Guilhem Lavaux
// Modified by: Vadim Zeitlin (almost full rewrite)
// Created:     04/22/98
// RCS-ID:      $Id$
// Copyright:   (c) Guilhem Lavaux
// Licence:     wxWindows licence
/////////////////////////////////////////////////////////////////////////////

// ============================================================================
// declarations
// ============================================================================

// ----------------------------------------------------------------------------
// headers
// ----------------------------------------------------------------------------

#if defined(__GNUG__) && !defined(NO_GCC_PRAGMA)
    #pragma implementation "tokenzr.h"
#endif

// For compilers that support precompilation, includes "wx.h".
#include "wx/wxprec.h"

#ifdef __BORLANDC__
    #pragma hdrstop
#endif

#include "wx/tokenzr.h"
#include "wx/arrstr.h"

// Required for wxIs... functions
#include <ctype.h>

// ============================================================================
// implementation
// ============================================================================

// ----------------------------------------------------------------------------
// wxStringTokenizer construction
// ----------------------------------------------------------------------------

wxStringTokenizer::wxStringTokenizer(const wxString& str,
                                     const wxString& delims,
                                     wxStringTokenizerMode mode)
{
    SetString(str, delims, mode);
}

void wxStringTokenizer::SetString(const wxString& str,
                                  const wxString& delims,
                                  wxStringTokenizerMode mode)
{
    if ( mode == wxTOKEN_DEFAULT )
    {
        // by default, we behave like strtok() if the delimiters are only
        // whitespace characters and as wxTOKEN_RET_EMPTY otherwise (for
        // whitespace delimiters, strtok() behaviour is better because we want
        // to count consecutive spaces as one delimiter)
        const wxChar *p;
        for ( p = delims.c_str(); *p; p++ )
        {
            if ( !wxIsspace(*p) )
                break;
        }

        if ( *p )
        {
            // not whitespace char in delims
            mode = wxTOKEN_RET_EMPTY;
        }
        else
        {
            // only whitespaces
            mode = wxTOKEN_STRTOK;
        }
    }

    m_delims = delims;
    m_mode = mode;

    Reinit(str);
}

void wxStringTokenizer::Reinit(const wxString& str)
{
    wxASSERT_MSG( IsOk(), _T("you should call SetString() first") );

    m_string = str;
    m_pos = 0;

    // empty string doesn't have any tokens
    m_hasMore = !m_string.empty();
}

// ----------------------------------------------------------------------------
// access to the tokens
// ----------------------------------------------------------------------------

// do we have more of them?
bool wxStringTokenizer::HasMoreTokens() const
{
    wxCHECK_MSG( IsOk(), false, _T("you should call SetString() first") );

    if ( m_string.find_first_not_of(m_delims) == wxString::npos )
    {
        // no non empty tokens left, but in 2 cases we still may return true if
        // GetNextToken() wasn't called yet for this empty token:
        //
        //   a) in wxTOKEN_RET_EMPTY_ALL mode we always do it
        //   b) in wxTOKEN_RET_EMPTY mode we do it in the special case of a
        //      string containing only the delimiter: then there is an empty
        //      token just before it
        return (m_mode == wxTOKEN_RET_EMPTY_ALL) ||
               (m_mode == wxTOKEN_RET_EMPTY && m_pos == 0)
                    ? m_hasMore : false;
    }
    else
    {
        // there are non delimiter characters left, hence we do have more
        // tokens
        return true;
    }
}

// count the number of tokens in the string
size_t wxStringTokenizer::CountTokens() const
{
    wxCHECK_MSG( IsOk(), 0, _T("you should call SetString() first") );

    // VZ: this function is IMHO not very useful, so it's probably not very
    //     important if it's implementation here is not as efficient as it
    //     could be - but OTOH like this we're sure to get the correct answer
    //     in all modes
    wxStringTokenizer *self = (wxStringTokenizer *)this;    // const_cast
    wxString stringInitial = m_string;

    size_t count = 0;
    while ( self->HasMoreTokens() )
    {
        count++;

        (void)self->GetNextToken();
    }

    self->Reinit(stringInitial);

    return count;
}

// ----------------------------------------------------------------------------
// token extraction
// ----------------------------------------------------------------------------

wxString wxStringTokenizer::GetNextToken()
{
    // strtok() doesn't return empty tokens, all other modes do
    bool allowEmpty = m_mode != wxTOKEN_STRTOK;

    wxString token;
    do
    {
        if ( !HasMoreTokens() )
        {
            break;
        }
        // find the end of this token
        size_t pos = m_string.find_first_of(m_delims);

        // and the start of the next one
        if ( pos == wxString::npos )
        {
            // no more delimiters, the token is everything till the end of
            // string
            token = m_string;

            m_pos += m_string.length();
            m_string.clear();

            // no more tokens in this string, even in wxTOKEN_RET_EMPTY_ALL
            // mode (we will return the trailing one right now in this case)
            m_hasMore = false;
        }
        else
        {
            size_t pos2 = pos + 1;

            // in wxTOKEN_RET_DELIMS mode we return the delimiter character
            // with token
            token = wxString(m_string, m_mode == wxTOKEN_RET_DELIMS ? pos2
                                                                    : pos);

            // remove token with the following it delimiter from string
            m_string.erase(0, pos2);

            // keep track of the position in the original string too
            m_pos += pos2;
        }
    }
    while ( !allowEmpty && token.empty() );

    return token;
}

// ----------------------------------------------------------------------------
// public functions
// ----------------------------------------------------------------------------

wxArrayString wxStringTokenize(const wxString& str,
                               const wxString& delims,
                               wxStringTokenizerMode mode)
{
    wxArrayString tokens;
    wxStringTokenizer tk(str, delims, mode);
    while ( tk.HasMoreTokens() )
    {
        tokens.Add(tk.GetNextToken());
    }

    return tokens;
}
Commit	Line	Data
f4ada568 GL	1	/////////////////////////////////////////////////////////////////////////////
	2	// Name: tokenzr.cpp
	3	// Purpose: String tokenizer
	4	// Author: Guilhem Lavaux
1e6feb95	5	// Modified by: Vadim Zeitlin (almost full rewrite)
f4ada568 GL	6	// Created: 04/22/98
	7	// RCS-ID: $Id$
	8	// Copyright: (c) Guilhem Lavaux
65571936	9	// Licence: wxWindows licence
f4ada568 GL	10	/////////////////////////////////////////////////////////////////////////////
f4ada568 GL	11
bbf8fc53 VZ	12	// ============================================================================
	13	// declarations
	14	// ============================================================================
	15
	16	// ----------------------------------------------------------------------------
	17	// headers
	18	// ----------------------------------------------------------------------------
	19
14f355c2	20	#if defined(__GNUG__) && !defined(NO_GCC_PRAGMA)
85833f5c	21	#pragma implementation "tokenzr.h"
f4ada568 GL	22	#endif
f4ada568 GL	23
fcc6dddd JS	24	// For compilers that support precompilation, includes "wx.h".
	25	#include "wx/wxprec.h"
	26
	27	#ifdef __BORLANDC__
85833f5c	28	#pragma hdrstop
fcc6dddd JS	29	#endif
fcc6dddd JS	30
f4ada568	31	#include "wx/tokenzr.h"
df5168c4	32	#include "wx/arrstr.h"
f4ada568	33
3f8e5072 JS	34	// Required for wxIs... functions
	35	#include <ctype.h>
	36
bbf8fc53 VZ	37	// ============================================================================
	38	// implementation
	39	// ============================================================================
	40
	41	// ----------------------------------------------------------------------------
	42	// wxStringTokenizer construction
	43	// ----------------------------------------------------------------------------
	44
7c968cee	45	wxStringTokenizer::wxStringTokenizer(const wxString& str,
f4ada568	46	const wxString& delims,
7c968cee	47	wxStringTokenizerMode mode)
bbf8fc53	48	{
7c968cee	49	SetString(str, delims, mode);
bbf8fc53 VZ	50	}
bbf8fc53 VZ	51
7c968cee	52	void wxStringTokenizer::SetString(const wxString& str,
bbf8fc53	53	const wxString& delims,
7c968cee	54	wxStringTokenizerMode mode)
f4ada568	55	{
7c968cee VZ	56	if ( mode == wxTOKEN_DEFAULT )
	57	{
	58	// by default, we behave like strtok() if the delimiters are only
	59	// whitespace characters and as wxTOKEN_RET_EMPTY otherwise (for
	60	// whitespace delimiters, strtok() behaviour is better because we want
	61	// to count consecutive spaces as one delimiter)
	62	const wxChar *p;
	63	for ( p = delims.c_str(); *p; p++ )
	64	{
	65	if ( !wxIsspace(*p) )
	66	break;
	67	}
	68
	69	if ( *p )
	70	{
	71	// not whitespace char in delims
	72	mode = wxTOKEN_RET_EMPTY;
	73	}
	74	else
	75	{
	76	// only whitespaces
	77	mode = wxTOKEN_STRTOK;
	78	}
	79	}
	80
85833f5c	81	m_delims = delims;
7c968cee	82	m_mode = mode;
bbf8fc53	83
7c968cee	84	Reinit(str);
f4ada568 GL	85	}
f4ada568 GL	86
7c968cee	87	void wxStringTokenizer::Reinit(const wxString& str)
f4ada568	88	{
7c968cee VZ	89	wxASSERT_MSG( IsOk(), _T("you should call SetString() first") );
	90
	91	m_string = str;
	92	m_pos = 0;
	93
	94	// empty string doesn't have any tokens
	95	m_hasMore = !m_string.empty();
f4ada568 GL	96	}
f4ada568 GL	97
bbf8fc53	98	// ----------------------------------------------------------------------------
7c968cee	99	// access to the tokens
bbf8fc53 VZ	100	// ----------------------------------------------------------------------------
bbf8fc53 VZ	101
7c968cee VZ	102	// do we have more of them?
7c968cee VZ	103	bool wxStringTokenizer::HasMoreTokens() const
f4ada568	104	{
cb719f2e	105	wxCHECK_MSG( IsOk(), false, _T("you should call SetString() first") );
7c968cee VZ	106
7c968cee VZ	107	if ( m_string.find_first_not_of(m_delims) == wxString::npos )
bbf8fc53	108	{
cb719f2e	109	// no non empty tokens left, but in 2 cases we still may return true if
1e6feb95 VZ	110	// GetNextToken() wasn't called yet for this empty token:
	111	//
	112	// a) in wxTOKEN_RET_EMPTY_ALL mode we always do it
	113	// b) in wxTOKEN_RET_EMPTY mode we do it in the special case of a
	114	// string containing only the delimiter: then there is an empty
	115	// token just before it
	116	return (m_mode == wxTOKEN_RET_EMPTY_ALL) \|\|
	117	(m_mode == wxTOKEN_RET_EMPTY && m_pos == 0)
cb719f2e	118	? m_hasMore : false;
7c968cee VZ	119	}
	120	else
	121	{
	122	// there are non delimiter characters left, hence we do have more
	123	// tokens
cb719f2e	124	return true;
7c968cee VZ	125	}
7c968cee VZ	126	}
bbf8fc53	127
7c968cee VZ	128	// count the number of tokens in the string
	129	size_t wxStringTokenizer::CountTokens() const
	130	{
	131	wxCHECK_MSG( IsOk(), 0, _T("you should call SetString() first") );
bbf8fc53	132
7c968cee VZ	133	// VZ: this function is IMHO not very useful, so it's probably not very
	134	// important if it's implementation here is not as efficient as it
	135	// could be - but OTOH like this we're sure to get the correct answer
	136	// in all modes
	137	wxStringTokenizer self = (wxStringTokenizer )this; // const_cast
	138	wxString stringInitial = m_string;
bbf8fc53	139
7c968cee VZ	140	size_t count = 0;
7c968cee VZ	141	while ( self->HasMoreTokens() )
bbf8fc53 VZ	142	{
bbf8fc53 VZ	143	count++;
7c968cee VZ	144
7c968cee VZ	145	(void)self->GetNextToken();
bbf8fc53 VZ	146	}
bbf8fc53 VZ	147
7c968cee VZ	148	self->Reinit(stringInitial);
7c968cee VZ	149
bbf8fc53 VZ	150	return count;
	151	}
	152
	153	// ----------------------------------------------------------------------------
	154	// token extraction
	155	// ----------------------------------------------------------------------------
	156
	157	wxString wxStringTokenizer::GetNextToken()
	158	{
7c968cee VZ	159	// strtok() doesn't return empty tokens, all other modes do
	160	bool allowEmpty = m_mode != wxTOKEN_STRTOK;
	161
bbf8fc53	162	wxString token;
7c968cee	163	do
bbf8fc53	164	{
7c968cee	165	if ( !HasMoreTokens() )
85833f5c	166	{
7c968cee	167	break;
85833f5c	168	}
7c968cee VZ	169	// find the end of this token
	170	size_t pos = m_string.find_first_of(m_delims);
	171
	172	// and the start of the next one
	173	if ( pos == wxString::npos )
85833f5c	174	{
7c968cee VZ	175	// no more delimiters, the token is everything till the end of
	176	// string
	177	token = m_string;
	178
	179	m_pos += m_string.length();
	180	m_string.clear();
bbf8fc53	181
7c968cee VZ	182	// no more tokens in this string, even in wxTOKEN_RET_EMPTY_ALL
7c968cee VZ	183	// mode (we will return the trailing one right now in this case)
cb719f2e	184	m_hasMore = false;
85833f5c	185	}
7c968cee VZ	186	else
	187	{
	188	size_t pos2 = pos + 1;
f4ada568	189
7c968cee VZ	190	// in wxTOKEN_RET_DELIMS mode we return the delimiter character
	191	// with token
	192	token = wxString(m_string, m_mode == wxTOKEN_RET_DELIMS ? pos2
	193	: pos);
dab58492	194
7c968cee VZ	195	// remove token with the following it delimiter from string
7c968cee VZ	196	m_string.erase(0, pos2);
bbf8fc53	197
7c968cee VZ	198	// keep track of the position in the original string too
	199	m_pos += pos2;
	200	}
85833f5c	201	}
7c968cee	202	while ( !allowEmpty && token.empty() );
bbf8fc53 VZ	203
bbf8fc53 VZ	204	return token;
f4ada568	205	}
1e6feb95 VZ	206
	207	// ----------------------------------------------------------------------------
	208	// public functions
	209	// ----------------------------------------------------------------------------
	210
	211	wxArrayString wxStringTokenize(const wxString& str,
	212	const wxString& delims,
	213	wxStringTokenizerMode mode)
	214	{
	215	wxArrayString tokens;
	216	wxStringTokenizer tk(str, delims, mode);
	217	while ( tk.HasMoreTokens() )
	218	{
	219	tokens.Add(tk.GetNextToken());
	220	}
	221
	222	return tokens;
	223	}