forked from boostorg/regex
Almost complete regex implementation now...
[SVN r22718]
This commit is contained in:
@@ -174,6 +174,7 @@ using std::distance;
|
||||
# ifdef BOOST_MSVC
|
||||
// warning suppression with VC6:
|
||||
# pragma warning(disable: 4800)
|
||||
# pragma warning(disable: 4786)
|
||||
# endif
|
||||
# define BOOST_REGEX_MAKE_BOOL(x) static_cast<bool>(x)
|
||||
#endif
|
||||
@@ -367,12 +368,14 @@ BOOST_REGEX_DECL void BOOST_REGEX_CALL reset_stack_guard_page();
|
||||
namespace boost{
|
||||
namespace re_detail{
|
||||
|
||||
BOOST_REGEX_DECL void BOOST_REGEX_CALL raise_runtime_error(const std::runtime_error& ex);
|
||||
|
||||
template <class traits>
|
||||
void raise_error(const traits& t, unsigned code)
|
||||
{
|
||||
(void)t; // warning suppression
|
||||
std::runtime_error e(t.error_string(code));
|
||||
throw_exception(e);
|
||||
::boost::re_detail::raise_runtime_error(e);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -89,7 +89,7 @@ class static_mutex
|
||||
{
|
||||
public:
|
||||
typedef scoped_static_mutex_lock scoped_lock;
|
||||
volatile boost::int32_t m_mutex;
|
||||
boost::int32_t m_mutex;
|
||||
};
|
||||
|
||||
#define BOOST_STATIC_MUTEX_INIT { 0, }
|
||||
|
||||
@@ -198,6 +198,7 @@ protected:
|
||||
re_syntax_base* m_last_state; // the last state we added
|
||||
bool m_icase; // true for case insensitive matches
|
||||
unsigned m_repeater_id; // the id of the next repeater
|
||||
bool m_has_backrefs; // true if there are actually any backrefs
|
||||
unsigned m_backrefs; // bitmask of permitted backrefs
|
||||
boost::uintmax_t m_bad_repeats; // bitmask of repeats we can't deduce a startmap for;
|
||||
typename traits::char_class_type m_word_mask; // mask used to determine if a character is a word character
|
||||
@@ -211,17 +212,19 @@ private:
|
||||
|
||||
void fixup_pointers(re_syntax_base* state);
|
||||
void create_startmaps(re_syntax_base* state);
|
||||
void create_startmap(re_syntax_base* state, unsigned char* map, unsigned int* pnull, unsigned char mask);
|
||||
int calculate_backstep(re_syntax_base* state);
|
||||
void create_startmap(re_syntax_base* state, unsigned char* l_map, unsigned int* pnull, unsigned char mask);
|
||||
unsigned get_restart_type(re_syntax_base* state);
|
||||
void set_all_masks(unsigned char* bits, unsigned char);
|
||||
bool is_bad_repeat(re_syntax_base* pt);
|
||||
void set_bad_repeat(re_syntax_base* pt);
|
||||
syntax_element_type get_repeat_type(re_syntax_base* state);
|
||||
void probe_leading_repeat(re_syntax_base* state);
|
||||
};
|
||||
|
||||
template <class charT, class traits>
|
||||
basic_regex_creator<charT, traits>::basic_regex_creator(regex_data<charT, traits>* data)
|
||||
: m_pdata(data), m_traits(data->m_traits), m_last_state(0), m_repeater_id(0), m_backrefs(0)
|
||||
: m_pdata(data), m_traits(data->m_traits), m_last_state(0), m_repeater_id(0), m_has_backrefs(false), m_backrefs(0)
|
||||
{
|
||||
m_pdata->m_data.clear();
|
||||
static const charT w = 'w';
|
||||
@@ -244,6 +247,9 @@ basic_regex_creator<charT, traits>::basic_regex_creator(regex_data<charT, traits
|
||||
template <class charT, class traits>
|
||||
re_syntax_base* basic_regex_creator<charT, traits>::append_state(syntax_element_type t, std::size_t s)
|
||||
{
|
||||
// if the state is a backref then make a note of it:
|
||||
if(t == syntax_element_backref)
|
||||
this->m_has_backrefs = true;
|
||||
// append a new state, start by aligning our last one:
|
||||
m_pdata->m_data.align();
|
||||
// set the offset to the next state in our last one:
|
||||
@@ -538,7 +544,7 @@ re_syntax_base* basic_regex_creator<charT, traits>::append_set(
|
||||
return 0; // invalid or unsupported equivalence class
|
||||
for(unsigned i = 0; i < (1u << CHAR_BIT); ++i)
|
||||
{
|
||||
charT c(i);
|
||||
charT c(static_cast<charT>(i));
|
||||
string_type s2 = this->m_traits.transform_primary(&c, &c+1);
|
||||
if(s == s2)
|
||||
result->_map[i] = true;
|
||||
@@ -585,6 +591,8 @@ void basic_regex_creator<charT, traits>::finalize(const charT* p1, const charT*
|
||||
create_startmap(m_pdata->m_first_state, m_pdata->m_startmap, &(m_pdata->m_can_be_null), mask_all);
|
||||
// get the restart type:
|
||||
m_pdata->m_restart_type = get_restart_type(m_pdata->m_first_state);
|
||||
// optimise a leading repeat if there is one:
|
||||
probe_leading_repeat(m_pdata->m_first_state);
|
||||
}
|
||||
|
||||
template <class charT, class traits>
|
||||
@@ -645,6 +653,11 @@ void basic_regex_creator<charT, traits>::create_startmaps(re_syntax_base* state)
|
||||
// adjust the type of the state to allow for faster matching:
|
||||
state->type = this->get_repeat_type(state);
|
||||
return;
|
||||
case syntax_element_backstep:
|
||||
// we need to calculate how big the backstep is:
|
||||
static_cast<re_brace*>(state)->index
|
||||
= this->calculate_backstep(state->next.p);
|
||||
// fall through:
|
||||
default:
|
||||
state = state->next.p;
|
||||
}
|
||||
@@ -652,7 +665,65 @@ void basic_regex_creator<charT, traits>::create_startmaps(re_syntax_base* state)
|
||||
}
|
||||
|
||||
template <class charT, class traits>
|
||||
void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state, unsigned char* map, unsigned int* pnull, unsigned char mask)
|
||||
int basic_regex_creator<charT, traits>::calculate_backstep(re_syntax_base* state)
|
||||
{
|
||||
typedef typename traits::char_class_type mask_type;
|
||||
int result = 0;
|
||||
while(state)
|
||||
{
|
||||
switch(state->type)
|
||||
{
|
||||
case syntax_element_startmark:
|
||||
if((static_cast<re_brace*>(state)->index == -1)
|
||||
|| (static_cast<re_brace*>(state)->index == -2))
|
||||
{
|
||||
state = static_cast<re_jump*>(state->next.p)->alt.p->next.p;
|
||||
continue;
|
||||
}
|
||||
else if(static_cast<re_brace*>(state)->index == -3)
|
||||
{
|
||||
state = state->next.p->next.p;
|
||||
continue;
|
||||
}
|
||||
break;
|
||||
case syntax_element_endmark:
|
||||
if((static_cast<re_brace*>(state)->index == -1)
|
||||
|| (static_cast<re_brace*>(state)->index == -2))
|
||||
return result;
|
||||
case syntax_element_literal:
|
||||
result += static_cast<re_literal*>(state)->length;
|
||||
break;
|
||||
case syntax_element_wild:
|
||||
case syntax_element_set:
|
||||
result += 1;
|
||||
break;
|
||||
case syntax_element_backref:
|
||||
case syntax_element_rep:
|
||||
case syntax_element_combining:
|
||||
case syntax_element_dot_rep:
|
||||
case syntax_element_char_rep:
|
||||
case syntax_element_short_set_rep:
|
||||
case syntax_element_long_set_rep:
|
||||
case syntax_element_backstep:
|
||||
return -1;
|
||||
case syntax_element_long_set:
|
||||
if(static_cast<re_set_long<mask_type>*>(state)->singleton == 0)
|
||||
return -1;
|
||||
result += 1;
|
||||
break;
|
||||
case syntax_element_jump:
|
||||
state = static_cast<re_jump*>(state)->alt.p;
|
||||
continue;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
state = state->next.p;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
template <class charT, class traits>
|
||||
void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state, unsigned char* l_map, unsigned int* pnull, unsigned char mask)
|
||||
{
|
||||
int not_last_jump = 1;
|
||||
while(state)
|
||||
@@ -661,16 +732,16 @@ void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state,
|
||||
{
|
||||
case syntax_element_literal:
|
||||
{
|
||||
// don't set anything in *pnull, set each element in map
|
||||
// don't set anything in *pnull, set each element in l_map
|
||||
// that could match the first character in the literal:
|
||||
if(map)
|
||||
if(l_map)
|
||||
{
|
||||
map[0] |= mask_init;
|
||||
l_map[0] |= mask_init;
|
||||
charT first_char = *static_cast<charT*>(static_cast<void*>(static_cast<re_literal*>(state) + 1));
|
||||
for(unsigned int i = 0; i < (1u << CHAR_BIT); ++i)
|
||||
{
|
||||
if(m_traits.translate(static_cast<charT>(i), m_icase) == first_char)
|
||||
map[i] |= mask;
|
||||
l_map[i] |= mask;
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -678,11 +749,11 @@ void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state,
|
||||
case syntax_element_end_line:
|
||||
{
|
||||
// next character must be a line separator (if there is one):
|
||||
if(map)
|
||||
if(l_map)
|
||||
{
|
||||
map[0] |= mask_init;
|
||||
map['\n'] |= mask;
|
||||
map['\r'] |= mask;
|
||||
l_map[0] |= mask_init;
|
||||
l_map['\n'] |= mask;
|
||||
l_map['\r'] |= mask;
|
||||
}
|
||||
// now figure out if we can match a NULL string at this point:
|
||||
if(pnull)
|
||||
@@ -697,13 +768,13 @@ void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state,
|
||||
case syntax_element_wild:
|
||||
{
|
||||
// can't be null, any character can match:
|
||||
set_all_masks(map, mask);
|
||||
set_all_masks(l_map, mask);
|
||||
return;
|
||||
}
|
||||
case syntax_element_match:
|
||||
{
|
||||
// must be null, any character can match:
|
||||
set_all_masks(map, mask);
|
||||
set_all_masks(l_map, mask);
|
||||
if(pnull)
|
||||
*pnull |= mask;
|
||||
return;
|
||||
@@ -711,14 +782,14 @@ void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state,
|
||||
case syntax_element_word_start:
|
||||
{
|
||||
// recurse, then AND with all the word characters:
|
||||
create_startmap(state->next.p, map, pnull, mask);
|
||||
if(map)
|
||||
create_startmap(state->next.p, l_map, pnull, mask);
|
||||
if(l_map)
|
||||
{
|
||||
map[0] |= mask_init;
|
||||
l_map[0] |= mask_init;
|
||||
for(unsigned int i = 0; i < (1u << CHAR_BIT); ++i)
|
||||
{
|
||||
if(!m_traits.is_class(static_cast<charT>(i), m_word_mask))
|
||||
map[i] &= static_cast<unsigned char>(~mask);
|
||||
l_map[i] &= static_cast<unsigned char>(~mask);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -726,14 +797,14 @@ void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state,
|
||||
case syntax_element_word_end:
|
||||
{
|
||||
// recurse, then AND with all the word characters:
|
||||
create_startmap(state->next.p, map, pnull, mask);
|
||||
if(map)
|
||||
create_startmap(state->next.p, l_map, pnull, mask);
|
||||
if(l_map)
|
||||
{
|
||||
map[0] |= mask_init;
|
||||
l_map[0] |= mask_init;
|
||||
for(unsigned int i = 0; i < (1u << CHAR_BIT); ++i)
|
||||
{
|
||||
if(m_traits.is_class(static_cast<charT>(i), m_word_mask))
|
||||
map[i] &= static_cast<unsigned char>(~mask);
|
||||
l_map[i] &= static_cast<unsigned char>(~mask);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -746,32 +817,32 @@ void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state,
|
||||
return;
|
||||
}
|
||||
case syntax_element_long_set:
|
||||
if(map)
|
||||
if(l_map)
|
||||
{
|
||||
typedef typename traits::char_class_type mask_type;
|
||||
if(static_cast<re_set_long<mask_type>*>(state)->singleton)
|
||||
{
|
||||
map[0] |= mask_init;
|
||||
l_map[0] |= mask_init;
|
||||
for(unsigned int i = 0; i < (1u << CHAR_BIT); ++i)
|
||||
{
|
||||
charT c = static_cast<charT>(i);
|
||||
if(&c != re_is_set_member(&c, &c + 1, static_cast<re_set_long<mask_type>*>(state), *m_pdata))
|
||||
map[i] |= mask;
|
||||
l_map[i] |= mask;
|
||||
}
|
||||
}
|
||||
else
|
||||
set_all_masks(map, mask);
|
||||
set_all_masks(l_map, mask);
|
||||
}
|
||||
return;
|
||||
case syntax_element_set:
|
||||
if(map)
|
||||
if(l_map)
|
||||
{
|
||||
map[0] |= mask_init;
|
||||
l_map[0] |= mask_init;
|
||||
for(unsigned int i = 0; i < (1u << CHAR_BIT); ++i)
|
||||
{
|
||||
if(static_cast<re_set*>(state)->_map[
|
||||
static_cast<unsigned char>(m_traits.translate(static_cast<charT>(i), this->m_icase))])
|
||||
map[i] |= mask;
|
||||
l_map[i] |= mask;
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -790,14 +861,14 @@ void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state,
|
||||
re_alt* rep = static_cast<re_alt*>(state);
|
||||
if(rep->_map[0] & mask_init)
|
||||
{
|
||||
if(map)
|
||||
if(l_map)
|
||||
{
|
||||
// copy previous results:
|
||||
map[0] |= mask_init;
|
||||
l_map[0] |= mask_init;
|
||||
for(unsigned int i = 0; i <= UCHAR_MAX; ++i)
|
||||
{
|
||||
if(rep->_map[i] & mask_any)
|
||||
map[i] |= mask;
|
||||
l_map[i] |= mask;
|
||||
}
|
||||
}
|
||||
if(pnull)
|
||||
@@ -812,29 +883,53 @@ void basic_regex_creator<charT, traits>::create_startmap(re_syntax_base* state,
|
||||
// so take the union of the two options:
|
||||
if(is_bad_repeat(state))
|
||||
{
|
||||
set_all_masks(map, mask);
|
||||
set_all_masks(l_map, mask);
|
||||
return;
|
||||
}
|
||||
set_bad_repeat(state);
|
||||
create_startmap(state->next.p, map, pnull, mask);
|
||||
if((state->type == syntax_element_alt)
|
||||
create_startmap(state->next.p, l_map, pnull, mask);
|
||||
if((state->type == syntax_element_alt)
|
||||
|| (static_cast<re_repeat*>(state)->min == 0)
|
||||
|| (not_last_jump == 0))
|
||||
create_startmap(rep->alt.p, map, pnull, mask);
|
||||
create_startmap(rep->alt.p, l_map, pnull, mask);
|
||||
}
|
||||
}
|
||||
return;
|
||||
case syntax_element_soft_buffer_end:
|
||||
// match newline or null:
|
||||
if(map)
|
||||
if(l_map)
|
||||
{
|
||||
map[0] |= mask_init;
|
||||
map['\n'] |= mask;
|
||||
map['\r'] |= mask;
|
||||
l_map[0] |= mask_init;
|
||||
l_map['\n'] |= mask;
|
||||
l_map['\r'] |= mask;
|
||||
}
|
||||
if(pnull)
|
||||
*pnull |= mask;
|
||||
return;
|
||||
case syntax_element_endmark:
|
||||
// need to handle independent subs as a special case:
|
||||
if(static_cast<re_brace*>(state)->index == -3)
|
||||
{
|
||||
// can be null, any character can match:
|
||||
set_all_masks(l_map, mask);
|
||||
if(pnull)
|
||||
*pnull |= mask;
|
||||
return;
|
||||
}
|
||||
else
|
||||
{
|
||||
state = state->next.p;
|
||||
break;
|
||||
}
|
||||
|
||||
case syntax_element_startmark:
|
||||
// need to handle independent subs as a special case:
|
||||
if(static_cast<re_brace*>(state)->index == -3)
|
||||
{
|
||||
state = state->next.p->next.p;
|
||||
break;
|
||||
}
|
||||
// otherwise fall through:
|
||||
default:
|
||||
state = state->next.p;
|
||||
}
|
||||
@@ -962,6 +1057,48 @@ syntax_element_type basic_regex_creator<charT, traits>::get_repeat_type(re_synta
|
||||
return state->type;
|
||||
}
|
||||
|
||||
template <class charT, class traits>
|
||||
void basic_regex_creator<charT, traits>::probe_leading_repeat(re_syntax_base* state)
|
||||
{
|
||||
// enumerate our states, and see if we have a leading repeat
|
||||
// for which failed search restarts can be optimised;
|
||||
do
|
||||
{
|
||||
switch(state->type)
|
||||
{
|
||||
case syntax_element_startmark:
|
||||
if(static_cast<re_brace*>(state)->index >= 0)
|
||||
{
|
||||
state = state->next.p;
|
||||
continue;
|
||||
}
|
||||
return;
|
||||
case syntax_element_endmark:
|
||||
case syntax_element_start_line:
|
||||
case syntax_element_end_line:
|
||||
case syntax_element_word_boundary:
|
||||
case syntax_element_within_word:
|
||||
case syntax_element_word_start:
|
||||
case syntax_element_word_end:
|
||||
case syntax_element_buffer_start:
|
||||
case syntax_element_buffer_end:
|
||||
case syntax_element_restart_continue:
|
||||
state = state->next.p;
|
||||
break;
|
||||
case syntax_element_dot_rep:
|
||||
case syntax_element_char_rep:
|
||||
case syntax_element_short_set_rep:
|
||||
case syntax_element_long_set_rep:
|
||||
if(this->m_has_backrefs == 0)
|
||||
static_cast<re_repeat*>(state)->leading = true;
|
||||
// fall through:
|
||||
default:
|
||||
return;
|
||||
}
|
||||
}while(state);
|
||||
}
|
||||
|
||||
|
||||
} // namespace re_detail
|
||||
|
||||
} // namespace boost
|
||||
|
||||
@@ -249,16 +249,14 @@ bool basic_regex_parser<charT, traits>::parse_open_paren()
|
||||
//
|
||||
if((this->flags() & (regbase::main_option_type | regbase::no_perl_ex)) == 0)
|
||||
{
|
||||
if(m_traits.syntax_type(*m_position) == regex_constants::syntax_question)
|
||||
if(this->m_traits.syntax_type(*m_position) == regex_constants::syntax_question)
|
||||
return parse_perl_extension();
|
||||
}
|
||||
//
|
||||
// update our mark count, and append the required state:
|
||||
//
|
||||
unsigned markid;
|
||||
if(this->flags() & regbase::nosubs)
|
||||
markid = 0;
|
||||
else
|
||||
unsigned markid = 0;
|
||||
if(0 == (this->flags() & regbase::nosubs))
|
||||
markid = ++m_mark_count;
|
||||
re_brace* pb = static_cast<re_brace*>(this->append_state(syntax_element_startmark, sizeof(re_brace)));
|
||||
pb->index = markid;
|
||||
@@ -1070,6 +1068,10 @@ bool basic_regex_parser<charT, traits>::parse_backref()
|
||||
template <class charT, class traits>
|
||||
bool basic_regex_parser<charT, traits>::parse_QE()
|
||||
{
|
||||
#ifdef BOOST_MSVC
|
||||
#pragma warning(push)
|
||||
#pragma warning(disable:4127)
|
||||
#endif
|
||||
//
|
||||
// parse a \Q...\E sequence:
|
||||
//
|
||||
@@ -1104,6 +1106,9 @@ bool basic_regex_parser<charT, traits>::parse_QE()
|
||||
++start;
|
||||
}
|
||||
return true;
|
||||
#ifdef BOOST_MSVC
|
||||
#pragma warning(pop)
|
||||
#endif
|
||||
}
|
||||
|
||||
template <class charT, class traits>
|
||||
@@ -1114,7 +1119,7 @@ bool basic_regex_parser<charT, traits>::parse_perl_extension()
|
||||
//
|
||||
// backup some state, and prepare the way:
|
||||
//
|
||||
int markid;
|
||||
int markid = 0;
|
||||
std::ptrdiff_t jump_offset = 0;
|
||||
re_brace* pb = static_cast<re_brace*>(this->append_state(syntax_element_startmark, sizeof(re_brace)));
|
||||
std::ptrdiff_t last_paren_start = this->getoffset(pb);
|
||||
@@ -1157,6 +1162,35 @@ bool basic_regex_parser<charT, traits>::parse_perl_extension()
|
||||
this->m_pdata->m_data.align();
|
||||
m_alt_insert_point = this->m_pdata->m_data.size();
|
||||
break;
|
||||
case regex_constants::escape_type_left_word:
|
||||
{
|
||||
// a lookbehind assertion:
|
||||
if(++m_position == m_end)
|
||||
fail(REG_BADRPT, m_position - m_base);
|
||||
regex_constants::syntax_type t = this->m_traits.syntax_type(*m_position);
|
||||
if(t == regex_constants::syntax_not)
|
||||
pb->index = markid = -2;
|
||||
else if(t == regex_constants::syntax_equal)
|
||||
pb->index = markid = -1;
|
||||
else
|
||||
fail(REG_BADRPT, m_position - m_base);
|
||||
++m_position;
|
||||
jump_offset = this->getoffset(this->append_state(syntax_element_jump, sizeof(re_jump)));
|
||||
this->append_state(syntax_element_backstep, sizeof(re_brace));
|
||||
this->m_pdata->m_data.align();
|
||||
m_alt_insert_point = this->m_pdata->m_data.size();
|
||||
break;
|
||||
}
|
||||
case regex_constants::escape_type_right_word:
|
||||
//
|
||||
// an independent sub-expression:
|
||||
//
|
||||
pb->index = markid = -3;
|
||||
++m_position;
|
||||
jump_offset = this->getoffset(this->append_state(syntax_element_jump, sizeof(re_jump)));
|
||||
this->m_pdata->m_data.align();
|
||||
m_alt_insert_point = this->m_pdata->m_data.size();
|
||||
break;
|
||||
default:
|
||||
fail(REG_BADRPT, m_position - m_base);
|
||||
}
|
||||
@@ -1180,6 +1214,11 @@ bool basic_regex_parser<charT, traits>::parse_perl_extension()
|
||||
this->m_pdata->m_data.align();
|
||||
re_jump* jmp = static_cast<re_jump*>(this->getaddress(jump_offset));
|
||||
jmp->alt.i = this->m_pdata->m_data.size() - this->getoffset(jmp);
|
||||
if(this->m_last_state == jmp)
|
||||
{
|
||||
// Oops... we didn't have anything inside the assertion:
|
||||
fail(REG_EMPTY, m_position - m_base);
|
||||
}
|
||||
}
|
||||
//
|
||||
// append closing parenthesis state:
|
||||
|
||||
@@ -29,6 +29,15 @@
|
||||
#include <boost/regex/v4/primary_transform.hpp>
|
||||
#endif
|
||||
|
||||
#ifdef BOOST_HAS_ABI_HEADERS
|
||||
# include BOOST_ABI_PREFIX
|
||||
#endif
|
||||
|
||||
#ifdef BOOST_MSVC
|
||||
#pragma warning(push)
|
||||
#pragma warning(disable:4786)
|
||||
#endif
|
||||
|
||||
namespace boost{
|
||||
|
||||
//
|
||||
@@ -58,8 +67,8 @@ public:
|
||||
const charT* getnext() { return this->gptr(); }
|
||||
protected:
|
||||
std::basic_streambuf<charT, traits>* setbuf(char_type* s, streamsize n);
|
||||
typename parser_buf<charT, traits>::pos_type seekpos(pos_type sp, ::std::ios_base::openmode which);
|
||||
typename parser_buf<charT, traits>::pos_type seekoff(off_type off, ::std::ios_base::seekdir way, ::std::ios_base::openmode which);
|
||||
//typename parser_buf<charT, traits>::pos_type seekpos(pos_type sp, ::std::ios_base::openmode which);
|
||||
//typename parser_buf<charT, traits>::pos_type seekoff(off_type off, ::std::ios_base::seekdir way, ::std::ios_base::openmode which);
|
||||
private:
|
||||
parser_buf& operator=(const parser_buf&);
|
||||
parser_buf(const parser_buf&);
|
||||
@@ -73,6 +82,7 @@ parser_buf<charT, traits>::setbuf(char_type* s, streamsize n)
|
||||
return this;
|
||||
}
|
||||
|
||||
#if 0
|
||||
template<class charT, class traits>
|
||||
typename parser_buf<charT, traits>::pos_type
|
||||
parser_buf<charT, traits>::seekoff(off_type off, ::std::ios_base::seekdir way, ::std::ios_base::openmode which)
|
||||
@@ -131,7 +141,7 @@ parser_buf<charT, traits>::seekpos(pos_type sp, ::std::ios_base::openmode which)
|
||||
}
|
||||
return pos_type(off_type(-1));
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
//
|
||||
// class cpp_regex_traits_base:
|
||||
@@ -308,9 +318,25 @@ class cpp_regex_traits_implementation : public cpp_regex_traits_char_layer<charT
|
||||
{
|
||||
public:
|
||||
typedef typename cpp_regex_traits<charT>::char_class_type char_class_type;
|
||||
typedef typename std::ctype<charT>::mask native_mask_type;
|
||||
BOOST_STATIC_CONSTANT(char_class_type, mask_blank = 1u << 16);
|
||||
BOOST_STATIC_CONSTANT(char_class_type, mask_word = 1u << 17);
|
||||
BOOST_STATIC_CONSTANT(char_class_type, mask_unicode = 1u << 18);
|
||||
#ifdef __GNUC__
|
||||
BOOST_STATIC_CONSTANT(native_mask_type,
|
||||
mask_base =
|
||||
std::ctype<charT>::alnum
|
||||
| std::ctype<charT>::alpha
|
||||
| std::ctype<charT>::cntrl
|
||||
| std::ctype<charT>::digit
|
||||
| std::ctype<charT>::graph
|
||||
| std::ctype<charT>::lower
|
||||
| std::ctype<charT>::print
|
||||
| std::ctype<charT>::punct
|
||||
| std::ctype<charT>::space
|
||||
| std::ctype<charT>::upper
|
||||
| std::ctype<charT>::xdigit);
|
||||
#else
|
||||
BOOST_STATIC_CONSTANT(char_class_type,
|
||||
mask_base =
|
||||
std::ctype<charT>::alnum
|
||||
@@ -324,6 +350,7 @@ public:
|
||||
| std::ctype<charT>::space
|
||||
| std::ctype<charT>::upper
|
||||
| std::ctype<charT>::xdigit);
|
||||
#endif
|
||||
|
||||
//BOOST_STATIC_ASSERT(0 == (mask_base & (mask_word | mask_unicode)));
|
||||
|
||||
@@ -346,9 +373,9 @@ public:
|
||||
char_class_type result = lookup_classname_imp(p1, p2);
|
||||
if(result == 0)
|
||||
{
|
||||
string_type s(p1, p2);
|
||||
this->m_pctype->tolower(&*s.begin(), &*s.end());
|
||||
result = lookup_classname_imp(&*s.begin(), &*s.end());
|
||||
string_type temp(p1, p2);
|
||||
this->m_pctype->tolower(&*temp.begin(), &*temp.begin() + temp.size());
|
||||
result = lookup_classname_imp(&*temp.begin(), &*temp.begin() + temp.size());
|
||||
}
|
||||
return result;
|
||||
}
|
||||
@@ -388,20 +415,20 @@ typename cpp_regex_traits_implementation<charT>::string_type
|
||||
// the best we can do is translate to lower case, then get a regular sort key:
|
||||
{
|
||||
result.assign(p1, p2);
|
||||
m_pctype->tolower(&*result.begin(), &*result.end());
|
||||
result = this->m_pcollate->transform(&*result.begin(), &*result.end());
|
||||
this->m_pctype->tolower(&*result.begin(), &*result.begin() + result.size());
|
||||
result = this->m_pcollate->transform(&*result.begin(), &*result.begin() + result.size());
|
||||
break;
|
||||
}
|
||||
case sort_fixed:
|
||||
{
|
||||
// get a regular sort key, and then truncate it:
|
||||
result.assign(this->m_pcollate->transform(&*result.begin(), &*result.end()));
|
||||
result.assign(this->m_pcollate->transform(&*result.begin(), &*result.begin() + result.size()));
|
||||
result.erase(this->m_collate_delim);
|
||||
break;
|
||||
}
|
||||
case sort_delim:
|
||||
// get a regular sort key, and then truncate everything after the delim:
|
||||
result.assign(this->m_pcollate->transform(&*result.begin(), &*result.end()));
|
||||
result.assign(this->m_pcollate->transform(&*result.begin(), &*result.begin() + result.size()));
|
||||
std::size_t i;
|
||||
for(i = 0; i < result.size(); ++i)
|
||||
{
|
||||
@@ -425,10 +452,30 @@ typename cpp_regex_traits_implementation<charT>::string_type
|
||||
if(pos != m_custom_collate_names.end())
|
||||
return pos->second;
|
||||
}
|
||||
#ifndef BOOST_NO_TEMPLATED_ITERATOR_CONSTRUCTORS
|
||||
std::string name(p1, p2);
|
||||
#else
|
||||
std::string name;
|
||||
const charT* p0 = p1;
|
||||
while(p0 != p2)
|
||||
name.append(1, char(*p0++));
|
||||
#endif
|
||||
name = lookup_default_collate_name(name);
|
||||
#ifndef BOOST_NO_TEMPLATED_ITERATOR_CONSTRUCTORS
|
||||
if(name.size())
|
||||
return string_type(name.begin(), name.end());
|
||||
#else
|
||||
if(name.size())
|
||||
{
|
||||
string_type result;
|
||||
typedef std::string::const_iterator iter;
|
||||
iter b = name.begin();
|
||||
iter e = name.end();
|
||||
while(b != e)
|
||||
result.append(1, charT(*b++));
|
||||
return result;
|
||||
}
|
||||
#endif
|
||||
if(p2 - p1 == 1)
|
||||
return string_type(1, *p1);
|
||||
return string_type();
|
||||
@@ -731,4 +778,12 @@ static_mutex& cpp_regex_traits<charT>::get_mutex_inst()
|
||||
|
||||
} // boost
|
||||
|
||||
#ifdef BOOST_MSVC
|
||||
#pragma warning(pop)
|
||||
#endif
|
||||
|
||||
#ifdef BOOST_HAS_ABI_HEADERS
|
||||
# include BOOST_ABI_SUFFIX
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
@@ -333,6 +333,7 @@ private:
|
||||
bool match_char_repeat();
|
||||
bool match_dot_repeat_fast();
|
||||
bool match_dot_repeat_slow();
|
||||
bool match_backstep();
|
||||
bool backtrack_till_match(unsigned count);
|
||||
|
||||
// find procs stored in s_find_vtable:
|
||||
|
||||
@@ -659,6 +659,17 @@ bool perl_matcher<BidiIterator, Allocator, traits>::match_restart_continue()
|
||||
return false;
|
||||
}
|
||||
|
||||
template <class BidiIterator, class Allocator, class traits>
|
||||
bool perl_matcher<BidiIterator, Allocator, traits>::match_backstep()
|
||||
{
|
||||
std::ptrdiff_t maxlen = std::distance(search_base, position);
|
||||
if(maxlen < static_cast<const re_brace*>(pstate)->index)
|
||||
return false;
|
||||
std::advance(position, -static_cast<const re_brace*>(pstate)->index);
|
||||
pstate = pstate->next.p;
|
||||
return true;
|
||||
}
|
||||
|
||||
template <class BidiIterator, class Allocator, class traits>
|
||||
bool perl_matcher<BidiIterator, Allocator, traits>::find_restart_any()
|
||||
{
|
||||
@@ -737,7 +748,7 @@ bool perl_matcher<BidiIterator, Allocator, traits>::find_restart_line()
|
||||
return true;
|
||||
while(position != last)
|
||||
{
|
||||
while((position != last) && (*position != '\n'))
|
||||
while((position != last) && !is_separator(*position))
|
||||
++position;
|
||||
if(position == last)
|
||||
return false;
|
||||
|
||||
@@ -113,7 +113,7 @@ struct saved_single_repeat : public saved_state
|
||||
template <class BidiIterator, class Allocator, class traits>
|
||||
bool perl_matcher<BidiIterator, Allocator, traits>::match_all_states()
|
||||
{
|
||||
static matcher_proc_type const s_match_vtable[26] =
|
||||
static matcher_proc_type const s_match_vtable[27] =
|
||||
{
|
||||
(&perl_matcher<BidiIterator, Allocator, traits>::match_startmark),
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_endmark,
|
||||
@@ -141,6 +141,7 @@ bool perl_matcher<BidiIterator, Allocator, traits>::match_all_states()
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_char_repeat,
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_set_repeat,
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_long_set_repeat,
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_backstep,
|
||||
};
|
||||
|
||||
push_recursion_stopper();
|
||||
@@ -717,8 +718,9 @@ bool perl_matcher<BidiIterator, Allocator, traits>::match_long_set_repeat()
|
||||
#ifdef __BORLANDC__
|
||||
#pragma option push -w-8008 -w-8066 -w-8004
|
||||
#endif
|
||||
typedef typename traits::char_class_type mask_type;
|
||||
const re_repeat* rep = static_cast<const re_repeat*>(pstate);
|
||||
const re_set_long<typename traits::char_class_type>* set = static_cast<const re_set_long<typename traits::char_class_type>*>(pstate->next.p);
|
||||
const re_set_long<mask_type>* set = static_cast<const re_set_long<mask_type>*>(pstate->next.p);
|
||||
std::size_t count = 0;
|
||||
//
|
||||
// start by working out how much we can skip:
|
||||
@@ -1207,6 +1209,7 @@ bool perl_matcher<BidiIterator, Allocator, traits>::unwind_short_set_repeat(bool
|
||||
template <class BidiIterator, class Allocator, class traits>
|
||||
bool perl_matcher<BidiIterator, Allocator, traits>::unwind_long_set_repeat(bool r)
|
||||
{
|
||||
typedef typename traits::char_class_type mask_type;
|
||||
saved_single_repeat<BidiIterator>* pmp = static_cast<saved_single_repeat<BidiIterator>*>(m_backup_state);
|
||||
|
||||
// if we have a match, just discard this state:
|
||||
@@ -1219,7 +1222,7 @@ bool perl_matcher<BidiIterator, Allocator, traits>::unwind_long_set_repeat(bool
|
||||
const re_repeat* rep = pmp->rep;
|
||||
std::size_t count = pmp->count;
|
||||
pstate = rep->next.p;
|
||||
const re_set_long<typename traits::char_class_type>* set = static_cast<const re_set_long<typename traits::char_class_type>*>(pstate);
|
||||
const re_set_long<mask_type>* set = static_cast<const re_set_long<mask_type>*>(pstate);
|
||||
position = pmp->last_position;
|
||||
|
||||
assert(rep->type == syntax_element_long_set_rep);
|
||||
|
||||
@@ -48,7 +48,7 @@ public:
|
||||
template <class BidiIterator, class Allocator, class traits>
|
||||
bool perl_matcher<BidiIterator, Allocator, traits>::match_all_states()
|
||||
{
|
||||
static matcher_proc_type const s_match_vtable[26] =
|
||||
static matcher_proc_type const s_match_vtable[27] =
|
||||
{
|
||||
(&perl_matcher<BidiIterator, Allocator, traits>::match_startmark),
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_endmark,
|
||||
@@ -76,6 +76,7 @@ bool perl_matcher<BidiIterator, Allocator, traits>::match_all_states()
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_char_repeat,
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_set_repeat,
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_long_set_repeat,
|
||||
&perl_matcher<BidiIterator, Allocator, traits>::match_backstep,
|
||||
};
|
||||
|
||||
if(state_count > max_state_count)
|
||||
|
||||
@@ -0,0 +1,132 @@
|
||||
/*
|
||||
*
|
||||
* Copyright (c) 1998-2002
|
||||
* Dr John Maddock
|
||||
*
|
||||
* Use, modification and distribution are subject to the
|
||||
* Boost Software License, Version 1.0. (See accompanying file
|
||||
* LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt)
|
||||
*
|
||||
*/
|
||||
|
||||
/*
|
||||
* LOCATION: see http://www.boost.org for most recent version.
|
||||
* FILE: primary_transform.hpp
|
||||
* VERSION: see <boost/version.hpp>
|
||||
* DESCRIPTION: Heuristically determines the sort string format in use
|
||||
* by the current locale.
|
||||
*/
|
||||
|
||||
#ifndef BOOST_REGEX_PRIMARY_TRANSFORM
|
||||
#define BOOST_REGEX_PRIMARY_TRANSFORM
|
||||
|
||||
#ifdef BOOST_HAS_ABI_HEADERS
|
||||
# include BOOST_ABI_PREFIX
|
||||
#endif
|
||||
|
||||
namespace boost{
|
||||
namespace re_detail{
|
||||
|
||||
|
||||
enum{
|
||||
sort_C,
|
||||
sort_fixed,
|
||||
sort_delim,
|
||||
sort_unknown
|
||||
};
|
||||
|
||||
template <class S, class charT>
|
||||
unsigned count_chars(const S& s, charT c)
|
||||
{
|
||||
//
|
||||
// Count how many occurances of character c occur
|
||||
// in string s: if c is a delimeter between collation
|
||||
// fields, then this should be the same value for all
|
||||
// sort keys:
|
||||
//
|
||||
unsigned int count = 0;
|
||||
for(unsigned pos = 0; pos < s.size(); ++pos)
|
||||
{
|
||||
if(s[pos] == c) ++count;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
|
||||
template <class traits, class charT>
|
||||
unsigned find_sort_syntax(const traits* pt, charT* delim)
|
||||
{
|
||||
//
|
||||
// compare 'a' with 'A' to see how similar they are,
|
||||
// should really use a-accute but we can't portably do that,
|
||||
//
|
||||
typedef typename traits::string_type string_type;
|
||||
typedef typename traits::char_type char_type;
|
||||
|
||||
// Suppress incorrect warning for MSVC
|
||||
(void)pt;
|
||||
|
||||
char_type a[2] = {'a', '\0', };
|
||||
string_type sa(pt->transform(a, a+1));
|
||||
if(sa == a)
|
||||
{
|
||||
*delim = 0;
|
||||
return sort_C;
|
||||
}
|
||||
char_type A[2] = { 'A', '\0', };
|
||||
string_type sA(pt->transform(A, A+1));
|
||||
char_type c[2] = { ';', '\0', };
|
||||
string_type sc(pt->transform(c, c+1));
|
||||
|
||||
int pos = 0;
|
||||
while((pos <= static_cast<int>(sa.size())) && (pos <= static_cast<int>(sA.size())) && (sa[pos] == sA[pos])) ++pos;
|
||||
--pos;
|
||||
if(pos < 0)
|
||||
{
|
||||
*delim = 0;
|
||||
return sort_unknown;
|
||||
}
|
||||
//
|
||||
// at this point sa[pos] is either the end of a fixed width field
|
||||
// or the character that acts as a delimiter:
|
||||
//
|
||||
charT maybe_delim = sa[pos];
|
||||
if((pos != 0) && (count_chars(sa, maybe_delim) == count_chars(sA, maybe_delim)) && (count_chars(sa, maybe_delim) == count_chars(sc, maybe_delim)))
|
||||
{
|
||||
*delim = maybe_delim;
|
||||
return sort_delim;
|
||||
}
|
||||
//
|
||||
// OK doen't look like a delimiter, try for fixed width field:
|
||||
//
|
||||
if((sa.size() == sA.size()) && (sa.size() == sc.size()))
|
||||
{
|
||||
// note assumes that the fixed width field is less than
|
||||
// numeric_limits<charT>::max(), should be true for all types
|
||||
// I can't imagine 127 character fields...
|
||||
*delim = static_cast<charT>(++pos);
|
||||
return sort_fixed;
|
||||
}
|
||||
//
|
||||
// don't know what it is:
|
||||
//
|
||||
*delim = 0;
|
||||
return sort_unknown;
|
||||
}
|
||||
|
||||
|
||||
} // namespace re_detail
|
||||
} // namespace boost
|
||||
|
||||
#ifdef BOOST_HAS_ABI_HEADERS
|
||||
# include BOOST_ABI_SUFFIX
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -41,9 +41,6 @@
|
||||
#ifndef BOOST_REGEX_FWD_HPP
|
||||
#include <boost/regex_fwd.hpp>
|
||||
#endif
|
||||
#ifndef BOOST_REGEX_STACK_HPP
|
||||
#include <boost/regex/v4/regex_stack.hpp>
|
||||
#endif
|
||||
#ifndef BOOST_REGEX_RAW_BUFFER_HPP
|
||||
#include <boost/regex/v4/regex_raw_buffer.hpp>
|
||||
#endif
|
||||
|
||||
@@ -35,6 +35,10 @@
|
||||
#include <boost/regex/v4/cpp_regex_traits.hpp>
|
||||
#endif
|
||||
|
||||
#ifdef BOOST_HAS_ABI_HEADERS
|
||||
# include BOOST_ABI_PREFIX
|
||||
#endif
|
||||
|
||||
namespace boost{
|
||||
|
||||
template <class charT, class implementationT >
|
||||
@@ -45,5 +49,9 @@ struct regex_traits : public implementationT
|
||||
|
||||
} // namespace boost
|
||||
|
||||
#ifdef BOOST_HAS_ABI_HEADERS
|
||||
# include BOOST_ABI_SUFFIX
|
||||
#endif
|
||||
|
||||
#endif // include
|
||||
|
||||
|
||||
@@ -19,6 +19,10 @@
|
||||
#ifndef BOOST_REGEX_TRAITS_DEFAULTS_HPP_INCLUDED
|
||||
#define BOOST_REGEX_TRAITS_DEFAULTS_HPP_INCLUDED
|
||||
|
||||
#ifdef BOOST_HAS_ABI_HEADERS
|
||||
# include BOOST_ABI_PREFIX
|
||||
#endif
|
||||
|
||||
namespace boost{ namespace re_detail{
|
||||
|
||||
BOOST_REGEX_DECL const char* BOOST_REGEX_CALL get_default_syntax(regex_constants::syntax_type n);
|
||||
@@ -77,6 +81,11 @@ inline bool is_separator(charT c)
|
||||
{
|
||||
return BOOST_REGEX_MAKE_BOOL((c == '\n') || (c == '\r') || (static_cast<int>(c) == 0x2028) || (static_cast<int>(c) == 0x2029));
|
||||
}
|
||||
template <>
|
||||
inline bool is_separator<char>(char c)
|
||||
{
|
||||
return BOOST_REGEX_MAKE_BOOL((c == '\n') || (c == '\r'));
|
||||
}
|
||||
|
||||
//
|
||||
// get a default collating element:
|
||||
@@ -99,7 +108,7 @@ struct character_pointer_range
|
||||
}
|
||||
bool operator == (const character_pointer_range& r)const
|
||||
{
|
||||
return (std::distance(p1, p2) == std::distance(r.p1, r.p2)) && std::equal(p1, p2, r.p1);
|
||||
return ((p2 - p1) == (r.p2 - r.p1)) && std::equal(p1, p2, r.p1);
|
||||
}
|
||||
};
|
||||
template <class charT>
|
||||
@@ -183,4 +192,8 @@ int parse_value(const charT*& p1, const charT* p2, const traits& traits_inst, in
|
||||
} // re_detail
|
||||
} // boost
|
||||
|
||||
#ifdef BOOST_HAS_ABI_HEADERS
|
||||
# include BOOST_ABI_SUFFIX
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
@@ -106,7 +106,9 @@ enum syntax_element_type
|
||||
syntax_element_dot_rep = syntax_element_restart_continue + 1,
|
||||
syntax_element_char_rep = syntax_element_dot_rep + 1,
|
||||
syntax_element_short_set_rep = syntax_element_char_rep + 1,
|
||||
syntax_element_long_set_rep = syntax_element_short_set_rep + 1
|
||||
syntax_element_long_set_rep = syntax_element_short_set_rep + 1,
|
||||
// a backstep for lookbehind repeats:
|
||||
syntax_element_backstep = syntax_element_long_set_rep + 1
|
||||
};
|
||||
|
||||
#ifdef BOOST_REGEX_DEBUG
|
||||
|
||||
Reference in New Issue
Block a user