libroot: Synchronize glibc regex with upstream 2.20.
More recent versions require more changes to adapt.
This commit is contained in:
@@ -1,7 +1,7 @@
|
||||
SubDir HAIKU_TOP src system libroot posix glibc regex ;
|
||||
|
||||
SubDirHdrs $(HAIKU_TOP) headers ;
|
||||
SubDirCcFlags -D_REGEX_RE_COMP -D_DEFAULT_SOURCE -DHAVE_STDBOOL_H ;
|
||||
SubDirCcFlags -D_REGEX_RE_COMP -D_DEFAULT_SOURCE -DHAVE_STDBOOL_H -DHAVE_STDINT_H ;
|
||||
|
||||
local architectureObject ;
|
||||
for architectureObject in [ MultiArchSubDirSetup ] {
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
/* Extended regular expression matching and search library.
|
||||
Copyright (C) 2002,2003,2004,2005,2006,2007,2009
|
||||
Free Software Foundation, Inc.
|
||||
Copyright (C) 2002-2014 Free Software Foundation, Inc.
|
||||
This file is part of the GNU C Library.
|
||||
Contributed by Isamu Hasegawa <[email protected]>.
|
||||
|
||||
@@ -15,9 +14,14 @@
|
||||
Lesser General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Lesser General Public
|
||||
License along with the GNU C Library; if not, write to the Free
|
||||
Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA
|
||||
02111-1307 USA. */
|
||||
License along with the GNU C Library; if not, see
|
||||
<http://www.gnu.org/licenses/>. */
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#ifdef _LIBC
|
||||
# include <locale/weight.h>
|
||||
#endif
|
||||
|
||||
static reg_errcode_t re_compile_internal (regex_t *preg, const char * pattern,
|
||||
size_t length, reg_syntax_t syntax);
|
||||
@@ -121,7 +125,7 @@ static bin_tree_t *duplicate_tree (const bin_tree_t *src, re_dfa_t *dfa);
|
||||
static void free_token (re_token_t *node);
|
||||
static reg_errcode_t free_tree (void *extra, bin_tree_t *node);
|
||||
static reg_errcode_t mark_opt_subexp (void *extra, bin_tree_t *node);
|
||||
|
||||
|
||||
/* This table gives an error message for each of the error codes listed
|
||||
in regex.h. Obviously the order here has to be same as there.
|
||||
POSIX doesn't require that we do anything for REG_NOERROR,
|
||||
@@ -201,7 +205,7 @@ const size_t __re_error_msgid_idx[] attribute_hidden =
|
||||
REG_ESIZE_IDX,
|
||||
REG_ERPAREN_IDX
|
||||
};
|
||||
|
||||
|
||||
/* Entry points for GNU code. */
|
||||
|
||||
/* re_compile_pattern is the GNU regular expression compiler: it
|
||||
@@ -377,7 +381,7 @@ re_compile_fastmap_iter (regex_t *bufp, const re_dfastate_t *init_state,
|
||||
applies to multibyte character sets; for single byte character
|
||||
sets, the SIMPLE_BRACKET again suffices. */
|
||||
if (dfa->mb_cur_max > 1
|
||||
&& (cset->nchar_classes || cset->non_match
|
||||
&& (cset->nchar_classes || cset->non_match || cset->nranges
|
||||
# ifdef _LIBC
|
||||
|| cset->nequiv_classes
|
||||
# endif /* _LIBC */
|
||||
@@ -410,8 +414,8 @@ re_compile_fastmap_iter (regex_t *bufp, const re_dfastate_t *init_state,
|
||||
!= (size_t) -1)
|
||||
re_set_fastmap (fastmap, false, *(unsigned char *) buf);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif /* RE_ENABLE_I18N */
|
||||
else if (type == OP_PERIOD
|
||||
@@ -427,7 +431,7 @@ re_compile_fastmap_iter (regex_t *bufp, const re_dfastate_t *init_state,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* Entry point for POSIX code. */
|
||||
/* regcomp takes a regular expression as a string and compiles it.
|
||||
|
||||
@@ -616,7 +620,7 @@ free_dfa_content (re_dfa_t *dfa)
|
||||
re_dfastate_t *state = entry->array[j];
|
||||
free_state (state);
|
||||
}
|
||||
re_free (entry->array);
|
||||
re_free (entry->array);
|
||||
}
|
||||
re_free (dfa->state_table);
|
||||
#ifdef RE_ENABLE_I18N
|
||||
@@ -653,7 +657,7 @@ regfree (preg)
|
||||
#ifdef _LIBC
|
||||
weak_alias (__regfree, regfree)
|
||||
#endif
|
||||
|
||||
|
||||
/* Entry points compatible with 4.2 BSD regex library. We don't define
|
||||
them unless specifically requested. */
|
||||
|
||||
@@ -722,7 +726,7 @@ libc_freeres_fn (free_mem)
|
||||
#endif
|
||||
|
||||
#endif /* _REGEX_RE_COMP */
|
||||
|
||||
|
||||
/* Internal entry point.
|
||||
Compile the regular expression PATTERN, whose length is LENGTH.
|
||||
SYNTAX indicate regular expression's syntax. */
|
||||
@@ -927,12 +931,49 @@ static void
|
||||
internal_function
|
||||
init_word_char (re_dfa_t *dfa)
|
||||
{
|
||||
int i, j, ch;
|
||||
dfa->word_ops_used = 1;
|
||||
for (i = 0, ch = 0; i < BITSET_WORDS; ++i)
|
||||
{
|
||||
int i = 0;
|
||||
int ch = 0;
|
||||
if (BE (dfa->map_notascii == 0, 1))
|
||||
{
|
||||
if (sizeof (dfa->word_char[0]) == 8)
|
||||
{
|
||||
/* The extra temporaries here avoid "implicitly truncated"
|
||||
warnings in the case when this is dead code, i.e. 32-bit. */
|
||||
const uint64_t wc0 = UINT64_C (0x03ff000000000000);
|
||||
const uint64_t wc1 = UINT64_C (0x07fffffe87fffffe);
|
||||
dfa->word_char[0] = wc0;
|
||||
dfa->word_char[1] = wc1;
|
||||
i = 2;
|
||||
}
|
||||
else if (sizeof (dfa->word_char[0]) == 4)
|
||||
{
|
||||
dfa->word_char[0] = UINT32_C (0x00000000);
|
||||
dfa->word_char[1] = UINT32_C (0x03ff0000);
|
||||
dfa->word_char[2] = UINT32_C (0x87fffffe);
|
||||
dfa->word_char[3] = UINT32_C (0x07fffffe);
|
||||
i = 4;
|
||||
}
|
||||
else
|
||||
abort ();
|
||||
ch = 128;
|
||||
|
||||
if (BE (dfa->is_utf8, 1))
|
||||
{
|
||||
memset (&dfa->word_char[i], '\0', (SBC_MAX - ch) / 8);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
int j;
|
||||
for (; i < BITSET_WORDS; ++i)
|
||||
for (j = 0; j < BITSET_WORD_BITS; ++j, ++ch)
|
||||
if (isalnum (ch) || ch == '_')
|
||||
dfa->word_char[i] |= (bitset_word_t) 1 << j;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Free the work area which are only used while compiling. */
|
||||
@@ -1000,7 +1041,11 @@ create_initial_state (re_dfa_t *dfa)
|
||||
int dest_idx = dfa->edests[node_idx].elems[0];
|
||||
if (!re_node_set_contains (&init_nodes, dest_idx))
|
||||
{
|
||||
re_node_set_merge (&init_nodes, dfa->eclosures + dest_idx);
|
||||
reg_errcode_t err = re_node_set_merge (&init_nodes,
|
||||
dfa->eclosures
|
||||
+ dest_idx);
|
||||
if (err != REG_NOERROR)
|
||||
return err;
|
||||
i = 0;
|
||||
}
|
||||
}
|
||||
@@ -1032,7 +1077,7 @@ create_initial_state (re_dfa_t *dfa)
|
||||
re_node_set_free (&init_nodes);
|
||||
return REG_NOERROR;
|
||||
}
|
||||
|
||||
|
||||
#ifdef RE_ENABLE_I18N
|
||||
/* If it is possible to do searching in single byte encoding instead of UTF-8
|
||||
to speed things up, set dfa->mb_cur_max to 1, clear is_utf8 and change
|
||||
@@ -1066,8 +1111,8 @@ optimize_utf8 (re_dfa_t *dfa)
|
||||
}
|
||||
break;
|
||||
case OP_PERIOD:
|
||||
has_period = 1;
|
||||
break;
|
||||
has_period = 1;
|
||||
break;
|
||||
case OP_BACK_REF:
|
||||
case OP_ALT:
|
||||
case END_OF_RE:
|
||||
@@ -1080,7 +1125,7 @@ optimize_utf8 (re_dfa_t *dfa)
|
||||
case SIMPLE_BRACKET:
|
||||
/* Just double check. The non-ASCII range starts at 0x80. */
|
||||
assert (0x80 % BITSET_WORD_BITS == 0);
|
||||
for (i = 0x80 / BITSET_WORD_BITS; i < BITSET_WORDS; ++i)
|
||||
for (i = 0x80 / BITSET_WORD_BITS; i < BITSET_WORDS; ++i)
|
||||
if (dfa->nodes[node].opr.sbcset[i])
|
||||
return;
|
||||
break;
|
||||
@@ -1104,7 +1149,7 @@ optimize_utf8 (re_dfa_t *dfa)
|
||||
dfa->has_mb_node = dfa->nbackref > 0 || has_period;
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
/* Analyze the structure tree, and calculate "first", "next", "edest",
|
||||
"eclosure", and "inveclosure". */
|
||||
|
||||
@@ -1161,7 +1206,7 @@ analyze (regex_t *preg)
|
||||
{
|
||||
dfa->inveclosures = re_malloc (re_node_set, dfa->nodes_len);
|
||||
if (BE (dfa->inveclosures == NULL, 0))
|
||||
return REG_ESPACE;
|
||||
return REG_ESPACE;
|
||||
ret = calc_inveclosure (dfa);
|
||||
}
|
||||
|
||||
@@ -1183,16 +1228,16 @@ postorder (bin_tree_t *root, reg_errcode_t (fn (void *, bin_tree_t *)),
|
||||
if that's the only child). */
|
||||
while (node->left || node->right)
|
||||
if (node->left)
|
||||
node = node->left;
|
||||
else
|
||||
node = node->right;
|
||||
node = node->left;
|
||||
else
|
||||
node = node->right;
|
||||
|
||||
do
|
||||
{
|
||||
reg_errcode_t err = fn (extra, node);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
return err;
|
||||
if (node->parent == NULL)
|
||||
if (node->parent == NULL)
|
||||
return REG_NOERROR;
|
||||
prev = node;
|
||||
node = node->parent;
|
||||
@@ -1226,7 +1271,7 @@ preorder (bin_tree_t *root, reg_errcode_t (fn (void *, bin_tree_t *)),
|
||||
prev = node;
|
||||
node = node->parent;
|
||||
if (!node)
|
||||
return REG_NOERROR;
|
||||
return REG_NOERROR;
|
||||
}
|
||||
node = node->right;
|
||||
}
|
||||
@@ -1249,13 +1294,13 @@ optimize_subexps (void *extra, bin_tree_t *node)
|
||||
}
|
||||
|
||||
else if (node->token.type == SUBEXP
|
||||
&& node->left && node->left->token.type == SUBEXP)
|
||||
&& node->left && node->left->token.type == SUBEXP)
|
||||
{
|
||||
int other_idx = node->left->token.opr.idx;
|
||||
|
||||
node->left = node->left->left;
|
||||
if (node->left)
|
||||
node->left->parent = node;
|
||||
node->left->parent = node;
|
||||
|
||||
dfa->subexp_map[other_idx] = dfa->subexp_map[node->token.opr.idx];
|
||||
if (other_idx < BITSET_WORD_BITS)
|
||||
@@ -1340,9 +1385,9 @@ calc_first (void *extra, bin_tree_t *node)
|
||||
node->first = node;
|
||||
node->node_idx = re_dfa_add_node (dfa, node->token);
|
||||
if (BE (node->node_idx == -1, 0))
|
||||
return REG_ESPACE;
|
||||
return REG_ESPACE;
|
||||
if (node->token.type == ANCHOR)
|
||||
dfa->nodes[node->node_idx].constraint = node->token.opr.ctx_type;
|
||||
dfa->nodes[node->node_idx].constraint = node->token.opr.ctx_type;
|
||||
}
|
||||
return REG_NOERROR;
|
||||
}
|
||||
@@ -1364,7 +1409,7 @@ calc_next (void *extra, bin_tree_t *node)
|
||||
if (node->left)
|
||||
node->left->next = node->next;
|
||||
if (node->right)
|
||||
node->right->next = node->next;
|
||||
node->right->next = node->next;
|
||||
break;
|
||||
}
|
||||
return REG_NOERROR;
|
||||
@@ -1415,7 +1460,7 @@ link_nfa_nodes (void *extra, bin_tree_t *node)
|
||||
case OP_BACK_REF:
|
||||
dfa->nexts[idx] = node->next->node_idx;
|
||||
if (node->token.type == OP_BACK_REF)
|
||||
re_node_set_init_1 (dfa->edests + idx, dfa->nexts[idx]);
|
||||
err = re_node_set_init_1 (dfa->edests + idx, dfa->nexts[idx]);
|
||||
break;
|
||||
|
||||
default:
|
||||
@@ -1643,9 +1688,10 @@ static reg_errcode_t
|
||||
calc_eclosure_iter (re_node_set *new_set, re_dfa_t *dfa, int node, int root)
|
||||
{
|
||||
reg_errcode_t err;
|
||||
int i, incomplete;
|
||||
int i;
|
||||
re_node_set eclosure;
|
||||
incomplete = 0;
|
||||
int ret;
|
||||
int incomplete = 0;
|
||||
err = re_node_set_alloc (&eclosure, dfa->edests[node].nelem + 1);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
return err;
|
||||
@@ -1690,7 +1736,9 @@ calc_eclosure_iter (re_node_set *new_set, re_dfa_t *dfa, int node, int root)
|
||||
else
|
||||
eclosure_elem = dfa->eclosures[edest];
|
||||
/* Merge the epsilon closure of `edest'. */
|
||||
re_node_set_merge (&eclosure, &eclosure_elem);
|
||||
err = re_node_set_merge (&eclosure, &eclosure_elem);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
return err;
|
||||
/* If the epsilon closure of `edest' is incomplete,
|
||||
the epsilon closure of this node is also incomplete. */
|
||||
if (dfa->eclosures[edest].nelem == 0)
|
||||
@@ -1700,8 +1748,10 @@ calc_eclosure_iter (re_node_set *new_set, re_dfa_t *dfa, int node, int root)
|
||||
}
|
||||
}
|
||||
|
||||
/* Epsilon closures include itself. */
|
||||
re_node_set_insert (&eclosure, node);
|
||||
/* An epsilon closure includes itself. */
|
||||
ret = re_node_set_insert (&eclosure, node);
|
||||
if (BE (ret < 0, 0))
|
||||
return REG_ESPACE;
|
||||
if (incomplete && !root)
|
||||
dfa->eclosures[node].nelem = 0;
|
||||
else
|
||||
@@ -1709,7 +1759,7 @@ calc_eclosure_iter (re_node_set *new_set, re_dfa_t *dfa, int node, int root)
|
||||
*new_set = eclosure;
|
||||
return REG_NOERROR;
|
||||
}
|
||||
|
||||
|
||||
/* Functions for token which are used in the parser. */
|
||||
|
||||
/* Fetch a token from INPUT.
|
||||
@@ -2046,7 +2096,7 @@ peek_token_bracket (re_token_t *token, re_string_t *input, reg_syntax_t syntax)
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
/* Functions for parser. */
|
||||
|
||||
/* Entry point of the parser.
|
||||
@@ -2113,7 +2163,11 @@ parse_reg_exp (re_string_t *regexp, regex_t *preg, re_token_t *token,
|
||||
{
|
||||
branch = parse_branch (regexp, preg, token, syntax, nest, err);
|
||||
if (BE (*err != REG_NOERROR && branch == NULL, 0))
|
||||
return NULL;
|
||||
{
|
||||
if (tree != NULL)
|
||||
postorder (tree, free_tree, NULL);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
else
|
||||
branch = NULL;
|
||||
@@ -2152,16 +2206,21 @@ parse_branch (re_string_t *regexp, regex_t *preg, re_token_t *token,
|
||||
exp = parse_expression (regexp, preg, token, syntax, nest, err);
|
||||
if (BE (*err != REG_NOERROR && exp == NULL, 0))
|
||||
{
|
||||
if (tree != NULL)
|
||||
postorder (tree, free_tree, NULL);
|
||||
return NULL;
|
||||
}
|
||||
if (tree != NULL && exp != NULL)
|
||||
{
|
||||
tree = create_tree (dfa, tree, exp, CONCAT);
|
||||
if (tree == NULL)
|
||||
bin_tree_t *newtree = create_tree (dfa, tree, exp, CONCAT);
|
||||
if (newtree == NULL)
|
||||
{
|
||||
postorder (exp, free_tree, NULL);
|
||||
postorder (tree, free_tree, NULL);
|
||||
*err = REG_ESPACE;
|
||||
return NULL;
|
||||
}
|
||||
tree = newtree;
|
||||
}
|
||||
else if (tree == NULL)
|
||||
tree = exp;
|
||||
@@ -2285,7 +2344,7 @@ parse_expression (re_string_t *regexp, regex_t *preg, re_token_t *token,
|
||||
&& dfa->word_ops_used == 0)
|
||||
init_word_char (dfa);
|
||||
if (token->opr.ctx_type == WORD_DELIM
|
||||
|| token->opr.ctx_type == NOT_WORD_DELIM)
|
||||
|| token->opr.ctx_type == NOT_WORD_DELIM)
|
||||
{
|
||||
bin_tree_t *tree_first, *tree_last;
|
||||
if (token->opr.ctx_type == WORD_DELIM)
|
||||
@@ -2293,13 +2352,13 @@ parse_expression (re_string_t *regexp, regex_t *preg, re_token_t *token,
|
||||
token->opr.ctx_type = WORD_FIRST;
|
||||
tree_first = create_token_tree (dfa, NULL, NULL, token);
|
||||
token->opr.ctx_type = WORD_LAST;
|
||||
}
|
||||
else
|
||||
{
|
||||
}
|
||||
else
|
||||
{
|
||||
token->opr.ctx_type = INSIDE_WORD;
|
||||
tree_first = create_token_tree (dfa, NULL, NULL, token);
|
||||
token->opr.ctx_type = INSIDE_NOTWORD;
|
||||
}
|
||||
}
|
||||
tree_last = create_token_tree (dfa, NULL, NULL, token);
|
||||
tree = create_tree (dfa, tree_first, tree_last, OP_ALT);
|
||||
if (BE (tree_first == NULL || tree_last == NULL || tree == NULL, 0))
|
||||
@@ -2369,14 +2428,21 @@ parse_expression (re_string_t *regexp, regex_t *preg, re_token_t *token,
|
||||
while (token->type == OP_DUP_ASTERISK || token->type == OP_DUP_PLUS
|
||||
|| token->type == OP_DUP_QUESTION || token->type == OP_OPEN_DUP_NUM)
|
||||
{
|
||||
tree = parse_dup_op (tree, regexp, dfa, token, syntax, err);
|
||||
if (BE (*err != REG_NOERROR && tree == NULL, 0))
|
||||
return NULL;
|
||||
bin_tree_t *dup_tree = parse_dup_op (tree, regexp, dfa, token, syntax, err);
|
||||
if (BE (*err != REG_NOERROR && dup_tree == NULL, 0))
|
||||
{
|
||||
if (tree != NULL)
|
||||
postorder (tree, free_tree, NULL);
|
||||
return NULL;
|
||||
}
|
||||
tree = dup_tree;
|
||||
/* In BRE consecutive duplications are not allowed. */
|
||||
if ((syntax & RE_CONTEXT_INVALID_DUP)
|
||||
&& (token->type == OP_DUP_ASTERISK
|
||||
|| token->type == OP_OPEN_DUP_NUM))
|
||||
{
|
||||
if (tree != NULL)
|
||||
postorder (tree, free_tree, NULL);
|
||||
*err = REG_BADRPT;
|
||||
return NULL;
|
||||
}
|
||||
@@ -2410,7 +2476,11 @@ parse_sub_exp (re_string_t *regexp, regex_t *preg, re_token_t *token,
|
||||
{
|
||||
tree = parse_reg_exp (regexp, preg, token, syntax, nest, err);
|
||||
if (BE (*err == REG_NOERROR && token->type != OP_CLOSE_SUBEXP, 0))
|
||||
*err = REG_EPAREN;
|
||||
{
|
||||
if (tree != NULL)
|
||||
postorder (tree, free_tree, NULL);
|
||||
*err = REG_EPAREN;
|
||||
}
|
||||
if (BE (*err != REG_NOERROR, 0))
|
||||
return NULL;
|
||||
}
|
||||
@@ -2521,6 +2591,8 @@ parse_dup_op (bin_tree_t *elem, re_string_t *regexp, re_dfa_t *dfa,
|
||||
|
||||
/* Duplicate ELEM before it is marked optional. */
|
||||
elem = duplicate_tree (elem, dfa);
|
||||
if (BE (elem == NULL, 0))
|
||||
goto parse_dup_op_espace;
|
||||
old_tree = tree;
|
||||
}
|
||||
else
|
||||
@@ -2541,11 +2613,11 @@ parse_dup_op (bin_tree_t *elem, re_string_t *regexp, re_dfa_t *dfa,
|
||||
elem = duplicate_tree (elem, dfa);
|
||||
tree = create_tree (dfa, tree, elem, CONCAT);
|
||||
if (BE (elem == NULL || tree == NULL, 0))
|
||||
goto parse_dup_op_espace;
|
||||
goto parse_dup_op_espace;
|
||||
|
||||
tree = create_tree (dfa, tree, NULL, OP_ALT);
|
||||
if (BE (tree == NULL, 0))
|
||||
goto parse_dup_op_espace;
|
||||
goto parse_dup_op_espace;
|
||||
}
|
||||
|
||||
if (old_tree)
|
||||
@@ -2626,9 +2698,9 @@ build_range_exp (bitset_t sbcset, bracket_elem_t *start_elem,
|
||||
no MBCSET if dfa->mb_cur_max == 1. */
|
||||
if (mbcset)
|
||||
{
|
||||
/* Check the space of the arrays. */
|
||||
if (BE (*range_alloc == mbcset->nranges, 0))
|
||||
{
|
||||
/* Check the space of the arrays. */
|
||||
if (BE (*range_alloc == mbcset->nranges, 0))
|
||||
{
|
||||
/* There is not enough space, need realloc. */
|
||||
wchar_t *new_array_start, *new_array_end;
|
||||
int new_nranges;
|
||||
@@ -2638,9 +2710,9 @@ build_range_exp (bitset_t sbcset, bracket_elem_t *start_elem,
|
||||
/* Use realloc since mbcset->range_starts and mbcset->range_ends
|
||||
are NULL if *range_alloc == 0. */
|
||||
new_array_start = re_realloc (mbcset->range_starts, wchar_t,
|
||||
new_nranges);
|
||||
new_nranges);
|
||||
new_array_end = re_realloc (mbcset->range_ends, wchar_t,
|
||||
new_nranges);
|
||||
new_nranges);
|
||||
|
||||
if (BE (new_array_start == NULL || new_array_end == NULL, 0))
|
||||
return REG_ESPACE;
|
||||
@@ -2648,10 +2720,10 @@ build_range_exp (bitset_t sbcset, bracket_elem_t *start_elem,
|
||||
mbcset->range_starts = new_array_start;
|
||||
mbcset->range_ends = new_array_end;
|
||||
*range_alloc = new_nranges;
|
||||
}
|
||||
}
|
||||
|
||||
mbcset->range_starts[mbcset->nranges] = start_wc;
|
||||
mbcset->range_ends[mbcset->nranges++] = end_wc;
|
||||
mbcset->range_starts[mbcset->nranges] = start_wc;
|
||||
mbcset->range_ends[mbcset->nranges++] = end_wc;
|
||||
}
|
||||
|
||||
/* Build the table for single byte characters. */
|
||||
@@ -2728,40 +2800,29 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
|
||||
/* Local function for parse_bracket_exp used in _LIBC environement.
|
||||
Seek the collating symbol entry correspondings to NAME.
|
||||
Return the index of the symbol in the SYMB_TABLE. */
|
||||
Return the index of the symbol in the SYMB_TABLE,
|
||||
or -1 if not found. */
|
||||
|
||||
auto inline int32_t
|
||||
__attribute ((always_inline))
|
||||
seek_collating_symbol_entry (name, name_len)
|
||||
const unsigned char *name;
|
||||
size_t name_len;
|
||||
seek_collating_symbol_entry (const unsigned char *name, size_t name_len)
|
||||
{
|
||||
int32_t hash = elem_hash ((const char *) name, name_len);
|
||||
int32_t elem = hash % table_size;
|
||||
if (symb_table[2 * elem] != 0)
|
||||
{
|
||||
int32_t second = hash % (table_size - 2) + 1;
|
||||
int32_t elem;
|
||||
|
||||
do
|
||||
{
|
||||
/* First compare the hashing value. */
|
||||
if (symb_table[2 * elem] == hash
|
||||
/* Compare the length of the name. */
|
||||
&& name_len == extra[symb_table[2 * elem + 1]]
|
||||
/* Compare the name. */
|
||||
&& memcmp (name, &extra[symb_table[2 * elem + 1] + 1],
|
||||
name_len) == 0)
|
||||
{
|
||||
/* Yep, this is the entry. */
|
||||
break;
|
||||
}
|
||||
|
||||
/* Next entry. */
|
||||
elem += second;
|
||||
}
|
||||
while (symb_table[2 * elem] != 0);
|
||||
}
|
||||
return elem;
|
||||
for (elem = 0; elem < table_size; elem++)
|
||||
if (symb_table[2 * elem] != 0)
|
||||
{
|
||||
int32_t idx = symb_table[2 * elem + 1];
|
||||
/* Skip the name of collating element name. */
|
||||
idx += 1 + extra[idx];
|
||||
if (/* Compare the length of the name. */
|
||||
name_len == extra[idx]
|
||||
/* Compare the name. */
|
||||
&& memcmp (name, &extra[idx + 1], name_len) == 0)
|
||||
/* Yep, this is the entry. */
|
||||
return elem;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* Local function for parse_bracket_exp used in _LIBC environment.
|
||||
@@ -2770,8 +2831,7 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
|
||||
auto inline unsigned int
|
||||
__attribute ((always_inline))
|
||||
lookup_collation_sequence_value (br_elem)
|
||||
bracket_elem_t *br_elem;
|
||||
lookup_collation_sequence_value (bracket_elem_t *br_elem)
|
||||
{
|
||||
if (br_elem->type == SB_CHAR)
|
||||
{
|
||||
@@ -2799,7 +2859,7 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
int32_t elem, idx;
|
||||
elem = seek_collating_symbol_entry (br_elem->opr.name,
|
||||
sym_name_len);
|
||||
if (symb_table[2 * elem] != 0)
|
||||
if (elem != -1)
|
||||
{
|
||||
/* We found the entry. */
|
||||
idx = symb_table[2 * elem + 1];
|
||||
@@ -2817,7 +2877,7 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
/* Return the collation sequence value. */
|
||||
return *(unsigned int *) (extra + idx);
|
||||
}
|
||||
else if (symb_table[2 * elem] == 0 && sym_name_len == 1)
|
||||
else if (sym_name_len == 1)
|
||||
{
|
||||
/* No valid character. Match it as a single byte
|
||||
character. */
|
||||
@@ -2839,11 +2899,8 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
|
||||
auto inline reg_errcode_t
|
||||
__attribute ((always_inline))
|
||||
build_range_exp (sbcset, mbcset, range_alloc, start_elem, end_elem)
|
||||
re_charset_t *mbcset;
|
||||
int *range_alloc;
|
||||
bitset_t sbcset;
|
||||
bracket_elem_t *start_elem, *end_elem;
|
||||
build_range_exp (bitset_t sbcset, re_charset_t *mbcset, int *range_alloc,
|
||||
bracket_elem_t *start_elem, bracket_elem_t *end_elem)
|
||||
{
|
||||
unsigned int ch;
|
||||
uint32_t start_collseq;
|
||||
@@ -2870,8 +2927,8 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
build below suffices. */
|
||||
if (nrules > 0 || dfa->mb_cur_max > 1)
|
||||
{
|
||||
/* Check the space of the arrays. */
|
||||
if (BE (*range_alloc == mbcset->nranges, 0))
|
||||
/* Check the space of the arrays. */
|
||||
if (BE (*range_alloc == mbcset->nranges, 0))
|
||||
{
|
||||
/* There is not enough space, need realloc. */
|
||||
uint32_t *new_array_start;
|
||||
@@ -2883,18 +2940,18 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
new_array_start = re_realloc (mbcset->range_starts, uint32_t,
|
||||
new_nranges);
|
||||
new_array_end = re_realloc (mbcset->range_ends, uint32_t,
|
||||
new_nranges);
|
||||
new_nranges);
|
||||
|
||||
if (BE (new_array_start == NULL || new_array_end == NULL, 0))
|
||||
return REG_ESPACE;
|
||||
return REG_ESPACE;
|
||||
|
||||
mbcset->range_starts = new_array_start;
|
||||
mbcset->range_ends = new_array_end;
|
||||
*range_alloc = new_nranges;
|
||||
}
|
||||
|
||||
mbcset->range_starts[mbcset->nranges] = start_collseq;
|
||||
mbcset->range_ends[mbcset->nranges++] = end_collseq;
|
||||
mbcset->range_starts[mbcset->nranges] = start_collseq;
|
||||
mbcset->range_ends[mbcset->nranges++] = end_collseq;
|
||||
}
|
||||
|
||||
/* Build the table for single byte characters. */
|
||||
@@ -2922,25 +2979,22 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
|
||||
auto inline reg_errcode_t
|
||||
__attribute ((always_inline))
|
||||
build_collating_symbol (sbcset, mbcset, coll_sym_alloc, name)
|
||||
re_charset_t *mbcset;
|
||||
int *coll_sym_alloc;
|
||||
bitset_t sbcset;
|
||||
const unsigned char *name;
|
||||
build_collating_symbol (bitset_t sbcset, re_charset_t *mbcset,
|
||||
int *coll_sym_alloc, const unsigned char *name)
|
||||
{
|
||||
int32_t elem, idx;
|
||||
size_t name_len = strlen ((const char *) name);
|
||||
if (nrules != 0)
|
||||
{
|
||||
elem = seek_collating_symbol_entry (name, name_len);
|
||||
if (symb_table[2 * elem] != 0)
|
||||
if (elem != -1)
|
||||
{
|
||||
/* We found the entry. */
|
||||
idx = symb_table[2 * elem + 1];
|
||||
/* Skip the name of collating element name. */
|
||||
idx += 1 + extra[idx];
|
||||
}
|
||||
else if (symb_table[2 * elem] == 0 && name_len == 1)
|
||||
else if (name_len == 1)
|
||||
{
|
||||
/* No valid character, treat it as a normal
|
||||
character. */
|
||||
@@ -3020,6 +3074,10 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
if (BE (sbcset == NULL, 0))
|
||||
#endif /* RE_ENABLE_I18N */
|
||||
{
|
||||
re_free (sbcset);
|
||||
#ifdef RE_ENABLE_I18N
|
||||
re_free (mbcset);
|
||||
#endif
|
||||
*err = REG_ESPACE;
|
||||
return NULL;
|
||||
}
|
||||
@@ -3227,17 +3285,17 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
of having both SIMPLE_BRACKET and COMPLEX_BRACKET. */
|
||||
if (sbc_idx < BITSET_WORDS)
|
||||
{
|
||||
/* Build a tree for simple bracket. */
|
||||
br_token.type = SIMPLE_BRACKET;
|
||||
br_token.opr.sbcset = sbcset;
|
||||
work_tree = create_token_tree (dfa, NULL, NULL, &br_token);
|
||||
if (BE (work_tree == NULL, 0))
|
||||
goto parse_bracket_exp_espace;
|
||||
/* Build a tree for simple bracket. */
|
||||
br_token.type = SIMPLE_BRACKET;
|
||||
br_token.opr.sbcset = sbcset;
|
||||
work_tree = create_token_tree (dfa, NULL, NULL, &br_token);
|
||||
if (BE (work_tree == NULL, 0))
|
||||
goto parse_bracket_exp_espace;
|
||||
|
||||
/* Then join them by ALT node. */
|
||||
work_tree = create_tree (dfa, work_tree, mbc_tree, OP_ALT);
|
||||
if (BE (work_tree == NULL, 0))
|
||||
goto parse_bracket_exp_espace;
|
||||
/* Then join them by ALT node. */
|
||||
work_tree = create_tree (dfa, work_tree, mbc_tree, OP_ALT);
|
||||
if (BE (work_tree == NULL, 0))
|
||||
goto parse_bracket_exp_espace;
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -3256,7 +3314,7 @@ parse_bracket_exp (re_string_t *regexp, re_dfa_t *dfa, re_token_t *token,
|
||||
br_token.opr.sbcset = sbcset;
|
||||
work_tree = create_token_tree (dfa, NULL, NULL, &br_token);
|
||||
if (BE (work_tree == NULL, 0))
|
||||
goto parse_bracket_exp_espace;
|
||||
goto parse_bracket_exp_espace;
|
||||
}
|
||||
return work_tree;
|
||||
|
||||
@@ -3377,8 +3435,6 @@ build_equiv_class (bitset_t sbcset, const unsigned char *name)
|
||||
int32_t idx1, idx2;
|
||||
unsigned int ch;
|
||||
size_t len;
|
||||
/* This #include defines a local function! */
|
||||
# include <locale/weight.h>
|
||||
/* Calculate the index for equivalence class. */
|
||||
cp = name;
|
||||
table = (const int32_t *) _NL_CURRENT (LC_COLLATE, _NL_COLLATE_TABLEMB);
|
||||
@@ -3388,19 +3444,18 @@ build_equiv_class (bitset_t sbcset, const unsigned char *name)
|
||||
_NL_COLLATE_EXTRAMB);
|
||||
indirect = (const int32_t *) _NL_CURRENT (LC_COLLATE,
|
||||
_NL_COLLATE_INDIRECTMB);
|
||||
idx1 = findidx (&cp);
|
||||
if (BE (idx1 == 0 || cp < name + strlen ((const char *) name), 0))
|
||||
idx1 = findidx (table, indirect, extra, &cp, -1);
|
||||
if (BE (idx1 == 0 || *cp != '\0', 0))
|
||||
/* This isn't a valid character. */
|
||||
return REG_ECOLLATE;
|
||||
|
||||
/* Build single byte matcing table for this equivalence class. */
|
||||
char_buf[1] = (unsigned char) '\0';
|
||||
len = weights[idx1 & 0xffffff];
|
||||
for (ch = 0; ch < SBC_MAX; ++ch)
|
||||
{
|
||||
char_buf[0] = ch;
|
||||
cp = char_buf;
|
||||
idx2 = findidx (&cp);
|
||||
idx2 = findidx (table, indirect, extra, &cp, 1);
|
||||
/*
|
||||
idx2 = table[ch];
|
||||
*/
|
||||
@@ -3670,7 +3725,7 @@ fetch_number (re_string_t *input, re_token_t *token, reg_syntax_t syntax)
|
||||
}
|
||||
return num;
|
||||
}
|
||||
|
||||
|
||||
#ifdef RE_ENABLE_I18N
|
||||
static void
|
||||
free_charset (re_charset_t *cset)
|
||||
@@ -3686,7 +3741,7 @@ free_charset (re_charset_t *cset)
|
||||
re_free (cset);
|
||||
}
|
||||
#endif /* RE_ENABLE_I18N */
|
||||
|
||||
|
||||
/* Functions for binary tree operation. */
|
||||
|
||||
/* Create a tree node. */
|
||||
@@ -3809,7 +3864,7 @@ duplicate_tree (const bin_tree_t *root, re_dfa_t *dfa)
|
||||
node = node->parent;
|
||||
dup_node = dup_node->parent;
|
||||
if (!node)
|
||||
return dup_root;
|
||||
return dup_root;
|
||||
}
|
||||
node = node->right;
|
||||
p_new = &dup_node->right;
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/* Extended regular expression matching and search library.
|
||||
Copyright (C) 2002, 2003, 2005 Free Software Foundation, Inc.
|
||||
Copyright (C) 2002-2014 Free Software Foundation, Inc.
|
||||
This file is part of the GNU C Library.
|
||||
Contributed by Isamu Hasegawa <isamu@yamato.ibm.com>.
|
||||
|
||||
@@ -14,9 +14,8 @@
|
||||
Lesser General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Lesser General Public
|
||||
License along with the GNU C Library; if not, write to the Free
|
||||
Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA
|
||||
02111-1307 USA. */
|
||||
License along with the GNU C Library; if not, see
|
||||
<http://www.gnu.org/licenses/>. */
|
||||
|
||||
#ifdef HAVE_CONFIG_H
|
||||
#include "config.h"
|
||||
@@ -57,6 +56,9 @@
|
||||
#undefs RE_DUP_MAX and sets it to the right value. */
|
||||
#include <limits.h>
|
||||
|
||||
/* This header defines the MIN and MAX macros. */
|
||||
#include <sys/param.h>
|
||||
|
||||
#include <regex.h>
|
||||
#include "regex_internal.h"
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/* Extended regular expression matching and search library.
|
||||
Copyright (C) 2002, 2003, 2004, 2005, 2006 Free Software Foundation, Inc.
|
||||
Copyright (C) 2002-2014 Free Software Foundation, Inc.
|
||||
This file is part of the GNU C Library.
|
||||
Contributed by Isamu Hasegawa <isamu@yamato.ibm.com>.
|
||||
|
||||
@@ -14,9 +14,8 @@
|
||||
Lesser General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Lesser General Public
|
||||
License along with the GNU C Library; if not, write to the Free
|
||||
Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA
|
||||
02111-1307 USA. */
|
||||
License along with the GNU C Library; if not, see
|
||||
<http://www.gnu.org/licenses/>. */
|
||||
|
||||
static void re_string_construct_common (const char *str, int len,
|
||||
re_string_t *pstr,
|
||||
@@ -36,7 +35,7 @@ static re_dfastate_t *create_cd_newstate (const re_dfa_t *dfa,
|
||||
re_string_reconstruct before using the object. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_string_allocate (re_string_t *pstr, const char *str, int len, int init_len,
|
||||
RE_TRANSLATE_TYPE trans, int icase, const re_dfa_t *dfa)
|
||||
{
|
||||
@@ -64,7 +63,7 @@ re_string_allocate (re_string_t *pstr, const char *str, int len, int init_len,
|
||||
/* This function allocate the buffers, and initialize them. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_string_construct (re_string_t *pstr, const char *str, int len,
|
||||
RE_TRANSLATE_TYPE trans, int icase, const re_dfa_t *dfa)
|
||||
{
|
||||
@@ -127,13 +126,20 @@ re_string_construct (re_string_t *pstr, const char *str, int len,
|
||||
/* Helper functions for re_string_allocate, and re_string_construct. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_string_realloc_buffers (re_string_t *pstr, int new_buf_len)
|
||||
{
|
||||
#ifdef RE_ENABLE_I18N
|
||||
if (pstr->mb_cur_max > 1)
|
||||
{
|
||||
wint_t *new_wcs = re_realloc (pstr->wcs, wint_t, new_buf_len);
|
||||
wint_t *new_wcs;
|
||||
|
||||
/* Avoid overflow in realloc. */
|
||||
const size_t max_object_size = MAX (sizeof (wint_t), sizeof (int));
|
||||
if (BE (SIZE_MAX / max_object_size < new_buf_len, 0))
|
||||
return REG_ESPACE;
|
||||
|
||||
new_wcs = re_realloc (pstr->wcs, wint_t, new_buf_len);
|
||||
if (BE (new_wcs == NULL, 0))
|
||||
return REG_ESPACE;
|
||||
pstr->wcs = new_wcs;
|
||||
@@ -230,13 +236,8 @@ build_wcs_buffer (re_string_t *pstr)
|
||||
else
|
||||
p = (const char *) pstr->raw_mbs + pstr->raw_mbs_idx + byte_idx;
|
||||
mbclen = __mbrtowc (&wc, p, remain_len, &pstr->cur_state);
|
||||
if (BE (mbclen == (size_t) -2, 0))
|
||||
{
|
||||
/* The buffer doesn't have enough space, finish to build. */
|
||||
pstr->cur_state = prev_st;
|
||||
break;
|
||||
}
|
||||
else if (BE (mbclen == (size_t) -1 || mbclen == 0, 0))
|
||||
if (BE (mbclen == (size_t) -1 || mbclen == 0
|
||||
|| (mbclen == (size_t) -2 && pstr->bufs_len >= pstr->len), 0))
|
||||
{
|
||||
/* We treat these cases as a singlebyte character. */
|
||||
mbclen = 1;
|
||||
@@ -245,6 +246,12 @@ build_wcs_buffer (re_string_t *pstr)
|
||||
wc = pstr->trans[wc];
|
||||
pstr->cur_state = prev_st;
|
||||
}
|
||||
else if (BE (mbclen == (size_t) -2, 0))
|
||||
{
|
||||
/* The buffer doesn't have enough space, finish to build. */
|
||||
pstr->cur_state = prev_st;
|
||||
break;
|
||||
}
|
||||
|
||||
/* Write wide character and padding. */
|
||||
pstr->wcs[byte_idx++] = wc;
|
||||
@@ -260,7 +267,7 @@ build_wcs_buffer (re_string_t *pstr)
|
||||
but for REG_ICASE. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
build_wcs_upper_buffer (re_string_t *pstr)
|
||||
{
|
||||
mbstate_t prev_st;
|
||||
@@ -327,9 +334,11 @@ build_wcs_upper_buffer (re_string_t *pstr)
|
||||
for (remain_len = byte_idx + mbclen - 1; byte_idx < remain_len ;)
|
||||
pstr->wcs[byte_idx++] = WEOF;
|
||||
}
|
||||
else if (mbclen == (size_t) -1 || mbclen == 0)
|
||||
else if (mbclen == (size_t) -1 || mbclen == 0
|
||||
|| (mbclen == (size_t) -2 && pstr->bufs_len >= pstr->len))
|
||||
{
|
||||
/* It is an invalid character or '\0'. Just use the byte. */
|
||||
/* It is an invalid character, an incomplete character
|
||||
at the end of the string, or '\0'. Just use the byte. */
|
||||
int ch = pstr->raw_mbs[pstr->raw_mbs_idx + byte_idx];
|
||||
pstr->mbs[byte_idx] = ch;
|
||||
/* And also cast it to wide char. */
|
||||
@@ -423,8 +432,8 @@ build_wcs_upper_buffer (re_string_t *pstr)
|
||||
src_idx += mbclen;
|
||||
continue;
|
||||
}
|
||||
else
|
||||
memcpy (pstr->mbs + byte_idx, p, mbclen);
|
||||
else
|
||||
memcpy (pstr->mbs + byte_idx, p, mbclen);
|
||||
}
|
||||
else
|
||||
memcpy (pstr->mbs + byte_idx, p, mbclen);
|
||||
@@ -442,7 +451,8 @@ build_wcs_upper_buffer (re_string_t *pstr)
|
||||
for (remain_len = byte_idx + mbclen - 1; byte_idx < remain_len ;)
|
||||
pstr->wcs[byte_idx++] = WEOF;
|
||||
}
|
||||
else if (mbclen == (size_t) -1 || mbclen == 0)
|
||||
else if (mbclen == (size_t) -1 || mbclen == 0
|
||||
|| (mbclen == (size_t) -2 && pstr->bufs_len >= pstr->len))
|
||||
{
|
||||
/* It is an invalid character or '\0'. Just use the byte. */
|
||||
int ch = pstr->raw_mbs[pstr->raw_mbs_idx + src_idx];
|
||||
@@ -482,18 +492,18 @@ re_string_skip_chars (re_string_t *pstr, int new_raw_idx, wint_t *last_wc)
|
||||
mbstate_t prev_st;
|
||||
int rawbuf_idx;
|
||||
size_t mbclen;
|
||||
wchar_t wc = WEOF;
|
||||
wint_t wc = WEOF;
|
||||
|
||||
/* Skip the characters which are not necessary to check. */
|
||||
for (rawbuf_idx = pstr->raw_mbs_idx + pstr->valid_raw_len;
|
||||
rawbuf_idx < new_raw_idx;)
|
||||
{
|
||||
int remain_len;
|
||||
remain_len = pstr->len - rawbuf_idx;
|
||||
wchar_t wc2;
|
||||
int remain_len = pstr->raw_len - rawbuf_idx;
|
||||
prev_st = pstr->cur_state;
|
||||
mbclen = __mbrtowc (&wc, (const char *) pstr->raw_mbs + rawbuf_idx,
|
||||
mbclen = __mbrtowc (&wc2, (const char *) pstr->raw_mbs + rawbuf_idx,
|
||||
remain_len, &pstr->cur_state);
|
||||
if (BE (mbclen == (size_t) -2 || mbclen == (size_t) -1 || mbclen == 0, 0))
|
||||
if (BE ((ssize_t) mbclen <= 0, 0))
|
||||
{
|
||||
/* We treat these cases as a single byte character. */
|
||||
if (mbclen == 0 || remain_len == 0)
|
||||
@@ -503,10 +513,12 @@ re_string_skip_chars (re_string_t *pstr, int new_raw_idx, wint_t *last_wc)
|
||||
mbclen = 1;
|
||||
pstr->cur_state = prev_st;
|
||||
}
|
||||
else
|
||||
wc = (wint_t) wc2;
|
||||
/* Then proceed the next character. */
|
||||
rawbuf_idx += mbclen;
|
||||
}
|
||||
*last_wc = (wint_t) wc;
|
||||
*last_wc = wc;
|
||||
return rawbuf_idx;
|
||||
}
|
||||
#endif /* RE_ENABLE_I18N */
|
||||
@@ -559,7 +571,7 @@ re_string_translate_buffer (re_string_t *pstr)
|
||||
convert to upper case in case of REG_ICASE, apply translation. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_string_reconstruct (re_string_t *pstr, int idx, int eflags)
|
||||
{
|
||||
int offset = idx - pstr->raw_mbs_idx;
|
||||
@@ -667,7 +679,7 @@ re_string_reconstruct (re_string_t *pstr, int idx, int eflags)
|
||||
pstr->valid_len - offset);
|
||||
pstr->valid_len -= offset;
|
||||
pstr->valid_raw_len -= offset;
|
||||
#if DEBUG
|
||||
#if defined DEBUG && DEBUG
|
||||
assert (pstr->valid_len > 0);
|
||||
#endif
|
||||
}
|
||||
@@ -694,7 +706,7 @@ re_string_reconstruct (re_string_t *pstr, int idx, int eflags)
|
||||
|
||||
if (pstr->is_utf8)
|
||||
{
|
||||
const unsigned char *raw, *p, *q, *end;
|
||||
const unsigned char *raw, *p, *end;
|
||||
|
||||
/* Special case UTF-8. Multi-byte chars start with any
|
||||
byte other than 0x80 - 0xbf. */
|
||||
@@ -723,18 +735,18 @@ re_string_reconstruct (re_string_t *pstr, int idx, int eflags)
|
||||
unsigned char buf[6];
|
||||
size_t mbclen;
|
||||
|
||||
q = p;
|
||||
const unsigned char *pp = p;
|
||||
if (BE (pstr->trans != NULL, 0))
|
||||
{
|
||||
int i = mlen < 6 ? mlen : 6;
|
||||
while (--i >= 0)
|
||||
buf[i] = pstr->trans[p[i]];
|
||||
q = buf;
|
||||
pp = buf;
|
||||
}
|
||||
/* XXX Don't use mbrtowc, we know which conversion
|
||||
to use (UTF-8 -> UCS4). */
|
||||
memset (&cur_state, 0, sizeof (cur_state));
|
||||
mbclen = __mbrtowc (&wc2, (const char *) p, mlen,
|
||||
mbclen = __mbrtowc (&wc2, (const char *) pp, mlen,
|
||||
&cur_state);
|
||||
if (raw + offset - p <= mbclen
|
||||
&& mbclen < (size_t) -2)
|
||||
@@ -855,7 +867,7 @@ re_string_peek_byte_case (const re_string_t *pstr, int idx)
|
||||
}
|
||||
|
||||
static unsigned char
|
||||
internal_function __attribute ((pure))
|
||||
internal_function
|
||||
re_string_fetch_byte_case (re_string_t *pstr)
|
||||
{
|
||||
if (BE (!pstr->mbs_allocated, 1))
|
||||
@@ -924,7 +936,7 @@ re_string_context_at (const re_string_t *input, int idx, int eflags)
|
||||
int wc_idx = idx;
|
||||
while(input->wcs[wc_idx] == WEOF)
|
||||
{
|
||||
#ifdef DEBUG
|
||||
#if defined DEBUG && DEBUG
|
||||
/* It must not happen. */
|
||||
assert (wc_idx >= 0);
|
||||
#endif
|
||||
@@ -951,7 +963,7 @@ re_string_context_at (const re_string_t *input, int idx, int eflags)
|
||||
/* Functions for set operation. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_alloc (re_node_set *set, int size)
|
||||
{
|
||||
set->alloc = size;
|
||||
@@ -963,7 +975,7 @@ re_node_set_alloc (re_node_set *set, int size)
|
||||
}
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_init_1 (re_node_set *set, int elem)
|
||||
{
|
||||
set->alloc = 1;
|
||||
@@ -979,7 +991,7 @@ re_node_set_init_1 (re_node_set *set, int elem)
|
||||
}
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_init_2 (re_node_set *set, int elem1, int elem2)
|
||||
{
|
||||
set->alloc = 2;
|
||||
@@ -1009,7 +1021,7 @@ re_node_set_init_2 (re_node_set *set, int elem1, int elem2)
|
||||
}
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_init_copy (re_node_set *dest, const re_node_set *src)
|
||||
{
|
||||
dest->nelem = src->nelem;
|
||||
@@ -1034,7 +1046,7 @@ re_node_set_init_copy (re_node_set *dest, const re_node_set *src)
|
||||
Note: We assume dest->elems is NULL, when dest->alloc is 0. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_add_intersect (re_node_set *dest, const re_node_set *src1,
|
||||
const re_node_set *src2)
|
||||
{
|
||||
@@ -1049,7 +1061,7 @@ re_node_set_add_intersect (re_node_set *dest, const re_node_set *src1,
|
||||
int new_alloc = src1->nelem + src2->nelem + dest->alloc;
|
||||
int *new_elems = re_realloc (dest->elems, int, new_alloc);
|
||||
if (BE (new_elems == NULL, 0))
|
||||
return REG_ESPACE;
|
||||
return REG_ESPACE;
|
||||
dest->elems = new_elems;
|
||||
dest->alloc = new_alloc;
|
||||
}
|
||||
@@ -1068,8 +1080,8 @@ re_node_set_add_intersect (re_node_set *dest, const re_node_set *src1,
|
||||
while (id >= 0 && dest->elems[id] > src1->elems[i1])
|
||||
--id;
|
||||
|
||||
if (id < 0 || dest->elems[id] != src1->elems[i1])
|
||||
dest->elems[--sbase] = src1->elems[i1];
|
||||
if (id < 0 || dest->elems[id] != src1->elems[i1])
|
||||
dest->elems[--sbase] = src1->elems[i1];
|
||||
|
||||
if (--i1 < 0 || --i2 < 0)
|
||||
break;
|
||||
@@ -1099,20 +1111,20 @@ re_node_set_add_intersect (re_node_set *dest, const re_node_set *src1,
|
||||
if (delta > 0 && id >= 0)
|
||||
for (;;)
|
||||
{
|
||||
if (dest->elems[is] > dest->elems[id])
|
||||
{
|
||||
/* Copy from the top. */
|
||||
dest->elems[id + delta--] = dest->elems[is--];
|
||||
if (delta == 0)
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
/* Slide from the bottom. */
|
||||
dest->elems[id + delta] = dest->elems[id];
|
||||
if (--id < 0)
|
||||
break;
|
||||
}
|
||||
if (dest->elems[is] > dest->elems[id])
|
||||
{
|
||||
/* Copy from the top. */
|
||||
dest->elems[id + delta--] = dest->elems[is--];
|
||||
if (delta == 0)
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
/* Slide from the bottom. */
|
||||
dest->elems[id + delta] = dest->elems[id];
|
||||
if (--id < 0)
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
/* Copy remaining SRC elements. */
|
||||
@@ -1125,7 +1137,7 @@ re_node_set_add_intersect (re_node_set *dest, const re_node_set *src1,
|
||||
DEST. Return value indicate the error code or REG_NOERROR if succeeded. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_init_union (re_node_set *dest, const re_node_set *src1,
|
||||
const re_node_set *src2)
|
||||
{
|
||||
@@ -1178,7 +1190,7 @@ re_node_set_init_union (re_node_set *dest, const re_node_set *src1,
|
||||
DEST. Return value indicate the error code or REG_NOERROR if succeeded. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_merge (re_node_set *dest, const re_node_set *src)
|
||||
{
|
||||
int is, id, sbase, delta;
|
||||
@@ -1207,11 +1219,11 @@ re_node_set_merge (re_node_set *dest, const re_node_set *src)
|
||||
is = src->nelem - 1, id = dest->nelem - 1; is >= 0 && id >= 0; )
|
||||
{
|
||||
if (dest->elems[id] == src->elems[is])
|
||||
is--, id--;
|
||||
is--, id--;
|
||||
else if (dest->elems[id] < src->elems[is])
|
||||
dest->elems[--sbase] = src->elems[is--];
|
||||
dest->elems[--sbase] = src->elems[is--];
|
||||
else /* if (dest->elems[id] > src->elems[is]) */
|
||||
--id;
|
||||
--id;
|
||||
}
|
||||
|
||||
if (is >= 0)
|
||||
@@ -1233,21 +1245,21 @@ re_node_set_merge (re_node_set *dest, const re_node_set *src)
|
||||
for (;;)
|
||||
{
|
||||
if (dest->elems[is] > dest->elems[id])
|
||||
{
|
||||
{
|
||||
/* Copy from the top. */
|
||||
dest->elems[id + delta--] = dest->elems[is--];
|
||||
dest->elems[id + delta--] = dest->elems[is--];
|
||||
if (delta == 0)
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
/* Slide from the bottom. */
|
||||
dest->elems[id + delta] = dest->elems[id];
|
||||
{
|
||||
/* Slide from the bottom. */
|
||||
dest->elems[id + delta] = dest->elems[id];
|
||||
if (--id < 0)
|
||||
{
|
||||
/* Copy remaining SRC elements. */
|
||||
memcpy (dest->elems, dest->elems + sbase,
|
||||
delta * sizeof (int));
|
||||
delta * sizeof (int));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -1261,7 +1273,7 @@ re_node_set_merge (re_node_set *dest, const re_node_set *src)
|
||||
return -1 if an error is occured, return 1 otherwise. */
|
||||
|
||||
static int
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_insert (re_node_set *set, int elem)
|
||||
{
|
||||
int idx;
|
||||
@@ -1299,12 +1311,12 @@ re_node_set_insert (re_node_set *set, int elem)
|
||||
{
|
||||
idx = 0;
|
||||
for (idx = set->nelem; idx > 0; idx--)
|
||||
set->elems[idx] = set->elems[idx - 1];
|
||||
set->elems[idx] = set->elems[idx - 1];
|
||||
}
|
||||
else
|
||||
{
|
||||
for (idx = set->nelem; set->elems[idx - 1] > elem; idx--)
|
||||
set->elems[idx] = set->elems[idx - 1];
|
||||
set->elems[idx] = set->elems[idx - 1];
|
||||
}
|
||||
|
||||
/* Insert the new element. */
|
||||
@@ -1318,7 +1330,7 @@ re_node_set_insert (re_node_set *set, int elem)
|
||||
Return -1 if an error is occured, return 1 otherwise. */
|
||||
|
||||
static int
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_node_set_insert_last (re_node_set *set, int elem)
|
||||
{
|
||||
/* Realloc if we need. */
|
||||
@@ -1404,8 +1416,11 @@ re_dfa_add_node (re_dfa_t *dfa, re_token_t token)
|
||||
re_node_set *new_edests, *new_eclosures;
|
||||
re_token_t *new_nodes;
|
||||
|
||||
/* Avoid overflows. */
|
||||
if (BE (new_nodes_alloc < dfa->nodes_alloc, 0))
|
||||
/* Avoid overflows in realloc. */
|
||||
const size_t max_object_size = MAX (sizeof (re_token_t),
|
||||
MAX (sizeof (re_node_set),
|
||||
sizeof (int)));
|
||||
if (BE (SIZE_MAX / max_object_size < new_nodes_alloc, 0))
|
||||
return -1;
|
||||
|
||||
new_nodes = re_realloc (dfa->nodes, re_token_t, new_nodes_alloc);
|
||||
@@ -1458,7 +1473,7 @@ calc_state_hash (const re_node_set *nodes, unsigned int context)
|
||||
optimization. */
|
||||
|
||||
static re_dfastate_t *
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_acquire_state (reg_errcode_t *err, const re_dfa_t *dfa,
|
||||
const re_node_set *nodes)
|
||||
{
|
||||
@@ -1502,7 +1517,7 @@ re_acquire_state (reg_errcode_t *err, const re_dfa_t *dfa,
|
||||
optimization. */
|
||||
|
||||
static re_dfastate_t *
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
re_acquire_state_context (reg_errcode_t *err, const re_dfa_t *dfa,
|
||||
const re_node_set *nodes, unsigned int context)
|
||||
{
|
||||
@@ -1539,6 +1554,7 @@ re_acquire_state_context (reg_errcode_t *err, const re_dfa_t *dfa,
|
||||
indicates the error code if failed. */
|
||||
|
||||
static reg_errcode_t
|
||||
__attribute_warn_unused_result__
|
||||
register_state (const re_dfa_t *dfa, re_dfastate_t *newstate,
|
||||
unsigned int hash)
|
||||
{
|
||||
@@ -1554,7 +1570,8 @@ register_state (const re_dfa_t *dfa, re_dfastate_t *newstate,
|
||||
{
|
||||
int elem = newstate->nodes.elems[i];
|
||||
if (!IS_EPSILON_NODE (dfa->nodes[elem].type))
|
||||
re_node_set_insert_last (&newstate->non_eps_nodes, elem);
|
||||
if (re_node_set_insert_last (&newstate->non_eps_nodes, elem) < 0)
|
||||
return REG_ESPACE;
|
||||
}
|
||||
|
||||
spot = dfa->state_table + (hash & dfa->state_hash_mask);
|
||||
@@ -1592,7 +1609,7 @@ free_state (re_dfastate_t *state)
|
||||
Return the new state if succeeded, otherwise return NULL. */
|
||||
|
||||
static re_dfastate_t *
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
create_ci_newstate (const re_dfa_t *dfa, const re_node_set *nodes,
|
||||
unsigned int hash)
|
||||
{
|
||||
@@ -1642,7 +1659,7 @@ create_ci_newstate (const re_dfa_t *dfa, const re_node_set *nodes,
|
||||
Return the new state if succeeded, otherwise return NULL. */
|
||||
|
||||
static re_dfastate_t *
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
create_cd_newstate (const re_dfa_t *dfa, const re_node_set *nodes,
|
||||
unsigned int context, unsigned int hash)
|
||||
{
|
||||
@@ -1691,7 +1708,9 @@ create_cd_newstate (const re_dfa_t *dfa, const re_node_set *nodes,
|
||||
free_state (newstate);
|
||||
return NULL;
|
||||
}
|
||||
re_node_set_init_copy (newstate->entrance_nodes, nodes);
|
||||
if (re_node_set_init_copy (newstate->entrance_nodes, nodes)
|
||||
!= REG_NOERROR)
|
||||
return NULL;
|
||||
nctx_nodes = 0;
|
||||
newstate->has_constraint = 1;
|
||||
}
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/* Extended regular expression matching and search library.
|
||||
Copyright (C) 2002-2005, 2007, 2008 Free Software Foundation, Inc.
|
||||
Copyright (C) 2002-2014 Free Software Foundation, Inc.
|
||||
This file is part of the GNU C Library.
|
||||
Contributed by Isamu Hasegawa <isamu@yamato.ibm.com>.
|
||||
|
||||
@@ -14,9 +14,8 @@
|
||||
Lesser General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Lesser General Public
|
||||
License along with the GNU C Library; if not, write to the Free
|
||||
Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA
|
||||
02111-1307 USA. */
|
||||
License along with the GNU C Library; if not, see
|
||||
<http://www.gnu.org/licenses/>. */
|
||||
|
||||
#ifndef _REGEX_INTERNAL_H
|
||||
#define _REGEX_INTERNAL_H 1
|
||||
@@ -74,7 +73,7 @@
|
||||
# ifdef _LIBC
|
||||
# undef gettext
|
||||
# define gettext(msgid) \
|
||||
INTUSE(__dcgettext) (_libc_intl_domainname, msgid, LC_MESSAGES)
|
||||
__dcgettext (_libc_intl_domainname, msgid, LC_MESSAGES)
|
||||
# endif
|
||||
#else
|
||||
# define gettext(msgid) (msgid)
|
||||
@@ -91,7 +90,7 @@
|
||||
# define SIZE_MAX ((size_t) -1)
|
||||
#endif
|
||||
|
||||
#if (defined MB_CUR_MAX && HAVE_LOCALE_H && HAVE_WCTYPE_H && HAVE_WCHAR_H && HAVE_WCRTOMB && HAVE_MBRTOWC && HAVE_WCSCOLL) || _LIBC
|
||||
#if (defined MB_CUR_MAX && HAVE_WCTYPE_H && HAVE_ISWCTYPE) || _LIBC
|
||||
# define RE_ENABLE_I18N
|
||||
#endif
|
||||
|
||||
@@ -99,7 +98,6 @@
|
||||
# define BE(expr, val) __builtin_expect (expr, val)
|
||||
#else
|
||||
# define BE(expr, val) (expr)
|
||||
# define inline
|
||||
#endif
|
||||
|
||||
/* Number of single byte character. */
|
||||
@@ -123,10 +121,8 @@
|
||||
# define attribute_hidden
|
||||
#endif /* not _LIBC */
|
||||
|
||||
#ifdef __GNUC__
|
||||
# define __attribute(arg) __attribute__ (arg)
|
||||
#else
|
||||
# define __attribute(arg)
|
||||
#if __GNUC__ < 3 + (__GNUC_MINOR__ < 1)
|
||||
# define __attribute__(arg)
|
||||
#endif
|
||||
|
||||
extern const char __re_error_msgid[] attribute_hidden;
|
||||
@@ -380,10 +376,11 @@ typedef struct re_dfa_t re_dfa_t;
|
||||
|
||||
#ifndef _LIBC
|
||||
# ifdef __i386__
|
||||
# define internal_function __attribute ((regparm (3), stdcall))
|
||||
# define internal_function __attribute__ ((regparm (3), stdcall))
|
||||
# else
|
||||
# define internal_function
|
||||
# endif
|
||||
# define __attribute_warn_unused_result__
|
||||
#endif
|
||||
|
||||
#ifndef NOT_IN_libc
|
||||
@@ -399,7 +396,7 @@ static void build_upper_buffer (re_string_t *pstr) internal_function;
|
||||
static void re_string_translate_buffer (re_string_t *pstr) internal_function;
|
||||
static unsigned int re_string_context_at (const re_string_t *input, int idx,
|
||||
int eflags)
|
||||
internal_function __attribute ((pure));
|
||||
internal_function __attribute__ ((pure));
|
||||
#endif
|
||||
#define re_string_peek_byte(pstr, offset) \
|
||||
((pstr)->mbs[(pstr)->cur_idx + offset])
|
||||
@@ -664,7 +661,7 @@ struct re_dfa_t
|
||||
(re_node_set_remove_at (set, re_node_set_contains (set, id) - 1))
|
||||
#define re_node_set_empty(p) ((p)->nelem = 0)
|
||||
#define re_node_set_free(set) re_free ((set)->elems)
|
||||
|
||||
|
||||
|
||||
typedef enum
|
||||
{
|
||||
@@ -688,7 +685,7 @@ typedef struct
|
||||
|
||||
|
||||
/* Inline functions for bitset operation. */
|
||||
static inline void
|
||||
static void __attribute__ ((unused))
|
||||
bitset_not (bitset_t set)
|
||||
{
|
||||
int bitset_i;
|
||||
@@ -696,7 +693,7 @@ bitset_not (bitset_t set)
|
||||
set[bitset_i] = ~set[bitset_i];
|
||||
}
|
||||
|
||||
static inline void
|
||||
static void __attribute__ ((unused))
|
||||
bitset_merge (bitset_t dest, const bitset_t src)
|
||||
{
|
||||
int bitset_i;
|
||||
@@ -704,7 +701,7 @@ bitset_merge (bitset_t dest, const bitset_t src)
|
||||
dest[bitset_i] |= src[bitset_i];
|
||||
}
|
||||
|
||||
static inline void
|
||||
static void __attribute__ ((unused))
|
||||
bitset_mask (bitset_t dest, const bitset_t src)
|
||||
{
|
||||
int bitset_i;
|
||||
@@ -714,8 +711,8 @@ bitset_mask (bitset_t dest, const bitset_t src)
|
||||
|
||||
#ifdef RE_ENABLE_I18N
|
||||
/* Inline functions for re_string. */
|
||||
static inline int
|
||||
internal_function __attribute ((pure))
|
||||
static int
|
||||
internal_function __attribute__ ((pure, unused))
|
||||
re_string_char_size_at (const re_string_t *pstr, int idx)
|
||||
{
|
||||
int byte_idx;
|
||||
@@ -727,8 +724,8 @@ re_string_char_size_at (const re_string_t *pstr, int idx)
|
||||
return byte_idx;
|
||||
}
|
||||
|
||||
static inline wint_t
|
||||
internal_function __attribute ((pure))
|
||||
static wint_t
|
||||
internal_function __attribute__ ((pure, unused))
|
||||
re_string_wchar_at (const re_string_t *pstr, int idx)
|
||||
{
|
||||
if (pstr->mb_cur_max == 1)
|
||||
@@ -737,15 +734,17 @@ re_string_wchar_at (const re_string_t *pstr, int idx)
|
||||
}
|
||||
|
||||
# ifndef NOT_IN_libc
|
||||
# ifdef _LIBC
|
||||
# include <locale/weight.h>
|
||||
# endif
|
||||
|
||||
static int
|
||||
internal_function __attribute ((pure))
|
||||
internal_function __attribute__ ((pure, unused))
|
||||
re_string_elem_size_at (const re_string_t *pstr, int idx)
|
||||
{
|
||||
# ifdef _LIBC
|
||||
const unsigned char *p, *extra;
|
||||
const int32_t *table, *indirect;
|
||||
int32_t tmp;
|
||||
# include <locale/weight.h>
|
||||
uint_fast32_t nrules = _NL_CURRENT_WORD (LC_COLLATE, _NL_COLLATE_NRULES);
|
||||
|
||||
if (nrules != 0)
|
||||
@@ -756,7 +755,7 @@ re_string_elem_size_at (const re_string_t *pstr, int idx)
|
||||
indirect = (const int32_t *) _NL_CURRENT (LC_COLLATE,
|
||||
_NL_COLLATE_INDIRECTMB);
|
||||
p = pstr->mbs + idx;
|
||||
tmp = findidx (&p);
|
||||
findidx (table, indirect, extra, &p, pstr->len - idx);
|
||||
return p - pstr->mbs - idx;
|
||||
}
|
||||
else
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/* Extended regular expression matching and search library.
|
||||
Copyright (C) 2002, 2003, 2004, 2005, 2007, 2009 Free Software Foundation, Inc.
|
||||
Copyright (C) 2002-2014 Free Software Foundation, Inc.
|
||||
This file is part of the GNU C Library.
|
||||
Contributed by Isamu Hasegawa <isamu@yamato.ibm.com>.
|
||||
|
||||
@@ -14,9 +14,10 @@
|
||||
Lesser General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Lesser General Public
|
||||
License along with the GNU C Library; if not, write to the Free
|
||||
Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA
|
||||
02111-1307 USA. */
|
||||
License along with the GNU C Library; if not, see
|
||||
<http://www.gnu.org/licenses/>. */
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
static reg_errcode_t match_ctx_init (re_match_context_t *cache, int eflags,
|
||||
int n) internal_function;
|
||||
@@ -198,7 +199,7 @@ static int group_nodes_into_DFAstates (const re_dfa_t *dfa,
|
||||
static int check_node_accept (const re_match_context_t *mctx,
|
||||
const re_token_t *node, int idx)
|
||||
internal_function;
|
||||
static reg_errcode_t extend_buffers (re_match_context_t *mctx)
|
||||
static reg_errcode_t extend_buffers (re_match_context_t *mctx, int min_len)
|
||||
internal_function;
|
||||
|
||||
/* Entry point for POSIX code. */
|
||||
@@ -368,16 +369,16 @@ re_search_2_stub (bufp, string1, length1, string2, length2, start, range, regs,
|
||||
const char *str;
|
||||
int rval;
|
||||
int len = length1 + length2;
|
||||
int free_str = 0;
|
||||
char *s = NULL;
|
||||
|
||||
if (BE (length1 < 0 || length2 < 0 || stop < 0, 0))
|
||||
if (BE (length1 < 0 || length2 < 0 || stop < 0 || len < length1, 0))
|
||||
return -2;
|
||||
|
||||
/* Concatenate the strings. */
|
||||
if (length2 > 0)
|
||||
if (length1 > 0)
|
||||
{
|
||||
char *s = re_malloc (char, len);
|
||||
s = re_malloc (char, len);
|
||||
|
||||
if (BE (s == NULL, 0))
|
||||
return -2;
|
||||
@@ -388,17 +389,14 @@ re_search_2_stub (bufp, string1, length1, string2, length2, start, range, regs,
|
||||
memcpy (s + length1, string2, length2);
|
||||
#endif
|
||||
str = s;
|
||||
free_str = 1;
|
||||
}
|
||||
else
|
||||
str = string2;
|
||||
else
|
||||
str = string1;
|
||||
|
||||
rval = re_search_stub (bufp, str, len, start, range, stop, regs,
|
||||
ret_len);
|
||||
if (free_str)
|
||||
re_free ((char *) str);
|
||||
rval = re_search_stub (bufp, str, len, start, range, stop, regs, ret_len);
|
||||
re_free (s);
|
||||
return rval;
|
||||
}
|
||||
|
||||
@@ -512,9 +510,14 @@ re_copy_regs (regs, pmatch, nregs, regs_allocated)
|
||||
if (regs_allocated == REGS_UNALLOCATED)
|
||||
{ /* No. So allocate them with malloc. */
|
||||
regs->start = re_malloc (regoff_t, need_regs);
|
||||
regs->end = re_malloc (regoff_t, need_regs);
|
||||
if (BE (regs->start == NULL, 0) || BE (regs->end == NULL, 0))
|
||||
if (BE (regs->start == NULL, 0))
|
||||
return REGS_UNALLOCATED;
|
||||
regs->end = re_malloc (regoff_t, need_regs);
|
||||
if (BE (regs->end == NULL, 0))
|
||||
{
|
||||
re_free (regs->start);
|
||||
return REGS_UNALLOCATED;
|
||||
}
|
||||
regs->num_regs = need_regs;
|
||||
}
|
||||
else if (regs_allocated == REGS_REALLOCATE)
|
||||
@@ -524,9 +527,15 @@ re_copy_regs (regs, pmatch, nregs, regs_allocated)
|
||||
if (BE (need_regs > regs->num_regs, 0))
|
||||
{
|
||||
regoff_t *new_start = re_realloc (regs->start, regoff_t, need_regs);
|
||||
regoff_t *new_end = re_realloc (regs->end, regoff_t, need_regs);
|
||||
if (BE (new_start == NULL, 0) || BE (new_end == NULL, 0))
|
||||
regoff_t *new_end;
|
||||
if (BE (new_start == NULL, 0))
|
||||
return REGS_UNALLOCATED;
|
||||
new_end = re_realloc (regs->end, regoff_t, need_regs);
|
||||
if (BE (new_end == NULL, 0))
|
||||
{
|
||||
re_free (new_start);
|
||||
return REGS_UNALLOCATED;
|
||||
}
|
||||
regs->start = new_start;
|
||||
regs->end = new_end;
|
||||
regs->num_regs = need_regs;
|
||||
@@ -617,6 +626,7 @@ re_exec (s)
|
||||
(START + RANGE >= 0 && START + RANGE <= LENGTH) */
|
||||
|
||||
static reg_errcode_t
|
||||
__attribute_warn_unused_result__
|
||||
re_search_internal (preg, string, length, start, range, stop, nmatch, pmatch,
|
||||
eflags)
|
||||
const regex_t *preg;
|
||||
@@ -668,7 +678,7 @@ re_search_internal (preg, string, length, start, range, stop, nmatch, pmatch,
|
||||
|| !preg->newline_anchor))
|
||||
{
|
||||
if (start != 0 && start + range != 0)
|
||||
return REG_NOMATCH;
|
||||
return REG_NOMATCH;
|
||||
start = range = 0;
|
||||
}
|
||||
|
||||
@@ -693,6 +703,13 @@ re_search_internal (preg, string, length, start, range, stop, nmatch, pmatch,
|
||||
multi character collating element. */
|
||||
if (nmatch > 1 || dfa->has_mb_node)
|
||||
{
|
||||
/* Avoid overflow. */
|
||||
if (BE (SIZE_MAX / sizeof (re_dfastate_t *) <= mctx.input.bufs_len, 0))
|
||||
{
|
||||
err = REG_ESPACE;
|
||||
goto free_return;
|
||||
}
|
||||
|
||||
mctx.state_log = re_malloc (re_dfastate_t *, mctx.input.bufs_len + 1);
|
||||
if (BE (mctx.state_log == NULL, 0))
|
||||
{
|
||||
@@ -800,10 +817,10 @@ re_search_internal (preg, string, length, start, range, stop, nmatch, pmatch,
|
||||
break;
|
||||
match_first += incr;
|
||||
if (match_first < left_lim || match_first > right_lim)
|
||||
{
|
||||
err = REG_NOMATCH;
|
||||
goto free_return;
|
||||
}
|
||||
{
|
||||
err = REG_NOMATCH;
|
||||
goto free_return;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -917,14 +934,14 @@ re_search_internal (preg, string, length, start, range, stop, nmatch, pmatch,
|
||||
}
|
||||
|
||||
if (dfa->subexp_map)
|
||||
for (reg_idx = 0; reg_idx + 1 < nmatch; reg_idx++)
|
||||
if (dfa->subexp_map[reg_idx] != reg_idx)
|
||||
{
|
||||
pmatch[reg_idx + 1].rm_so
|
||||
= pmatch[dfa->subexp_map[reg_idx] + 1].rm_so;
|
||||
pmatch[reg_idx + 1].rm_eo
|
||||
= pmatch[dfa->subexp_map[reg_idx] + 1].rm_eo;
|
||||
}
|
||||
for (reg_idx = 0; reg_idx + 1 < nmatch; reg_idx++)
|
||||
if (dfa->subexp_map[reg_idx] != reg_idx)
|
||||
{
|
||||
pmatch[reg_idx + 1].rm_so
|
||||
= pmatch[dfa->subexp_map[reg_idx] + 1].rm_so;
|
||||
pmatch[reg_idx + 1].rm_eo
|
||||
= pmatch[dfa->subexp_map[reg_idx] + 1].rm_eo;
|
||||
}
|
||||
}
|
||||
|
||||
free_return:
|
||||
@@ -936,6 +953,7 @@ re_search_internal (preg, string, length, start, range, stop, nmatch, pmatch,
|
||||
}
|
||||
|
||||
static reg_errcode_t
|
||||
__attribute_warn_unused_result__
|
||||
prune_impossible_nodes (mctx)
|
||||
re_match_context_t *mctx;
|
||||
{
|
||||
@@ -950,6 +968,11 @@ prune_impossible_nodes (mctx)
|
||||
#endif
|
||||
match_last = mctx->match_last;
|
||||
halt_node = mctx->last_node;
|
||||
|
||||
/* Avoid overflow. */
|
||||
if (BE (SIZE_MAX / sizeof (re_dfastate_t *) <= match_last, 0))
|
||||
return REG_ESPACE;
|
||||
|
||||
sifted_states = re_malloc (re_dfastate_t *, match_last + 1);
|
||||
if (BE (sifted_states == NULL, 0))
|
||||
{
|
||||
@@ -1069,7 +1092,7 @@ acquire_init_state_context (reg_errcode_t *err, const re_match_context_t *mctx,
|
||||
index of the buffer. */
|
||||
|
||||
static int
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
check_matching (re_match_context_t *mctx, int fl_longest_match,
|
||||
int *p_match_first)
|
||||
{
|
||||
@@ -1108,7 +1131,7 @@ check_matching (re_match_context_t *mctx, int fl_longest_match,
|
||||
{
|
||||
err = transit_state_bkref (mctx, &cur_state->nodes);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
return err;
|
||||
return err;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1134,17 +1157,18 @@ check_matching (re_match_context_t *mctx, int fl_longest_match,
|
||||
re_dfastate_t *old_state = cur_state;
|
||||
int next_char_idx = re_string_cur_idx (&mctx->input) + 1;
|
||||
|
||||
if (BE (next_char_idx >= mctx->input.bufs_len, 0)
|
||||
|| (BE (next_char_idx >= mctx->input.valid_len, 0)
|
||||
&& mctx->input.valid_len < mctx->input.len))
|
||||
{
|
||||
err = extend_buffers (mctx);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
if ((BE (next_char_idx >= mctx->input.bufs_len, 0)
|
||||
&& mctx->input.bufs_len < mctx->input.len)
|
||||
|| (BE (next_char_idx >= mctx->input.valid_len, 0)
|
||||
&& mctx->input.valid_len < mctx->input.len))
|
||||
{
|
||||
err = extend_buffers (mctx, next_char_idx + 1);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
{
|
||||
assert (err == REG_ESPACE);
|
||||
return -2;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
cur_state = transit_state (&err, mctx, cur_state);
|
||||
if (mctx->state_log != NULL)
|
||||
@@ -1263,20 +1287,20 @@ proceed_next_node (const re_match_context_t *mctx, int nregs, regmatch_t *regs,
|
||||
int candidate = edests->elems[i];
|
||||
if (!re_node_set_contains (cur_nodes, candidate))
|
||||
continue;
|
||||
if (dest_node == -1)
|
||||
if (dest_node == -1)
|
||||
dest_node = candidate;
|
||||
|
||||
else
|
||||
else
|
||||
{
|
||||
/* In order to avoid infinite loop like "(a*)*", return the second
|
||||
epsilon-transition if the first was already considered. */
|
||||
epsilon-transition if the first was already considered. */
|
||||
if (re_node_set_contains (eps_via_nodes, dest_node))
|
||||
return candidate;
|
||||
return candidate;
|
||||
|
||||
/* Otherwise, push the second epsilon-transition on the fail stack. */
|
||||
else if (fs != NULL
|
||||
&& push_fail_stack (fs, *pidx, candidate, nregs, regs,
|
||||
eps_via_nodes))
|
||||
eps_via_nodes))
|
||||
return -2;
|
||||
|
||||
/* We know we are going to exit. */
|
||||
@@ -1342,7 +1366,7 @@ proceed_next_node (const re_match_context_t *mctx, int nregs, regmatch_t *regs,
|
||||
}
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
push_fail_stack (struct re_fail_stack_t *fs, int str_idx, int dest_node,
|
||||
int nregs, regmatch_t *regs, re_node_set *eps_via_nodes)
|
||||
{
|
||||
@@ -1389,7 +1413,7 @@ pop_fail_stack (struct re_fail_stack_t *fs, int *pidx, int nregs,
|
||||
pmatch[i].rm_so == pmatch[i].rm_eo == -1 for 0 < i < nmatch. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
set_regs (const regex_t *preg, const re_match_context_t *mctx, size_t nmatch,
|
||||
regmatch_t *pmatch, int fl_backtrack)
|
||||
{
|
||||
@@ -1624,7 +1648,7 @@ sift_states_backward (const re_match_context_t *mctx, re_sift_context_t *sctx)
|
||||
if (mctx->state_log[str_idx])
|
||||
{
|
||||
err = build_sifted_states (mctx, sctx, str_idx, &cur_dest);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
goto free_return;
|
||||
}
|
||||
|
||||
@@ -1643,7 +1667,7 @@ sift_states_backward (const re_match_context_t *mctx, re_sift_context_t *sctx)
|
||||
}
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
build_sifted_states (const re_match_context_t *mctx, re_sift_context_t *sctx,
|
||||
int str_idx, re_node_set *cur_dest)
|
||||
{
|
||||
@@ -1710,12 +1734,13 @@ clean_state_log_if_needed (re_match_context_t *mctx, int next_state_log_idx)
|
||||
{
|
||||
int top = mctx->state_log_top;
|
||||
|
||||
if (next_state_log_idx >= mctx->input.bufs_len
|
||||
if ((next_state_log_idx >= mctx->input.bufs_len
|
||||
&& mctx->input.bufs_len < mctx->input.len)
|
||||
|| (next_state_log_idx >= mctx->input.valid_len
|
||||
&& mctx->input.valid_len < mctx->input.len))
|
||||
{
|
||||
reg_errcode_t err;
|
||||
err = extend_buffers (mctx);
|
||||
err = extend_buffers (mctx, next_state_log_idx + 1);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
return err;
|
||||
}
|
||||
@@ -1805,7 +1830,7 @@ update_cur_sifted_state (const re_match_context_t *mctx,
|
||||
}
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
add_epsilon_src_nodes (const re_dfa_t *dfa, re_node_set *dest_nodes,
|
||||
const re_node_set *candidates)
|
||||
{
|
||||
@@ -1820,10 +1845,14 @@ add_epsilon_src_nodes (const re_dfa_t *dfa, re_node_set *dest_nodes,
|
||||
{
|
||||
err = re_node_set_alloc (&state->inveclosure, dest_nodes->nelem);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
return REG_ESPACE;
|
||||
return REG_ESPACE;
|
||||
for (i = 0; i < dest_nodes->nelem; i++)
|
||||
re_node_set_merge (&state->inveclosure,
|
||||
dfa->inveclosures + dest_nodes->elems[i]);
|
||||
{
|
||||
err = re_node_set_merge (&state->inveclosure,
|
||||
dfa->inveclosures + dest_nodes->elems[i]);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
return REG_ESPACE;
|
||||
}
|
||||
}
|
||||
return re_node_set_add_intersect (dest_nodes, candidates,
|
||||
&state->inveclosure);
|
||||
@@ -1935,7 +1964,7 @@ check_dst_limits_calc_pos_1 (const re_match_context_t *mctx, int boundaries,
|
||||
{
|
||||
struct re_backref_cache_entry *ent = mctx->bkref_ents + bkref_idx;
|
||||
do
|
||||
{
|
||||
{
|
||||
int dst, cpos;
|
||||
|
||||
if (ent->node != node)
|
||||
@@ -1956,9 +1985,9 @@ check_dst_limits_calc_pos_1 (const re_match_context_t *mctx, int boundaries,
|
||||
if (dst == from_node)
|
||||
{
|
||||
if (boundaries & 1)
|
||||
return -1;
|
||||
return -1;
|
||||
else /* if (boundaries & 2) */
|
||||
return 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
cpos =
|
||||
@@ -1972,7 +2001,7 @@ check_dst_limits_calc_pos_1 (const re_match_context_t *mctx, int boundaries,
|
||||
if (subexp_idx < BITSET_WORD_BITS)
|
||||
ent->eps_reachable_subexps_map
|
||||
&= ~((bitset_word_t) 1 << subexp_idx);
|
||||
}
|
||||
}
|
||||
while (ent++->more);
|
||||
}
|
||||
break;
|
||||
@@ -2114,7 +2143,7 @@ check_subexp_limits (const re_dfa_t *dfa, re_node_set *dest_nodes,
|
||||
}
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
sift_states_bkref (const re_match_context_t *mctx, re_sift_context_t *sctx,
|
||||
int str_idx, const re_node_set *candidates)
|
||||
{
|
||||
@@ -2197,7 +2226,7 @@ sift_states_bkref (const re_match_context_t *mctx, re_sift_context_t *sctx,
|
||||
re_node_set_remove (&local_sctx.limits, enabled_idx);
|
||||
|
||||
/* mctx->bkref_ents may have changed, reload the pointer. */
|
||||
entry = mctx->bkref_ents + enabled_idx;
|
||||
entry = mctx->bkref_ents + enabled_idx;
|
||||
}
|
||||
while (enabled_idx++, entry++->more);
|
||||
}
|
||||
@@ -2244,7 +2273,7 @@ sift_states_iter_mb (const re_match_context_t *mctx, re_sift_context_t *sctx,
|
||||
update the destination of STATE_LOG. */
|
||||
|
||||
static re_dfastate_t *
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
transit_state (reg_errcode_t *err, re_match_context_t *mctx,
|
||||
re_dfastate_t *state)
|
||||
{
|
||||
@@ -2278,7 +2307,7 @@ transit_state (reg_errcode_t *err, re_match_context_t *mctx,
|
||||
|
||||
trtable = state->word_trtable;
|
||||
if (BE (trtable != NULL, 1))
|
||||
{
|
||||
{
|
||||
unsigned int context;
|
||||
context
|
||||
= re_string_context_at (&mctx->input,
|
||||
@@ -2324,21 +2353,21 @@ merge_state_with_log (reg_errcode_t *err, re_match_context_t *mctx,
|
||||
unsigned int context;
|
||||
re_node_set next_nodes, *log_nodes, *table_nodes = NULL;
|
||||
/* If (state_log[cur_idx] != 0), it implies that cur_idx is
|
||||
the destination of a multibyte char/collating element/
|
||||
back reference. Then the next state is the union set of
|
||||
these destinations and the results of the transition table. */
|
||||
the destination of a multibyte char/collating element/
|
||||
back reference. Then the next state is the union set of
|
||||
these destinations and the results of the transition table. */
|
||||
pstate = mctx->state_log[cur_idx];
|
||||
log_nodes = pstate->entrance_nodes;
|
||||
if (next_state != NULL)
|
||||
{
|
||||
table_nodes = next_state->entrance_nodes;
|
||||
*err = re_node_set_init_union (&next_nodes, table_nodes,
|
||||
{
|
||||
table_nodes = next_state->entrance_nodes;
|
||||
*err = re_node_set_init_union (&next_nodes, table_nodes,
|
||||
log_nodes);
|
||||
if (BE (*err != REG_NOERROR, 0))
|
||||
if (BE (*err != REG_NOERROR, 0))
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
else
|
||||
next_nodes = *log_nodes;
|
||||
next_nodes = *log_nodes;
|
||||
/* Note: We already add the nodes of the initial state,
|
||||
then we don't need to add them here. */
|
||||
|
||||
@@ -2346,12 +2375,12 @@ merge_state_with_log (reg_errcode_t *err, re_match_context_t *mctx,
|
||||
re_string_cur_idx (&mctx->input) - 1,
|
||||
mctx->eflags);
|
||||
next_state = mctx->state_log[cur_idx]
|
||||
= re_acquire_state_context (err, dfa, &next_nodes, context);
|
||||
= re_acquire_state_context (err, dfa, &next_nodes, context);
|
||||
/* We don't need to check errors here, since the return value of
|
||||
this function is next_state and ERR is already set. */
|
||||
this function is next_state and ERR is already set. */
|
||||
|
||||
if (table_nodes != NULL)
|
||||
re_node_set_free (&next_nodes);
|
||||
re_node_set_free (&next_nodes);
|
||||
}
|
||||
|
||||
if (BE (dfa->nbackref, 0) && next_state != NULL)
|
||||
@@ -2392,9 +2421,9 @@ find_recover_state (reg_errcode_t *err, re_match_context_t *mctx)
|
||||
|
||||
do
|
||||
{
|
||||
if (++cur_str_idx > max)
|
||||
return NULL;
|
||||
re_string_skip_bytes (&mctx->input, 1);
|
||||
if (++cur_str_idx > max)
|
||||
return NULL;
|
||||
re_string_skip_bytes (&mctx->input, 1);
|
||||
}
|
||||
while (mctx->state_log[cur_str_idx] == NULL);
|
||||
|
||||
@@ -2501,7 +2530,7 @@ transit_state_mb (re_match_context_t *mctx, re_dfastate_t *pstate)
|
||||
re_dfastate_t *dest_state;
|
||||
|
||||
if (!dfa->nodes[cur_node_idx].accept_mb)
|
||||
continue;
|
||||
continue;
|
||||
|
||||
if (dfa->nodes[cur_node_idx].constraint)
|
||||
{
|
||||
@@ -2669,7 +2698,7 @@ transit_state_bkref (re_match_context_t *mctx, const re_node_set *nodes)
|
||||
delay these checking for prune_impossible_nodes(). */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
get_subexp (re_match_context_t *mctx, int bkref_node, int bkref_str_idx)
|
||||
{
|
||||
const re_dfa_t *const dfa = mctx->dfa;
|
||||
@@ -2682,7 +2711,7 @@ get_subexp (re_match_context_t *mctx, int bkref_node, int bkref_str_idx)
|
||||
const struct re_backref_cache_entry *entry
|
||||
= mctx->bkref_ents + cache_idx;
|
||||
do
|
||||
if (entry->node == bkref_node)
|
||||
if (entry->node == bkref_node)
|
||||
return REG_NOERROR; /* We already checked it. */
|
||||
while (entry++->more);
|
||||
}
|
||||
@@ -2765,7 +2794,7 @@ get_subexp (re_match_context_t *mctx, int bkref_node, int bkref_str_idx)
|
||||
if (bkref_str_off >= mctx->input.len)
|
||||
break;
|
||||
|
||||
err = extend_buffers (mctx);
|
||||
err = extend_buffers (mctx, bkref_str_off + 1);
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
return err;
|
||||
|
||||
@@ -2869,7 +2898,7 @@ find_subexp_node (const re_dfa_t *dfa, const re_node_set *nodes,
|
||||
Return REG_NOERROR if it can arrive, or REG_NOMATCH otherwise. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
check_arrival (re_match_context_t *mctx, state_array_t *path, int top_node,
|
||||
int top_str, int last_node, int last_str, int type)
|
||||
{
|
||||
@@ -3030,7 +3059,7 @@ check_arrival (re_match_context_t *mctx, state_array_t *path, int top_node,
|
||||
Can't we unify them? */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
check_arrival_add_next_nodes (re_match_context_t *mctx, int str_idx,
|
||||
re_node_set *cur_nodes, re_node_set *next_nodes)
|
||||
{
|
||||
@@ -3162,7 +3191,7 @@ check_arrival_expand_ecl (const re_dfa_t *dfa, re_node_set *cur_nodes,
|
||||
problematic append it to DST_NODES. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
check_arrival_expand_ecl_sub (const re_dfa_t *dfa, re_node_set *dst_nodes,
|
||||
int target, int ex_subexp, int type)
|
||||
{
|
||||
@@ -3206,7 +3235,7 @@ check_arrival_expand_ecl_sub (const re_dfa_t *dfa, re_node_set *dst_nodes,
|
||||
in MCTX->BKREF_ENTS. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
expand_bkref_cache (re_match_context_t *mctx, re_node_set *cur_nodes,
|
||||
int cur_str, int subexp_num, int type)
|
||||
{
|
||||
@@ -3347,6 +3376,8 @@ build_trtable (const re_dfa_t *dfa, re_dfastate_t *state)
|
||||
{
|
||||
state->trtable = (re_dfastate_t **)
|
||||
calloc (sizeof (re_dfastate_t *), SBC_MAX);
|
||||
if (BE (state->trtable == NULL, 0))
|
||||
return 0;
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
@@ -3356,6 +3387,13 @@ build_trtable (const re_dfa_t *dfa, re_dfastate_t *state)
|
||||
if (BE (err != REG_NOERROR, 0))
|
||||
goto out_free;
|
||||
|
||||
/* Avoid arithmetic overflow in size calculation. */
|
||||
if (BE ((((SIZE_MAX - (sizeof (re_node_set) + sizeof (bitset_t)) * SBC_MAX)
|
||||
/ (3 * sizeof (re_dfastate_t *)))
|
||||
< ndests),
|
||||
0))
|
||||
goto out_free;
|
||||
|
||||
if (__libc_use_alloca ((sizeof (re_node_set) + sizeof (bitset_t)) * SBC_MAX
|
||||
+ ndests * 3 * sizeof (re_dfastate_t *)))
|
||||
dest_states = (re_dfastate_t **)
|
||||
@@ -3564,13 +3602,13 @@ group_nodes_into_DFAstates (const re_dfa_t *dfa, const re_dfastate_t *state,
|
||||
}
|
||||
#ifdef RE_ENABLE_I18N
|
||||
else if (type == OP_UTF8_PERIOD)
|
||||
{
|
||||
{
|
||||
memset (accepts, '\xff', sizeof (bitset_t) / 2);
|
||||
if (!(dfa->syntax & RE_DOT_NEWLINE))
|
||||
bitset_clear (accepts, '\n');
|
||||
if (dfa->syntax & RE_DOT_NOT_NULL)
|
||||
bitset_clear (accepts, '\0');
|
||||
}
|
||||
}
|
||||
#endif
|
||||
else
|
||||
continue;
|
||||
@@ -3711,6 +3749,10 @@ group_nodes_into_DFAstates (const re_dfa_t *dfa, const re_dfastate_t *state,
|
||||
one collating element like '.', '[a-z]', opposite to the other nodes
|
||||
can only accept one byte. */
|
||||
|
||||
# ifdef _LIBC
|
||||
# include <locale/weight.h>
|
||||
# endif
|
||||
|
||||
static int
|
||||
internal_function
|
||||
check_node_accept_bytes (const re_dfa_t *dfa, int node_idx,
|
||||
@@ -3775,7 +3817,7 @@ check_node_accept_bytes (const re_dfa_t *dfa, int node_idx,
|
||||
if (node->type == OP_PERIOD)
|
||||
{
|
||||
if (char_len <= 1)
|
||||
return 0;
|
||||
return 0;
|
||||
/* FIXME: I don't think this if is needed, as both '\n'
|
||||
and '\0' are char_len == 1. */
|
||||
/* '.' accepts any one character except the following two cases. */
|
||||
@@ -3830,8 +3872,6 @@ check_node_accept_bytes (const re_dfa_t *dfa, int node_idx,
|
||||
const int32_t *table, *indirect;
|
||||
const unsigned char *weights, *extra;
|
||||
const char *collseqwc;
|
||||
/* This #include defines a local function! */
|
||||
# include <locale/weight.h>
|
||||
|
||||
/* match with collating_symbol? */
|
||||
if (cset->ncoll_syms)
|
||||
@@ -3887,7 +3927,7 @@ check_node_accept_bytes (const re_dfa_t *dfa, int node_idx,
|
||||
_NL_CURRENT (LC_COLLATE, _NL_COLLATE_EXTRAMB);
|
||||
indirect = (const int32_t *)
|
||||
_NL_CURRENT (LC_COLLATE, _NL_COLLATE_INDIRECTMB);
|
||||
int32_t idx = findidx (&cp);
|
||||
int32_t idx = findidx (table, indirect, extra, &cp, elem_len);
|
||||
if (idx > 0)
|
||||
for (i = 0; i < cset->nequiv_classes; ++i)
|
||||
{
|
||||
@@ -3998,7 +4038,7 @@ find_collation_sequence_value (const unsigned char *mbs, size_t mbs_len)
|
||||
/* Skip the collation sequence value. */
|
||||
idx += sizeof (uint32_t);
|
||||
/* Skip the wide char sequence of the collating element. */
|
||||
idx = idx + sizeof (uint32_t) * (extra[idx] + 1);
|
||||
idx = idx + sizeof (uint32_t) * (*(int32_t *) (extra + idx) + 1);
|
||||
/* If we found the entry, return the sequence value. */
|
||||
if (found)
|
||||
return *(uint32_t *) (extra + idx);
|
||||
@@ -4025,18 +4065,18 @@ check_node_accept (const re_match_context_t *mctx, const re_token_t *node,
|
||||
{
|
||||
case CHARACTER:
|
||||
if (node->opr.c != ch)
|
||||
return 0;
|
||||
return 0;
|
||||
break;
|
||||
|
||||
case SIMPLE_BRACKET:
|
||||
if (!bitset_contain (node->opr.sbcset, ch))
|
||||
return 0;
|
||||
return 0;
|
||||
break;
|
||||
|
||||
#ifdef RE_ENABLE_I18N
|
||||
case OP_UTF8_PERIOD:
|
||||
if (ch >= 0x80)
|
||||
return 0;
|
||||
return 0;
|
||||
/* FALLTHROUGH */
|
||||
#endif
|
||||
case OP_PERIOD:
|
||||
@@ -4065,14 +4105,20 @@ check_node_accept (const re_match_context_t *mctx, const re_token_t *node,
|
||||
/* Extend the buffers, if the buffers have run out. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
extend_buffers (re_match_context_t *mctx)
|
||||
internal_function __attribute_warn_unused_result__
|
||||
extend_buffers (re_match_context_t *mctx, int min_len)
|
||||
{
|
||||
reg_errcode_t ret;
|
||||
re_string_t *pstr = &mctx->input;
|
||||
|
||||
/* Double the lengthes of the buffers. */
|
||||
ret = re_string_realloc_buffers (pstr, pstr->bufs_len * 2);
|
||||
/* Avoid overflow. */
|
||||
if (BE (INT_MAX / 2 / sizeof (re_dfastate_t *) <= pstr->bufs_len, 0))
|
||||
return REG_ESPACE;
|
||||
|
||||
/* Double the lengthes of the buffers, but allocate at least MIN_LEN. */
|
||||
ret = re_string_realloc_buffers (pstr,
|
||||
MAX (min_len,
|
||||
MIN (pstr->len, pstr->bufs_len * 2)));
|
||||
if (BE (ret != REG_NOERROR, 0))
|
||||
return ret;
|
||||
|
||||
@@ -4124,7 +4170,7 @@ extend_buffers (re_match_context_t *mctx)
|
||||
/* Initialize MCTX. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
match_ctx_init (re_match_context_t *mctx, int eflags, int n)
|
||||
{
|
||||
mctx->eflags = eflags;
|
||||
@@ -4197,7 +4243,7 @@ match_ctx_free (re_match_context_t *mctx)
|
||||
*/
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
match_ctx_add_entry (re_match_context_t *mctx, int node, int str_idx, int from,
|
||||
int to)
|
||||
{
|
||||
@@ -4269,7 +4315,7 @@ search_cur_bkref_entry (const re_match_context_t *mctx, int str_idx)
|
||||
at STR_IDX. */
|
||||
|
||||
static reg_errcode_t
|
||||
internal_function
|
||||
internal_function __attribute_warn_unused_result__
|
||||
match_ctx_add_subtop (re_match_context_t *mctx, int node, int str_idx)
|
||||
{
|
||||
#ifdef DEBUG
|
||||
|
||||
Reference in New Issue
Block a user