* Import the r_regex api in libr/util/regex from OpenBSD source

- Added a r2-like API on top of it
  - Make RSearch and RMagic use this new api, so
* Only load default magicpath files when no file is passed to RMagic
* Initial work on r_listrange optimization in RAnal
  - #define USE_NEW_FCN_STORE
  - Still work-in-progress
* Implemented a RPoolFactory singleton api to accelerate
  allocations of little objects in the future
* Fix sys/mingw32.sh for osx
* Added sys/maemo.sh
This commit is contained in:
pancake 2011-09-14 02:07:06 +02:00
parent 597fc9198f
commit e8af14966b
31 changed files with 5283 additions and 161 deletions

View file

@ -7,6 +7,6 @@
# -- pancake
acr -p
if [ -n "$1" ]; then
echo "./configure $@"
./configure $@
echo "./configure $@"
./configure $@
fi

View file

@ -32,6 +32,7 @@ R_API RAnal *r_anal_new() {
anal->reg = r_reg_new ();
anal->lineswidth = 0;
anal->fcns = r_anal_fcn_list_new ();
anal->fcnstore = r_listrange_new ();
anal->refs = r_anal_ref_list_new ();
anal->vartypes = r_anal_var_type_list_new ();
r_anal_set_bits (anal, 32);
@ -50,11 +51,10 @@ R_API RAnal *r_anal_new() {
R_API RAnal *r_anal_free(RAnal *anal) {
if (anal) {
/* TODO: Free a->anals here */
if (anal->fcns)
r_list_free (anal->fcns);
if (anal->vartypes)
r_list_free (anal->vartypes);
/* TODO: Free anals here */
r_listrange_free (anal->fcnstore);
r_list_free (anal->fcns);
r_list_free (anal->vartypes);
}
free (anal);
return NULL;
@ -95,8 +95,7 @@ R_API int r_anal_use(RAnal *anal, const char *name) {
}
R_API int r_anal_set_reg_profile(RAnal *anal) {
if (anal)
if (anal->cur && anal->cur->set_reg_profile)
if (anal && anal->cur && anal->cur->set_reg_profile)
return anal->cur->set_reg_profile (anal);
return R_FALSE;
}

View file

@ -5,26 +5,27 @@
#include <r_util.h>
#include <r_list.h>
#define VERBOSE if(0)
/* work in progress */
#define USE_NEW_FCN_STORE 0
R_API RAnalFcn *r_anal_fcn_new() {
RAnalFcn *fcn = R_NEW (RAnalFcn);
if (fcn) {
memset (fcn, 0, sizeof (RAnalFcn));
fcn->addr = -1;
fcn->stack = 0;
fcn->vars = r_anal_var_list_new ();
fcn->refs = r_anal_ref_list_new ();
fcn->xrefs = r_anal_ref_list_new ();
fcn->bbs = r_anal_bb_list_new ();
fcn->fingerprint = NULL;
fcn->diff = r_anal_diff_new ();
}
if (!fcn) return NULL;
memset (fcn, 0, sizeof (RAnalFcn));
fcn->addr = -1;
fcn->stack = 0;
fcn->vars = r_anal_var_list_new ();
fcn->refs = r_anal_ref_list_new ();
fcn->xrefs = r_anal_ref_list_new ();
fcn->bbs = r_anal_bb_list_new ();
fcn->fingerprint = NULL;
fcn->diff = r_anal_diff_new ();
return fcn;
}
R_API RList *r_anal_fcn_list_new() {
RList *list = r_list_new ();
if (!list) return NULL;
list->free = &r_anal_fcn_free;
return list;
}
@ -55,7 +56,7 @@ R_API int r_anal_fcn(RAnal *anal, RAnalFcn *fcn, ut64 addr, ut8 *buf, ut64 len,
while (idx < len) {
if ((oplen = r_anal_op (anal, &op, addr+idx, buf+idx, len-idx)) == 0) {
if (idx == 0) {
VERBOSE eprintf ("Unknown opcode at 0x%08"PFMT64x"\n", addr+idx);
// eprintf ("Unknown opcode at 0x%08"PFMT64x"\n", addr+idx);
return R_ANAL_RET_END;
} else break;
}
@ -116,14 +117,8 @@ R_API int r_anal_fcn(RAnal *anal, RAnalFcn *fcn, ut64 addr, ut8 *buf, ut64 len,
}
R_API int r_anal_fcn_add(RAnal *anal, ut64 addr, ut64 size, const char *name, int type, RAnalDiff *diff) {
RAnalFcn *fcn = NULL, *fcni;
RListIter *iter;
int append = 0;
r_list_foreach (anal->fcns, iter, fcni)
if (addr == fcni->addr) {
fcn = fcni;
break;
}
RAnalFcn *fcn = r_anal_fcn_find (anal, addr, R_ANAL_FCN_TYPE_ROOT);
if (fcn == NULL) {
if (!(fcn = r_anal_fcn_new ()))
return R_FALSE;
@ -141,40 +136,64 @@ R_API int r_anal_fcn_add(RAnal *anal, ut64 addr, ut64 size, const char *name, in
if (diff->name)
fcn->diff->name = strdup (diff->name);
}
#if USE_NEW_FCN_STORE
if (append) r_listrange_add (anal->fcnstore, fcn);
#else
if (append) r_list_append (anal->fcns, fcn);
#endif
return R_TRUE;
}
R_API int r_anal_fcn_del(RAnal *anal, ut64 addr) {
RAnalFcn *fcni;
RListIter it, *iter;
if (addr == 0) {
r_list_free (anal->fcns);
if (!(anal->fcns = r_anal_fcn_list_new ()))
return R_FALSE;
} else
r_list_foreach (anal->fcns, iter, fcni) {
if (addr >= fcni->addr && addr < fcni->addr+fcni->size) {
it.n = iter->n;
r_list_delete (anal->fcns, iter);
iter = &it;
} else {
#if USE_NEW_FCN_STORE
// XXX: must only get the function if starting at 0?
RAnalFcn *f = r_listrange_find_in_range (anal->fcnstore, addr);
if (f) r_listrange_del (anal->fcnstore, f);
#else
RAnalFcn *fcni;
RListIter it, *iter;
r_list_foreach (anal->fcns, iter, fcni) {
if (addr >= fcni->addr && addr < fcni->addr+fcni->size) {
it.n = iter->n;
r_list_delete (anal->fcns, iter);
iter = &it;
}
}
#endif
}
return R_TRUE;
}
R_API RAnalFcn *r_anal_fcn_find(RAnal *anal, ut64 addr, int type) {
int root = type & R_ANAL_FCN_TYPE_ROOT;
#if USE_NEW_FCN_STORE
// TODO: type is ignored here? wtf.. we need more work on fcnstore
if (root) r_listrange_find_root (anal->fcnstore, addr);
return r_listrange_find_in_range (anal->fcnstore, addr);
#else
RAnalFcn *fcn, *ret = NULL;
RListIter *iter;
r_list_foreach (anal->fcns, iter, fcn) {
if (type == R_ANAL_FCN_TYPE_NULL || (fcn->type & type))
if (addr == fcn->addr ||
(ret == NULL && (addr > fcn->addr && addr < fcn->addr+fcn->size)))
ret = fcn;
if (type == R_ANAL_FCN_TYPE_NULL || (fcn->type & type)) {
if (root) {
if (addr == fcn->addr)
ret = fcn;
} else {
if (addr == fcn->addr || (ret == NULL && (addr > fcn->addr && addr < fcn->addr+fcn->size)))
ret = fcn;
}
}
}
return ret;
#endif
}
/* rename RAnalFcnBB.add() */
R_API int r_anal_fcn_add_bb(RAnalFcn *fcn, ut64 addr, ut64 size, ut64 jump, ut64 fail, int type, RAnalDiff *diff) {
RAnalBlock *bb = NULL, *bbi;
RListIter *iter;
@ -185,7 +204,8 @@ R_API int r_anal_fcn_add_bb(RAnalFcn *fcn, ut64 addr, ut64 size, ut64 jump, ut64
bb = bbi;
mid = 0;
break;
} else if (addr > bbi->addr && addr < bbi->addr+bbi->size)
} else
if (addr > bbi->addr && addr < bbi->addr+bbi->size)
mid = 1;
}
if (mid)
@ -211,6 +231,7 @@ R_API int r_anal_fcn_add_bb(RAnalFcn *fcn, ut64 addr, ut64 size, ut64 jump, ut64
return R_TRUE;
}
// TODO: rename fcn_bb_split()
R_API int r_anal_fcn_split_bb(RAnalFcn *fcn, RAnalBlock *bb, ut64 addr) {
RAnalBlock *bbi;
RAnalOp *opi;
@ -253,6 +274,7 @@ R_API int r_anal_fcn_split_bb(RAnalFcn *fcn, RAnalBlock *bb, ut64 addr) {
return R_ANAL_RET_NEW;
}
// TODO: rename fcn_bb_overlap()
R_API int r_anal_fcn_overlap_bb(RAnalFcn *fcn, RAnalBlock *bb) {
RAnalBlock *bbi;
RListIter nit; // hack to make r_list_unlink not fail that hard
@ -290,9 +312,7 @@ R_API int r_anal_fcn_cc(RAnalFcn *fcn) {
int ret = 0, retbb;
r_list_foreach (fcn->bbs, iter, bbi) {
if ((bbi->type & R_ANAL_BB_TYPE_LAST))
retbb = 1;
else retbb = 0;
retbb = ((bbi->type & R_ANAL_BB_TYPE_LAST))? 1: 0;
ret += bbi->conditional + retbb;
}
return ret;
@ -324,37 +344,32 @@ R_API char *r_anal_fcn_to_string(RAnal *a, RAnalFcn* fs) {
if (!(arg = r_anal_fcn_get_var (fs, i,
R_ANAL_VAR_TYPE_ARG|R_ANAL_VAR_TYPE_ARGREG)))
break;
if (arg->array>1) {
if (i) sign = r_str_concatf (sign, ", %s %s:%02x[%d]",
if (arg->array>1)
sign = r_str_concatf (sign, i?", %s %s:%02x[%d]":"%s %s:%02x[%d]",
arg->vartype, arg->name, arg->delta, arg->array);
else sign = r_str_concatf (sign, "%s %s:%02x[%d]",
arg->vartype, arg->name, arg->delta, arg->array);
} else {
if (i) sign = r_str_concatf (sign, ", %s %s:%02x",
else sign = r_str_concatf (sign, i?", %s %s:%02x":"%s %s:%02x",
arg->vartype, arg->name, arg->delta);
else sign = r_str_concatf (sign, "%s %s:%02x",
arg->vartype, arg->name, arg->delta);
}
}
return (sign = r_str_concatf (sign, ");"));
}
// TODO: This function is not fully implemented
R_API int r_anal_fcn_from_string(RAnal *a, RAnalFcn *f, const char *_str) {
/* set function signature from string */
R_API int r_anal_fcn_from_string(RAnal *a, RAnalFcn *f, const char *sig) {
char *p, *q, *r, *str;
RAnalVar *var;
int i, arg;
if (!a || !f) {
if (!a || !f || !sig) {
eprintf ("r_anal_fcn_from_string: No function received\n");
return R_FALSE;
}
str = strdup (_str);
str = strdup (sig);
/* TODO : implement parser */
//r_list_destroy (fs->vars);
//set: fs->vars = r_list_new ();
//set: fs->name
eprintf ("ORIG=(%s)\n", _str);
eprintf ("ORIG=(%s)\n", sig);
p = strchr (str, '(');
if (!p) goto parsefail;
*p = 0;
@ -397,6 +412,7 @@ R_API int r_anal_fcn_from_string(RAnal *a, RAnalFcn *f, const char *_str) {
return R_TRUE;
parsefail:
free (str);
eprintf ("Function string parse fail\n");
return R_FALSE;
}

View file

@ -1,60 +1,101 @@
/* radare - LGPL - Copyright 2011 -- pancake<nopcode.org> */
/* this file contains a test implementation of the ~O(1) function search */
// TODO: REFACTOR: This must be a generic data structure named RListRange
// TODO: We need a standard struct named Surface1D {.addr, .size}, so we can
// simplify all this by just passing the offset of the field of the given ptr
// TODO: RListComparator does not supports *user
#define RANGEBITS 10
#define RANGE (1<<RANGEBITS)
#include <r_anal.h>
RAnalFcnStore* hl_new() {
RAnalFcnStore *s = R_NEW (RAnalFcnStore);
s->h = r_hashtable64_new();
s->l = r_list_new();
static int cmpfun(void *a, void *b) {
RAnalFcn *fa = (RAnalFcn*)a;
RAnalFcn *fb = (RAnalFcn*)b;
// TODO: swap sort order here or wtf?
return (fb->addr - fa->addr);
}
R_API RListRange* r_listrange_new () {
RListRange *s = R_NEW (RListRange);
s->h = r_hashtable64_new ();
s->l = r_list_new ();
return s;
}
static inline ut64 hl_key(ut64 addr) {
static inline ut64 r_listrange_key(ut64 addr) {
return (addr >> RANGEBITS);
}
static inline ut64 hl_next(ut64 addr) {
static inline ut64 r_listrange_next(ut64 addr) {
return (addr + RANGE);
}
void hl_free(RAnalFcnStore *s) {
R_API void r_listrange_free(RListRange *s) {
r_hashtable64_free (s->h);
r_list_destroy (s->l);
free (s);
}
static int cmpfun(void *a, void *b) {
// TODO
return 0;
}
void hl_add(RAnalFcnStore *s, RAnalFcn *f) {
R_API void r_listrange_add(RListRange *s, RAnalFcn *f) {
ut64 addr;
RList *list;
ut64 from = f->addr;
ut64 to = f->addr + f->size;
for (addr = from; addr<to; addr = hl_next (addr)) {
list = r_hashtable64_lookup (s->h, hl_key (addr));
if (!list) list = r_list_new ();
if (!r_list_contains (list, f)) // double rainbow :(
for (addr = from; addr<to; addr = r_listrange_next (addr)) {
ut32 key = r_listrange_key (addr);
list = r_hashtable64_lookup (s->h, key);
if (list) {
if (!r_list_contains (list, f))
r_list_add_sorted (list, f, cmpfun);
} else {
list = r_list_new ();
r_list_add_sorted (list, f, cmpfun);
r_hashtable64_insert (s->h, key, list);
}
}
r_list_add_sorted (s->l, f, cmpfun);
}
void hl_del(RAnalFcnStore *s, RAnalFcn *f) {
// TODO
R_API void r_listrange_del(RListRange *s, RAnalFcn *f) {
RList *list;
ut64 addr, from, to;
if (!f) return;
from = f->addr;
to = f->addr + f->size;
for (addr = from; addr<to; addr = r_listrange_next (addr)) {
list = r_hashtable64_lookup (s->h, r_listrange_key (addr));
if (list) r_list_delete_data (list, f);
}
r_list_delete_data (s->l, f);
}
RAnalFcn *hl_find(RAnalFcnStore* s, ut64 addr) {
R_API void r_listrange_resize(RListRange *s, RAnalFcn *f, int newsize) {
r_listrange_del (s, f);
f->size = newsize;
r_listrange_add (s, f);
}
R_API RAnalFcn *r_listrange_find_in_range(RListRange* s, ut64 addr) {
RAnalFcn *f;
RListIter *iter;
RList *list = r_hashtable64_lookup (s->h, hl_key (addr));
RList *list = r_hashtable64_lookup (s->h, r_listrange_key (addr));
if (list)
r_list_foreach (list, iter, f) {
if (addr >= f->addr && (addr < f->addr+f->size))
if (R_BETWEEN (f->addr, addr, f->addr+f->size))
return f;
}
return NULL;
}
R_API RAnalFcn *r_listrange_find_root(RListRange* s, ut64 addr) {
RAnalFcn *f;
RListIter *iter;
RList *list = r_hashtable64_lookup (s->h, r_listrange_key (addr));
if (list)
r_list_foreach (list, iter, f) {
if (addr == f->addr)
return f;
}
return NULL;
@ -62,7 +103,7 @@ RAnalFcn *hl_find(RAnalFcnStore* s, ut64 addr) {
#if 0
main() {
RHashTable64 *h = hl_new();
hl_add (h, f1);
RHashTable64 *h = r_listrange_new();
r_listrange_add (h, f1);
}
#endif

View file

@ -1587,11 +1587,11 @@ static void r_core_magic_at(RCore *core, const char *file, ut64 addr, int depth,
if (*file == ' ') file++;
if (!*file) file = NULL;
}
if (!oldfile || (file && strcmp (file, oldfile))) {
if (!oldfile || ck==NULL || (file && strcmp (file, oldfile))) {
// TODO: Move RMagic into RCore
r_magic_free (ck);
ck = r_magic_new (0);
if (r_magic_load (ck, MAGICPATH) == -1)
if ((file && *file) && (r_magic_load (ck, MAGICPATH) == -1))
eprintf ("failed r_magic_load ("MAGICPATH") %s\n", r_magic_error (ck));
}
if (file)

View file

@ -11,6 +11,7 @@
#include <r_util.h>
#include <r_syscall.h>
// TODO: Remove this define? /cc @nibble_ds
#define VERBOSE_ANAL if(0)
/* meta */
@ -159,6 +160,7 @@ typedef struct r_anal_t {
int split;
void *user;
RList *fcns;
RListRange *fcnstore;
RList *refs;
RList *vartypes;
RMeta *meta;
@ -279,7 +281,8 @@ enum {
R_ANAL_FCN_TYPE_FCN = 1,
R_ANAL_FCN_TYPE_LOC = 2,
R_ANAL_FCN_TYPE_SYM = 4,
R_ANAL_FCN_TYPE_IMP = 8
R_ANAL_FCN_TYPE_IMP = 8,
R_ANAL_FCN_TYPE_ROOT = 16 /* matching flag */
};
#define R_ANAL_VARSUBS 32
@ -380,6 +383,16 @@ typedef struct r_anal_plugin_t {
struct list_head list;
} RAnalPlugin;
/* --------- */ /* REFACTOR */ /* ---------- */
R_API RListRange* r_listrange_new ();
R_API void r_listrange_free(RListRange *s);
R_API void r_listrange_add(RListRange *s, RAnalFcn *f);
R_API void r_listrange_del(RListRange *s, RAnalFcn *f);
R_API void r_listrange_resize(RListRange *s, RAnalFcn *f, int newsize);
R_API RAnalFcn *r_listrange_find_in_range(RListRange* s, ut64 addr);
R_API RAnalFcn *r_listrange_find_root(RListRange* s, ut64 addr);
/* --------- */ /* REFACTOR */ /* ---------- */
#ifdef R_API
/* anal.c */
R_API RAnal *r_anal_new();

153
libr/include/r_regex.h Normal file
View file

@ -0,0 +1,153 @@
//#define _DARWIN_C_SOURCE
//#define _POSIX_C_SOURCE
/*
* Copyright (c) 2000 Apple Computer, Inc. All rights reserved.
*
* @APPLE_LICENSE_HEADER_START@
*
* This file contains Original Code and/or Modifications of Original Code
* as defined in and that are subject to the Apple Public Source License
* Version 2.0 (the 'License'). You may not use this file except in
* compliance with the License. Please obtain a copy of the License at
* http://www.opensource.apple.com/apsl/ and read it before using this
* file.
*
* The Original Code and all software distributed under the License are
* distributed on an 'AS IS' basis, WITHOUT WARRANTY OF ANY KIND, EITHER
* EXPRESS OR IMPLIED, AND APPLE HEREBY DISCLAIMS ALL SUCH WARRANTIES,
* INCLUDING WITHOUT LIMITATION, ANY WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE, QUIET ENJOYMENT OR NON-INFRINGEMENT.
* Please see the License for the specific language governing rights and
* limitations under the License.
*
* @APPLE_LICENSE_HEADER_END@
*/
/*-
* Copyright (c) 1992 Henry Spencer.
* Copyright (c) 1992, 1993
* The Regents of the University of California. All rights reserved.
*
* This code is derived from software contributed to Berkeley by
* Henry Spencer of the University of Toronto.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
* 3. All advertising materials mentioning features or use of this software
* must display the following acknowledgement:
* This product includes software developed by the University of
* California, Berkeley and its contributors.
* 4. Neither the name of the University nor the names of its contributors
* may be used to endorse or promote products derived from this software
* without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*
* @(#)regex.h 8.2 (Berkeley) 1/3/94
*/
#ifndef _R_REGEX_H_
#define _R_REGEX_H_
//#include <r_types.h>
//#define ut8 unsigned char
#define R_API
#include <sys/types.h>
#define __off_t off_t
#define __darwin_size_t size_t
/* types */
typedef __off_t regoff_t;
#ifndef _SIZE_T
#define _SIZE_T
typedef __darwin_size_t size_t;
#endif
typedef struct r_regex_t {
int re_magic;
size_t re_nsub; /* number of parenthesized subexpressions */
const char *re_endp; /* end pointer for R_REGEX_PEND */
struct re_guts *re_g; /* none of your business :-) */
} RRegex;
typedef struct r_regmatch_t {
regoff_t rm_so; /* start of match */
regoff_t rm_eo; /* end of match */
} RRegexMatch;
// TODO: rename to R_REGEX_ prefix
/* regcomp() flags */
#define R_REGEX_BASIC 0000
#define R_REGEX_EXTENDED 0001
#define R_REGEX_ICASE 0002
#define R_REGEX_NOSUB 0004
#define R_REGEX_NEWLINE 0010
#define R_REGEX_NOSPEC 0020
#define R_REGEX_PEND 0040
#define R_REGEX_DUMP 0200
/* regerror() flags */
#define R_REGEX_ENOSYS (-1) /* Reserved */
#define R_REGEX_NOMATCH 1
#define R_REGEX_BADPAT 2
#define R_REGEX_ECOLLATE 3
#define R_REGEX_ECTYPE 4
#define R_REGEX_EESCAPE 5
#define R_REGEX_ESUBREG 6
#define R_REGEX_EBRACK 7
#define R_REGEX_EPAREN 8
#define R_REGEX_EBRACE 9
#define R_REGEX_BADBR 10
#define R_REGEX_ERANGE 11
#define R_REGEX_ESPACE 12
#define R_REGEX_BADRPT 13
#if !defined(_POSIX_C_SOURCE) || defined(_DARWIN_C_SOURCE)
#define R_REGEX_EMPTY 14
#define R_REGEX_ASSERT 15
#define R_REGEX_INVARG 16
#define R_REGEX_ILLSEQ 17
#define R_REGEX_ATOI 255 /* convert name to number (!) */
#define R_REGEX_ITOA 0400 /* convert number to name (!) */
#endif /* (!_POSIX_C_SOURCE || _DARWIN_C_SOURCE) */
/* regexec() flags */
#define R_REGEX_NOTBOL 00001
#define R_REGEX_NOTEOL 00002
#if !defined(_POSIX_C_SOURCE) || defined(_DARWIN_C_SOURCE)
#define R_REGEX_STARTEND 00004
#define R_REGEX_TRACE 00400 /* tracing of execution */
#define R_REGEX_LARGE 01000 /* force large representation */
#define R_REGEX_BACKR 02000 /* force use of backref code */
#endif /* (!_POSIX_C_SOURCE || _DARWIN_C_SOURCE) */
R_API RRegex *r_regex_new (const char *pattern, const char *cflags);
R_API int r_regex_run (const char *pattern, const char *flags, const char *text);
R_API int r_regex_flags(const char *flags);
R_API int r_regex_comp(RRegex*, const char *, int);
R_API size_t r_regex_error(int, const RRegex*, char *, size_t);
/*
* gcc under c99 mode won't compile "[]" by itself. As a workaround,
* a dummy argument name is added.
*/
R_API int r_regex_exec(const RRegex *, const char *, size_t, RRegexMatch __pmatch[], int);
R_API void r_regex_free(RRegex *);
R_API void r_regex_fini(RRegex *);
#endif /* !_REGEX_H_ */

View file

@ -3,6 +3,7 @@
#include <r_types.h>
#include <btree.h>
#include <r_regex.h>
#include <r_list.h> // radare linked list
#include <r_flist.h> // radare fixed pointer array iterators
#include <list.h> // kernel linked list
@ -54,6 +55,11 @@ typedef struct r_mem_pool_t {
int poolcount;
} RMemoryPool;
typedef struct r_mem_pool_factory_t {
int limit;
RMemoryPool **pools;
} RPoolFactory;
typedef struct r_buf_t {
ut8 *buf;
int length;
@ -171,6 +177,13 @@ typedef struct r_mixed_t {
} RMixed;
/* TODO : THIS IS FROM See libr/anal/fcnstore.c for refactoring info */
typedef struct r_list_range_t {
RHashTable64 *h;
RList *l;
//RListComparator c;
} RListRange;
#ifdef R_API
R_API RMmap *r_file_mmap (const char *file, boolt rw);
@ -205,10 +218,21 @@ R_API int r_buf_fwrite_at (RBuffer *b, ut64 addr, ut8 *buf, const char *fmt, int
R_API void r_buf_free(RBuffer *b);
R_API ut64 r_mem_get_num(ut8 *b, int size, int endian);
R_API struct r_mem_pool_t* r_mem_pool_deinit(struct r_mem_pool_t *pool);
R_API struct r_mem_pool_t *r_mem_pool_new(int nodesize, int poolsize, int poolcount);
R_API struct r_mem_pool_t *r_mem_pool_free(struct r_mem_pool_t *pool);
R_API void* r_mem_pool_alloc(struct r_mem_pool_t *pool);
/* MEMORY POOL */
R_API RMemoryPool* r_mem_pool_deinit(struct r_mem_pool_t *pool);
R_API RMemoryPool *r_mem_pool_new(int nodesize, int poolsize, int poolcount);
R_API RMemoryPool *r_mem_pool_free(struct r_mem_pool_t *pool);
R_API void* r_mem_pool_alloc(RMemoryPool *pool);
/* FACTORY POOL */
R_API RPoolFactory *r_poolfactory_instance();
R_API void r_poolfactory_init (int limit);
R_API RPoolFactory* r_poolfactory_new(int limit);
R_API void *r_poolfactory_alloc(RPoolFactory *pf, int nodesize);
R_API void r_poolfactory_stats(RPoolFactory *pf);
R_API void r_poolfactory_free(RPoolFactory *pf);
R_API int r_mem_count(const ut8 **addr);
R_API RCache* r_cache_new();
R_API void r_cache_free(struct r_cache_t *c);

View file

@ -1,6 +1,7 @@
include ../config.mk
NAME=r_magic
DEPS=r_util
CFLAGS+=-I.
CFLAGS+=-DHAVE_CONFIG_H
OBJ=apprentice.o ascmagic.o compress.o fsmagic.o funcs.o is_tar.o magic.o print.o softmagic.o

View file

@ -31,6 +31,7 @@
*/
#include "file.h"
#include "r_regex.h"
#include <string.h>
#include <ctype.h>
#include <stdlib.h>
@ -255,21 +256,21 @@ static int match(RMagic *ms, struct r_magic *magic, ut32 nmagic, const ut8 *s, s
}
static int check_fmt(RMagic *ms, struct r_magic *m) {
regex_t rx;
RRegex rx;
int rc;
if (strchr (R_MAGIC_DESC, '%') == NULL)
return 0;
rc = regcomp (&rx, "%[-0-9\\.]*s", REG_EXTENDED|REG_NOSUB);
rc = r_regex_comp (&rx, "%[-0-9\\.]*s", R_REGEX_EXTENDED|R_REGEX_NOSUB);
if (rc) {
char errmsg[512];
(void)regerror (rc, &rx, errmsg, sizeof(errmsg));
r_regex_error (rc, &rx, errmsg, sizeof (errmsg));
file_magerror (ms, "regex error %d, (%s)", rc, errmsg);
return -1;
} else {
rc = regexec (&rx, R_MAGIC_DESC, 0, 0, 0);
regfree (&rx);
rc = r_regex_exec (&rx, R_MAGIC_DESC, 0, 0, 0);
r_regex_free (&rx);
return !rc;
}
}
@ -1300,25 +1301,25 @@ static int magiccheck(RMagic *ms, struct r_magic *m) {
}
case FILE_REGEX: {
int rc;
regex_t rx;
RRegex rx;
char errmsg[512];
if (ms->search.s == NULL)
return 0;
l = 0;
rc = regcomp(&rx, m->value.s,
REG_EXTENDED|REG_NEWLINE|
((m->str_flags & STRING_IGNORE_CASE) ? REG_ICASE : 0));
rc = r_regex_comp (&rx, m->value.s,
R_REGEX_EXTENDED|R_REGEX_NEWLINE|
((m->str_flags & STRING_IGNORE_CASE) ? R_REGEX_ICASE : 0));
if (rc) {
(void)regerror(rc, &rx, errmsg, sizeof(errmsg));
(void)r_regex_error(rc, &rx, errmsg, sizeof(errmsg));
file_magerror(ms, "regex error %d, (%s)",
rc, errmsg);
v = (ut64)-1;
} else {
regmatch_t pmatch[1];
#ifndef REG_STARTEND
#define REG_STARTEND 0
RRegexMatch pmatch[1];
#ifndef R_REGEX_STARTEND
#define R_REGEX_STARTEND 0
size_t l = ms->search.s_len - 1;
char c = ms->search.s[l];
((char *)(intptr_t)ms->search.s)[l] = '\0';
@ -1326,8 +1327,8 @@ static int magiccheck(RMagic *ms, struct r_magic *m) {
pmatch[0].rm_so = 0;
pmatch[0].rm_eo = ms->search.s_len;
#endif
rc = regexec (&rx, (const char *)ms->search.s, 1, pmatch, REG_STARTEND);
#if REG_STARTEND == 0
rc = r_regex_exec (&rx, (const char *)ms->search.s, 1, pmatch, R_REGEX_STARTEND);
#if R_REGEX_STARTEND == 0
((char *)(intptr_t)ms->search.s)[l] = c;
#endif
switch (rc) {
@ -1338,17 +1339,17 @@ static int magiccheck(RMagic *ms, struct r_magic *m) {
(size_t)(pmatch[0].rm_eo - pmatch[0].rm_so);
v = 0;
break;
case REG_NOMATCH:
case R_REGEX_NOMATCH:
v = 1;
break;
default:
(void)regerror(rc, &rx, errmsg, sizeof(errmsg));
(void)r_regex_error(rc, &rx, errmsg, sizeof(errmsg));
file_magerror(ms, "regexec error %d, (%s)",
rc, errmsg);
v = (ut64)-1;
break;
}
regfree(&rx);
r_regex_fini (&rx);
}
if (v == (ut64)-1)
return -1;

View file

@ -1,8 +1,7 @@
/* radare - LGPL - Copyright 2008-2010 pancake<nopcode.org> */
/* radare - LGPL - Copyright 2008-2011 pancake<nopcode.org> */
#include "r_search.h"
#if __UNIX__
#include <regex.h>
#include <r_regex.h>
R_API int r_search_regexp_update(void *_s, ut64 from, const ut8 *buf, int len) {
RSearch *s = (RSearch*)_s;
@ -16,27 +15,28 @@ R_API int r_search_regexp_update(void *_s, ut64 from, const ut8 *buf, int len) {
RSearchKeyword *kw;
r_list_foreach (s->kws, iter, kw) {
int reflags = REG_EXTENDED;
int reflags = R_REGEX_EXTENDED;
int ret, delta = 0;
regmatch_t matches[10];
regex_t compiled;
RRegexMatch matches[10];
RRegex compiled;
// TODO: memory leak with compiled foo
if (strchr (kw->binmask, 'i'))
reflags |= REG_ICASE;
reflags |= R_REGEX_ICASE;
if (regcomp (&compiled, kw->keyword, reflags)) {
if (r_regex_comp (&compiled, kw->keyword, reflags)) {
eprintf ("Cannot compile '%s' regexp\n",kw->keyword);
return -1;
}
foo:
ret = regexec (&compiled, buffer+delta, 1, matches, 0);
ret = r_regex_exec (&compiled, buffer+delta, 1, matches, 0);
if (ret) return 0;
do {
r_search_hit_new (s, kw, (ut64)(from+matches[0].rm_so+delta));
delta += matches[0].rm_so+1;
kw->count++;
count++;
} while (!regexec (&compiled, buffer+delta, 1, matches, 0));
} while (!r_regex_exec (&compiled, buffer+delta, 1, matches, 0));
if (delta == 0)
return 0;
@ -52,11 +52,3 @@ R_API int r_search_regexp_update(void *_s, ut64 from, const ut8 *buf, int len) {
}
return count;
}
#else
R_API int r_search_regexp_update(void *_s, ut64 from, const ut8 *buf, int len) {
eprintf ("r_search_regexp_update: unimplemented for this platform\n");
return -1;
}
#endif

View file

@ -1,10 +1,12 @@
include ../config.mk
NAME=r_util
OBJ=mem.o pool.o num.o str.o re.o hex.o file.o alloca.o range.o log.o
OBJ=mem.o pool.o num.o str.o hex.o file.o alloca.o range.o log.o
OBJ+=prof.o cache.o sys.o buf.o w32-sys.o base64.o name.o
OBJ+=list.o flist.o ht.o ht64.o mixed.o btree.o chmod.o
OBJ+=regex/regcomp.o regex/regerror.o regex/regexec.o
# DO NOT BUILD r_big api (not yet used and its buggy)
ifeq (1,0)
ifeq (${HAVE_LIB_GMP},1)

View file

@ -1,4 +1,4 @@
/* radare - LGPL - Copyright 2010 pancake<nopcode.org> */
/* radare - LGPL - Copyright 2010-2011 pancake<nopcode.org> */
#include <r_util.h>
#include <stdlib.h>
@ -20,9 +20,7 @@ R_API RMemoryPool* r_mem_pool_deinit(RMemoryPool *pool) {
}
R_API RMemoryPool *r_mem_pool_new(int nodesize, int poolsize, int poolcount) {
RMemoryPool *mp;
mp = R_NEW (RMemoryPool);
RMemoryPool *mp = R_NEW (RMemoryPool);
if (mp) {
if (poolsize<1)
poolsize = ALLOC_POOL_SIZE;
@ -61,3 +59,69 @@ R_API void* r_mem_pool_alloc(RMemoryPool *pool) {
// TODO: fix warning
return (void *)(&(pool->nodes[pool->npool][pool->ncount++]));
}
// TODO: not implemented
R_API int r_mem_pool_dealloc(RMemoryPool *pool, void *p) {
return R_FALSE;
}
/* poolfactory */
/* TODO: must tune */
static RPoolFactory single_pf = {0};
R_API RPoolFactory *r_poolfactory_instance() {
return &single_pf;
}
R_API void r_poolfactory_init (int limit) {
int size = limit * sizeof (RMemoryPool*);
single_pf.limit = limit+1;
free (single_pf.pools);
single_pf.pools = malloc (size);
memset (single_pf.pools, 0, size);
}
R_API RPoolFactory* r_poolfactory_new(int limit) {
if (limit>0) {
int size = sizeof (RMemoryPool*) * limit;
RPoolFactory *pf = R_NEW0 (RPoolFactory);
if (!pf) return NULL;
pf->limit = limit+1;
pf->pools = malloc (size);
memset (pf->pools, 0, size);
return pf;
}
return NULL;
}
R_API void *r_poolfactory_alloc(RPoolFactory *pf, int nodesize) {
if (nodesize > pf->limit)
return NULL;
if (!pf->pools[nodesize])
pf->pools[nodesize] = r_mem_pool_new (nodesize,
ALLOC_POOL_SIZE, ALLOC_POOL_COUNT);
return r_mem_pool_alloc (pf->pools[nodesize]);
}
// TODO: not implemented
R_API int r_poolfactory_dealloc(RPoolFactory *pool, void *p) {
return R_FALSE;
}
// TODO: add support for ranged limits, from-to
R_API void r_poolfactory_stats(RPoolFactory *pf) {
int i=0;
eprintf ("RPoolFactory stats:\n");
eprintf (" limits: %d\n", pf->limit);
for (i=0; i<pf->limit; i++) {
if (pf->pools[i])
eprintf (" size: %d\t npool: %d\t count: %d\n",
pf->pools[i]->nodesize,
pf->pools[i]->npool,
pf->pools[i]->ncount);
}
}
R_API void r_poolfactory_free(RPoolFactory *pf) {
free (pf->pools);
free (pf);
}

View file

@ -1,29 +0,0 @@
/* radare - LGPL - Copyright 2007-2010 pancake<nopcode.org> */
#include <r_util.h>
#if HAVE_REGEXP
#include <regex.h>
/* XXX: This code uses POSIX 2001 . can be nonportable */
#define NUM_MATCHES 16
#endif
/* returns 1 if 'str' matches 'reg' regexp */
R_API int r_str_re_match(const char *str, const char *reg) {
#if HAVE_REGEXP
regex_t preg;
regmatch_t pmatch[NUM_MATCHES];
if (regcomp(&preg, reg, REG_EXTENDED))
return -1;
return (regexec (&preg, str, NUM_MATCHES, pmatch, 0))?1:0;
#else
return -1;
#endif
}
R_API int r_str_re_replace(const char *str, const char *reg, const char *sub) {
/* TODO: not yet implemented */
return -1;
}
/* Added glob stuff here */

54
libr/util/regex/COPYRIGHT Normal file
View file

@ -0,0 +1,54 @@
$OpenBSD: COPYRIGHT,v 1.3 2003/06/02 20:18:36 millert Exp $
Copyright 1992, 1993, 1994 Henry Spencer. All rights reserved.
This software is not subject to any license of the American Telephone
and Telegraph Company or of the Regents of the University of California.
Permission is granted to anyone to use this software for any purpose on
any computer system, and to alter it and redistribute it, subject
to the following restrictions:
1. The author is not responsible for the consequences of use of this
software, no matter how awful, even if they arise from flaws in it.
2. The origin of this software must not be misrepresented, either by
explicit claim or by omission. Since few users ever read sources,
credits must appear in the documentation.
3. Altered versions must be plainly marked as such, and must not be
misrepresented as being the original software. Since few users
ever read sources, credits must appear in the documentation.
4. This notice may not be removed or altered.
=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=
/*-
* Copyright (c) 1994
* The Regents of the University of California. All rights reserved.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
* 3. Neither the name of the University nor the names of its contributors
* may be used to endorse or promote products derived from this software
* without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*
* @(#)COPYRIGHT 8.1 (Berkeley) 3/16/94
*/

13
libr/util/regex/Makefile Normal file
View file

@ -0,0 +1,13 @@
#CC=i386-mingw32-gcc
SRC= regcomp.c regerror.c regexec.c
CFLAGS+=-I.
CFLAGS+=-I../../include
#CFLAGS+=-Wall
#CFLAGS+=`pkg-config --cflags --libs r_util`
# implicit engine.c
all:
${CC} ${CFLAGS} test.c ${SRC}
clean:
rm -f *.o

5
libr/util/regex/README Normal file
View file

@ -0,0 +1,5 @@
Based on the OpenBSD's regex implementation
Modified to be portable (now compiles on windows, linux and *bsd including darwin)
cvs -qd anoncvs@anoncvs.ca.openbsd.org:/cvs get -P src/lib/libc/regex

68
libr/util/regex/cclass.h Normal file
View file

@ -0,0 +1,68 @@
/* $OpenBSD: cclass.h,v 1.5 2003/06/02 20:18:36 millert Exp $ */
/*-
* Copyright (c) 1992, 1993, 1994 Henry Spencer.
* Copyright (c) 1992, 1993, 1994
* The Regents of the University of California. All rights reserved.
*
* This code is derived from software contributed to Berkeley by
* Henry Spencer.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
* 3. Neither the name of the University nor the names of its contributors
* may be used to endorse or promote products derived from this software
* without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*
* @(#)cclass.h 8.3 (Berkeley) 3/20/94
*/
/* character-class table */
static struct cclass {
char *name;
char *chars;
char *multis;
} cclasses[] = {
{ "alnum", "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz\
0123456789", ""} ,
{ "alpha", "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz",
""} ,
{ "blank", " \t", ""} ,
{ "cntrl", "\007\b\t\n\v\f\r\1\2\3\4\5\6\16\17\20\21\22\23\24\
\25\26\27\30\31\32\33\34\35\36\37\177", ""} ,
{ "digit", "0123456789", ""} ,
{ "graph", "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz\
0123456789!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~",
""} ,
{ "lower", "abcdefghijklmnopqrstuvwxyz",
""} ,
{ "print", "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz\
0123456789!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~ ",
""} ,
{ "punct", "!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~",
""} ,
{ "space", "\t\n\v\f\r ", ""} ,
{ "upper", "ABCDEFGHIJKLMNOPQRSTUVWXYZ",
""} ,
{ "xdigit", "0123456789ABCDEFabcdef",
""} ,
{ NULL, 0, "" }
};

139
libr/util/regex/cname.h Normal file
View file

@ -0,0 +1,139 @@
/* $OpenBSD: cname.h,v 1.5 2003/06/02 20:18:36 millert Exp $ */
/*-
* Copyright (c) 1992, 1993, 1994 Henry Spencer.
* Copyright (c) 1992, 1993, 1994
* The Regents of the University of California. All rights reserved.
*
* This code is derived from software contributed to Berkeley by
* Henry Spencer.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
* 3. Neither the name of the University nor the names of its contributors
* may be used to endorse or promote products derived from this software
* without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*
* @(#)cname.h 8.3 (Berkeley) 3/20/94
*/
/* character-name table */
static struct cname {
char *name;
char code;
} cnames[] = {
{ "NUL", '\0' },
{ "SOH", '\001' },
{ "STX", '\002' },
{ "ETX", '\003' },
{ "EOT", '\004' },
{ "ENQ", '\005' },
{ "ACK", '\006' },
{ "BEL", '\007' },
{ "alert", '\007' },
{ "BS", '\010' },
{ "backspace", '\b' },
{ "HT", '\011' },
{ "tab", '\t' },
{ "LF", '\012' },
{ "newline", '\n' },
{ "VT", '\013' },
{ "vertical-tab", '\v' },
{ "FF", '\014' },
{ "form-feed", '\f' },
{ "CR", '\015' },
{ "carriage-return", '\r' },
{ "SO", '\016' },
{ "SI", '\017' },
{ "DLE", '\020' },
{ "DC1", '\021' },
{ "DC2", '\022' },
{ "DC3", '\023' },
{ "DC4", '\024' },
{ "NAK", '\025' },
{ "SYN", '\026' },
{ "ETB", '\027' },
{ "CAN", '\030' },
{ "EM", '\031' },
{ "SUB", '\032' },
{ "ESC", '\033' },
{ "IS4", '\034' },
{ "FS", '\034' },
{ "IS3", '\035' },
{ "GS", '\035' },
{ "IS2", '\036' },
{ "RS", '\036' },
{ "IS1", '\037' },
{ "US", '\037' },
{ "space", ' ' },
{ "exclamation-mark", '!' },
{ "quotation-mark", '"' },
{ "number-sign", '#' },
{ "dollar-sign", '$' },
{ "percent-sign", '%' },
{ "ampersand", '&' },
{ "apostrophe", '\'' },
{ "left-parenthesis", '(' },
{ "right-parenthesis", ')' },
{ "asterisk", '*' },
{ "plus-sign", '+' },
{ "comma", ',' },
{ "hyphen", '-' },
{ "hyphen-minus", '-' },
{ "period", '.' },
{ "full-stop", '.' },
{ "slash", '/' },
{ "solidus", '/' },
{ "zero", '0' },
{ "one", '1' },
{ "two", '2' },
{ "three", '3' },
{ "four", '4' },
{ "five", '5' },
{ "six", '6' },
{ "seven", '7' },
{ "eight", '8' },
{ "nine", '9' },
{ "colon", ':' },
{ "semicolon", ';' },
{ "less-than-sign", '<' },
{ "equals-sign", '=' },
{ "greater-than-sign", '>' },
{ "question-mark", '?' },
{ "commercial-at", '@' },
{ "left-square-bracket", '[' },
{ "backslash", '\\' },
{ "reverse-solidus", '\\' },
{ "right-square-bracket", ']' },
{ "circumflex", '^' },
{ "circumflex-accent", '^' },
{ "underscore", '_' },
{ "low-line", '_' },
{ "grave-accent", '`' },
{ "left-brace", '{' },
{ "left-curly-bracket", '{' },
{ "vertical-line", '|' },
{ "right-brace", '}' },
{ "right-curly-bracket", '}' },
{ "tilde", '~' },
{ "DEL", '\177' },
{ NULL, 0 }
};

1021
libr/util/regex/engine.c Normal file

File diff suppressed because it is too large Load diff

756
libr/util/regex/re_format.7 Normal file
View file

@ -0,0 +1,756 @@
.\" $OpenBSD: re_format.7,v 1.15 2010/07/15 20:51:38 schwarze Exp $
.\"
.\" Copyright (c) 1997, Phillip F Knaack. All rights reserved.
.\"
.\" Copyright (c) 1992, 1993, 1994 Henry Spencer.
.\" Copyright (c) 1992, 1993, 1994
.\" The Regents of the University of California. All rights reserved.
.\"
.\" This code is derived from software contributed to Berkeley by
.\" Henry Spencer.
.\"
.\" Redistribution and use in source and binary forms, with or without
.\" modification, are permitted provided that the following conditions
.\" are met:
.\" 1. Redistributions of source code must retain the above copyright
.\" notice, this list of conditions and the following disclaimer.
.\" 2. Redistributions in binary form must reproduce the above copyright
.\" notice, this list of conditions and the following disclaimer in the
.\" documentation and/or other materials provided with the distribution.
.\" 3. Neither the name of the University nor the names of its contributors
.\" may be used to endorse or promote products derived from this software
.\" without specific prior written permission.
.\"
.\" THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
.\" ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
.\" IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
.\" ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
.\" FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
.\" DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
.\" OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
.\" HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
.\" LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
.\" OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
.\" SUCH DAMAGE.
.\"
.\" @(#)re_format.7 8.3 (Berkeley) 3/20/94
.\"
.Dd $Mdocdate: July 15 2010 $
.Dt RE_FORMAT 7
.Os
.Sh NAME
.Nm re_format
.Nd POSIX regular expressions
.Sh DESCRIPTION
Regular expressions (REs),
as defined in
.St -p1003.1-2004 ,
come in two forms:
basic regular expressions
(BREs)
and extended regular expressions
(EREs).
Both forms of regular expressions are supported
by the interfaces described in
.Xr regex 3 .
Applications dealing with regular expressions
may use one or the other form
(or indeed both).
For example,
.Xr ed 1
uses BREs,
whilst
.Xr egrep 1
talks EREs.
Consult the manual page for the specific application to find out which
it uses.
.Pp
POSIX leaves some aspects of RE syntax and semantics open;
.Sq **
marks decisions on these aspects that
may not be fully portable to other POSIX implementations.
.Pp
This manual page first describes regular expressions in general,
specifically extended regular expressions,
and then discusses differences between them and basic regular expressions.
.Sh EXTENDED REGULAR EXPRESSIONS
An ERE is one** or more non-empty**
.Em branches ,
separated by
.Sq \*(Ba .
It matches anything that matches one of the branches.
.Pp
A branch is one** or more
.Em pieces ,
concatenated.
It matches a match for the first, followed by a match for the second, etc.
.Pp
A piece is an
.Em atom
possibly followed by a single**
.Sq * ,
.Sq + ,
.Sq ?\& ,
or
.Em bound .
An atom followed by
.Sq *
matches a sequence of 0 or more matches of the atom.
An atom followed by
.Sq +
matches a sequence of 1 or more matches of the atom.
An atom followed by
.Sq ?\&
matches a sequence of 0 or 1 matches of the atom.
.Pp
A bound is
.Sq {
followed by an unsigned decimal integer,
possibly followed by
.Sq ,\&
possibly followed by another unsigned decimal integer,
always followed by
.Sq } .
The integers must lie between 0 and
.Dv RE_DUP_MAX
(255**) inclusive,
and if there are two of them, the first may not exceed the second.
An atom followed by a bound containing one integer
.Ar i
and no comma matches
a sequence of exactly
.Ar i
matches of the atom.
An atom followed by a bound
containing one integer
.Ar i
and a comma matches
a sequence of
.Ar i
or more matches of the atom.
An atom followed by a bound
containing two integers
.Ar i
and
.Ar j
matches a sequence of
.Ar i
through
.Ar j
(inclusive) matches of the atom.
.Pp
An atom is a regular expression enclosed in
.Sq ()
(matching a part of the regular expression),
an empty set of
.Sq ()
(matching the null string)**,
a
.Em bracket expression
(see below),
.Sq .\&
(matching any single character),
.Sq ^
(matching the null string at the beginning of a line),
.Sq $
(matching the null string at the end of a line),
a
.Sq \e
followed by one of the characters
.Sq ^.[$()|*+?{\e
(matching that character taken as an ordinary character),
a
.Sq \e
followed by any other character**
(matching that character taken as an ordinary character,
as if the
.Sq \e
had not been present**),
or a single character with no other significance (matching that character).
A
.Sq {
followed by a character other than a digit is an ordinary character,
not the beginning of a bound**.
It is illegal to end an RE with
.Sq \e .
.Pp
A bracket expression is a list of characters enclosed in
.Sq [] .
It normally matches any single character from the list (but see below).
If the list begins with
.Sq ^ ,
it matches any single character
.Em not
from the rest of the list
(but see below).
If two characters in the list are separated by
.Sq - ,
this is shorthand for the full
.Em range
of characters between those two (inclusive) in the
collating sequence, e.g.\&
.Sq [0-9]
in ASCII matches any decimal digit.
It is illegal** for two ranges to share an endpoint, e.g.\&
.Sq a-c-e .
Ranges are very collating-sequence-dependent,
and portable programs should avoid relying on them.
.Pp
To include a literal
.Sq ]\&
in the list, make it the first character
(following a possible
.Sq ^ ) .
To include a literal
.Sq - ,
make it the first or last character,
or the second endpoint of a range.
To use a literal
.Sq -
as the first endpoint of a range,
enclose it in
.Sq [.
and
.Sq .]
to make it a collating element (see below).
With the exception of these and some combinations using
.Sq \&[
(see next paragraphs),
all other special characters, including
.Sq \e ,
lose their special significance within a bracket expression.
.Pp
Within a bracket expression, a collating element
(a character,
a multi-character sequence that collates as if it were a single character,
or a collating-sequence name for either)
enclosed in
.Sq [.
and
.Sq .]
stands for the sequence of characters of that collating element.
The sequence is a single element of the bracket expression's list.
A bracket expression containing a multi-character collating element
can thus match more than one character,
e.g. if the collating sequence includes a
.Sq ch
collating element,
then the RE
.Sq [[.ch.]]*c
matches the first five characters of
.Sq chchcc .
.Pp
Within a bracket expression, a collating element enclosed in
.Sq [=
and
.Sq =]
is an equivalence class, standing for the sequences of characters
of all collating elements equivalent to that one, including itself.
(If there are no other equivalent collating elements,
the treatment is as if the enclosing delimiters were
.Sq [.
and
.Sq .] . )
For example, if
.Sq x
and
.Sq y
are the members of an equivalence class,
then
.Sq [[=x=]] ,
.Sq [[=y=]] ,
and
.Sq [xy]
are all synonymous.
An equivalence class may not** be an endpoint of a range.
.Pp
Within a bracket expression, the name of a
.Em character class
enclosed
in
.Sq [:
and
.Sq :]
stands for the list of all characters belonging to that class.
Standard character class names are:
.Bd -literal -offset indent
alnum digit punct
alpha graph space
blank lower upper
cntrl print xdigit
.Ed
.Pp
These stand for the character classes defined in
.Xr ctype 3 .
A locale may provide others.
A character class may not be used as an endpoint of a range.
.Pp
There are two special cases** of bracket expressions:
the bracket expressions
.Sq [[:<:]]
and
.Sq [[:>:]]
match the null string at the beginning and end of a word, respectively.
A word is defined as a sequence of
characters starting and ending with a word character
which is neither preceded nor followed by
word characters.
A word character is an
.Em alnum
character (as defined by
.Xr ctype 3 )
or an underscore.
This is an extension,
compatible with but not specified by POSIX,
and should be used with
caution in software intended to be portable to other systems.
.Pp
In the event that an RE could match more than one substring of a given
string,
the RE matches the one starting earliest in the string.
If the RE could match more than one substring starting at that point,
it matches the longest.
Subexpressions also match the longest possible substrings, subject to
the constraint that the whole match be as long as possible,
with subexpressions starting earlier in the RE taking priority over
ones starting later.
Note that higher-level subexpressions thus take priority over
their lower-level component subexpressions.
.Pp
Match lengths are measured in characters, not collating elements.
A null string is considered longer than no match at all.
For example,
.Sq bb*
matches the three middle characters of
.Sq abbbc ;
.Sq (wee|week)(knights|nights)
matches all ten characters of
.Sq weeknights ;
when
.Sq (.*).*
is matched against
.Sq abc ,
the parenthesized subexpression matches all three characters;
and when
.Sq (a*)*
is matched against
.Sq bc ,
both the whole RE and the parenthesized subexpression match the null string.
.Pp
If case-independent matching is specified,
the effect is much as if all case distinctions had vanished from the
alphabet.
When an alphabetic that exists in multiple cases appears as an
ordinary character outside a bracket expression, it is effectively
transformed into a bracket expression containing both cases,
e.g.\&
.Sq x
becomes
.Sq [xX] .
When it appears inside a bracket expression,
all case counterparts of it are added to the bracket expression,
so that, for example,
.Sq [x]
becomes
.Sq [xX]
and
.Sq [^x]
becomes
.Sq [^xX] .
.Pp
No particular limit is imposed on the length of REs**.
Programs intended to be portable should not employ REs longer
than 256 bytes,
as an implementation can refuse to accept such REs and remain
POSIX-compliant.
.Pp
The following is a list of extended regular expressions:
.Bl -tag -width Ds
.It Ar c
Any character
.Ar c
not listed below matches itself.
.It \e Ns Ar c
Any backslash-escaped character
.Ar c
matches itself.
.It \&.
Matches any single character that is not a newline
.Pq Sq \en .
.It Bq Ar char-class
Matches any single character in
.Ar char-class .
To include a
.Ql \&]
in
.Ar char-class ,
it must be the first character.
A range of characters may be specified by separating the end characters
of the range with a
.Ql - ;
e.g.\&
.Ar a-z
specifies the lower case characters.
The following literal expressions can also be used in
.Ar char-class
to specify sets of characters:
.Bd -unfilled -offset indent
[:alnum:] [:cntrl:] [:lower:] [:space:]
[:alpha:] [:digit:] [:print:] [:upper:]
[:blank:] [:graph:] [:punct:] [:xdigit:]
.Ed
.Pp
If
.Ql -
appears as the first or last character of
.Ar char-class ,
then it matches itself.
All other characters in
.Ar char-class
match themselves.
.Pp
Patterns in
.Ar char-class
of the form
.Eo [.
.Ar col-elm
.Ec .]\&
or
.Eo [=
.Ar col-elm
.Ec =]\& ,
where
.Ar col-elm
is a collating element, are interpreted according to
.Xr setlocale 3
.Pq not currently supported .
.It Bq ^ Ns Ar char-class
Matches any single character, other than newline, not in
.Ar char-class .
.Ar char-class
is defined as above.
.It ^
If
.Sq ^
is the first character of a regular expression, then it
anchors the regular expression to the beginning of a line.
Otherwise, it matches itself.
.It $
If
.Sq $
is the last character of a regular expression,
it anchors the regular expression to the end of a line.
Otherwise, it matches itself.
.It [[:<:]]
Anchors the single character regular expression or subexpression
immediately following it to the beginning of a word.
.It [[:>:]]
Anchors the single character regular expression or subexpression
immediately following it to the end of a word.
.It Pq Ar re
Defines a subexpression
.Ar re .
Any set of characters enclosed in parentheses
matches whatever the set of characters without parentheses matches
(that is a long-winded way of saying the constructs
.Sq (re)
and
.Sq re
match identically).
.It *
Matches the single character regular expression or subexpression
immediately preceding it zero or more times.
If
.Sq *
is the first character of a regular expression or subexpression,
then it matches itself.
The
.Sq *
operator sometimes yields unexpected results.
For example, the regular expression
.Ar b*
matches the beginning of the string
.Qq abbb
(as opposed to the substring
.Qq bbb ) ,
since a null match is the only leftmost match.
.It +
Matches the singular character regular expression
or subexpression immediately preceding it
one or more times.
.It ?
Matches the singular character regular expression
or subexpression immediately preceding it
0 or 1 times.
.Sm off
.It Xo
.Pf { Ar n , m No }\ \&
.Pf { Ar n , No }\ \&
.Pf { Ar n No }
.Xc
.Sm on
Matches the single character regular expression or subexpression
immediately preceding it at least
.Ar n
and at most
.Ar m
times.
If
.Ar m
is omitted, then it matches at least
.Ar n
times.
If the comma is also omitted, then it matches exactly
.Ar n
times.
.It \*(Ba
Used to separate patterns.
For example,
the pattern
.Sq cat\*(Badog
matches either
.Sq cat
or
.Sq dog .
.El
.Sh BASIC REGULAR EXPRESSIONS
Basic regular expressions differ in several respects:
.Bl -bullet -offset 3n
.It
.Sq \*(Ba ,
.Sq + ,
and
.Sq ?\&
are ordinary characters and there is no equivalent
for their functionality.
.It
The delimiters for bounds are
.Sq \e{
and
.Sq \e} ,
with
.Sq {
and
.Sq }
by themselves ordinary characters.
.It
The parentheses for nested subexpressions are
.Sq \e(
and
.Sq \e) ,
with
.Sq \&(
and
.Sq )\&
by themselves ordinary characters.
.It
.Sq ^
is an ordinary character except at the beginning of the
RE or** the beginning of a parenthesized subexpression.
.It
.Sq $
is an ordinary character except at the end of the
RE or** the end of a parenthesized subexpression.
.It
.Sq *
is an ordinary character if it appears at the beginning of the
RE or the beginning of a parenthesized subexpression
(after a possible leading
.Sq ^ ) .
.It
Finally, there is one new type of atom, a
.Em back-reference :
.Sq \e
followed by a non-zero decimal digit
.Ar d
matches the same sequence of characters matched by the
.Ar d Ns th
parenthesized subexpression
(numbering subexpressions by the positions of their opening parentheses,
left to right),
so that, for example,
.Sq \e([bc]\e)\e1
matches
.Sq bb\&
or
.Sq cc
but not
.Sq bc .
.El
.Pp
The following is a list of basic regular expressions:
.Bl -tag -width Ds
.It Ar c
Any character
.Ar c
not listed below matches itself.
.It \e Ns Ar c
Any backslash-escaped character
.Ar c ,
except for
.Sq { ,
.Sq } ,
.Sq \&( ,
and
.Sq \&) ,
matches itself.
.It \&.
Matches any single character that is not a newline
.Pq Sq \en .
.It Bq Ar char-class
Matches any single character in
.Ar char-class .
To include a
.Ql \&]
in
.Ar char-class ,
it must be the first character.
A range of characters may be specified by separating the end characters
of the range with a
.Ql - ;
e.g.\&
.Ar a-z
specifies the lower case characters.
The following literal expressions can also be used in
.Ar char-class
to specify sets of characters:
.Bd -unfilled -offset indent
[:alnum:] [:cntrl:] [:lower:] [:space:]
[:alpha:] [:digit:] [:print:] [:upper:]
[:blank:] [:graph:] [:punct:] [:xdigit:]
.Ed
.Pp
If
.Ql -
appears as the first or last character of
.Ar char-class ,
then it matches itself.
All other characters in
.Ar char-class
match themselves.
.Pp
Patterns in
.Ar char-class
of the form
.Eo [.
.Ar col-elm
.Ec .]\&
or
.Eo [=
.Ar col-elm
.Ec =]\& ,
where
.Ar col-elm
is a collating element, are interpreted according to
.Xr setlocale 3
.Pq not currently supported .
.It Bq ^ Ns Ar char-class
Matches any single character, other than newline, not in
.Ar char-class .
.Ar char-class
is defined as above.
.It ^
If
.Sq ^
is the first character of a regular expression, then it
anchors the regular expression to the beginning of a line.
Otherwise, it matches itself.
.It $
If
.Sq $
is the last character of a regular expression,
it anchors the regular expression to the end of a line.
Otherwise, it matches itself.
.It [[:<:]]
Anchors the single character regular expression or subexpression
immediately following it to the beginning of a word.
.It [[:>:]]
Anchors the single character regular expression or subexpression
immediately following it to the end of a word.
.It \e( Ns Ar re Ns \e)
Defines a subexpression
.Ar re .
Subexpressions may be nested.
A subsequent backreference of the form
.Pf \e Ns Ar n ,
where
.Ar n
is a number in the range [1,9], expands to the text matched by the
.Ar n Ns th
subexpression.
For example, the regular expression
.Ar \e(.*\e)\e1
matches any string consisting of identical adjacent substrings.
Subexpressions are ordered relative to their left delimiter.
.It *
Matches the single character regular expression or subexpression
immediately preceding it zero or more times.
If
.Sq *
is the first character of a regular expression or subexpression,
then it matches itself.
The
.Sq *
operator sometimes yields unexpected results.
For example, the regular expression
.Ar b*
matches the beginning of the string
.Qq abbb
(as opposed to the substring
.Qq bbb ) ,
since a null match is the only leftmost match.
.Sm off
.It Xo
.Pf \e{ Ar n , m No \e}\ \&
.Pf \e{ Ar n , No \e}\ \&
.Pf \e{ Ar n No \e}
.Xc
.Sm on
Matches the single character regular expression or subexpression
immediately preceding it at least
.Ar n
and at most
.Ar m
times.
If
.Ar m
is omitted, then it matches at least
.Ar n
times.
If the comma is also omitted, then it matches exactly
.Ar n
times.
.El
.Sh SEE ALSO
.Xr ctype 3 ,
.Xr regex 3
.Sh STANDARDS
.St -p1003.1-2004 :
Base Definitions, Chapter 9 (Regular Expressions).
.Sh BUGS
Having two kinds of REs is a botch.
.Pp
The current POSIX spec says that
.Sq )\&
is an ordinary character in the absence of an unmatched
.Sq \&( ;
this was an unintentional result of a wording error,
and change is likely.
Avoid relying on it.
.Pp
Back-references are a dreadful botch,
posing major problems for efficient implementations.
They are also somewhat vaguely defined
(does
.Sq a\e(\e(b\e)*\e2\e)*d
match
.Sq abbbd ? ) .
Avoid using them.
.Pp
POSIX's specification of case-independent matching is vague.
The
.Dq one case implies all cases
definition given above
is the current consensus among implementors as to the right interpretation.
.Pp
The syntax for word boundaries is incredibly ugly.

1566
libr/util/regex/regcomp.c Normal file

File diff suppressed because it is too large Load diff

130
libr/util/regex/regerror.c Normal file
View file

@ -0,0 +1,130 @@
/* $OpenBSD: regerror.c,v 1.13 2005/08/05 13:03:00 espie Exp $ */
/*-
* Copyright (c) 1992, 1993, 1994 Henry Spencer.
* Copyright (c) 1992, 1993, 1994
* The Regents of the University of California. All rights reserved.
*
* This code is derived from software contributed to Berkeley by
* Henry Spencer.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
* 3. Neither the name of the University nor the names of its contributors
* may be used to endorse or promote products derived from this software
* without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*
* @(#)regerror.c 8.4 (Berkeley) 3/20/94
*/
#include <sys/types.h>
#include <stdio.h>
#include <string.h>
#include <ctype.h>
#include <limits.h>
#include <stdlib.h>
#include "r_regex.h"
#include "utils.h"
static char *regatoi(const RRegex*, char *, int);
static struct rerr {
int code;
char *name;
char *explain;
} rerrs[] = {
{ R_REGEX_NOMATCH, "R_REGEX_NOMATCH", "regexec() failed to match" },
{ R_REGEX_BADPAT, "R_REGEX_BADPAT", "invalid regular expression" },
{ R_REGEX_ECOLLATE, "R_REGEX_ECOLLATE", "invalid collating element" },
{ R_REGEX_ECTYPE, "R_REGEX_ECTYPE", "invalid character class" },
{ R_REGEX_EESCAPE, "R_REGEX_EESCAPE", "trailing backslash (\\)" },
{ R_REGEX_ESUBREG, "R_REGEX_ESUBREG", "invalid backreference number" },
{ R_REGEX_EBRACK, "R_REGEX_EBRACK", "brackets ([ ]) not balanced" },
{ R_REGEX_EPAREN, "R_REGEX_EPAREN", "parentheses not balanced" },
{ R_REGEX_EBRACE, "R_REGEX_EBRACE", "braces not balanced" },
{ R_REGEX_BADBR, "R_REGEX_BADBR", "invalid repetition count(s)" },
{ R_REGEX_ERANGE, "R_REGEX_ERANGE", "invalid character range" },
{ R_REGEX_ESPACE, "R_REGEX_ESPACE", "out of memory" },
{ R_REGEX_BADRPT, "R_REGEX_BADRPT", "repetition-operator operand invalid" },
{ R_REGEX_EMPTY, "R_REGEX_EMPTY", "empty (sub)expression" },
{ R_REGEX_ASSERT, "R_REGEX_ASSERT", "\"can't happen\" -- you found a bug" },
{ R_REGEX_INVARG, "R_REGEX_INVARG", "invalid argument to regex routine" },
{ 0, "", "*** unknown regexp error code ***" }
};
/*
- regerror - the interface to error numbers
= extern size_t regerror(int, const regex_t *, char *, size_t);
*/
/* ARGSUSED */
size_t
r_regex_error(int errcode, const RRegex *preg, char *errbuf, size_t errbuf_size)
{
struct rerr *r;
size_t len;
int target = errcode &~ R_REGEX_ITOA;
char *s;
char convbuf[50];
if (errcode == R_REGEX_ATOI)
s = regatoi(preg, convbuf, sizeof convbuf);
else {
for (r = rerrs; r->code != 0; r++)
if (r->code == target)
break;
if (errcode&R_REGEX_ITOA) {
if (r->code != 0) {
assert(strlen(r->name) < sizeof(convbuf));
(void) strlcpy(convbuf, r->name, sizeof convbuf);
} else
(void)snprintf(convbuf, sizeof convbuf,
"R_REGEX_0x%x", target);
s = convbuf;
} else
s = r->explain;
}
len = strlen(s) + 1;
if (errbuf_size > 0) {
strlcpy(errbuf, s, errbuf_size);
}
return(len);
}
/*
- regatoi - internal routine to implement R_REGEX_ATOI
*/
static char *
regatoi(const RRegex *preg, char *localbuf, int localbufsize)
{
struct rerr *r;
for (r = rerrs; r->code != 0; r++)
if (strcmp(r->name, preg->re_endp) == 0)
break;
if (r->code == 0)
return("0");
(void)snprintf(localbuf, localbufsize, "%d", r->code);
return(localbuf);
}

667
libr/util/regex/regex.3 Normal file
View file

@ -0,0 +1,667 @@
.\" $OpenBSD: regex.3,v 1.21 2007/05/31 19:19:30 jmc Exp $
.\"
.\" Copyright (c) 1997, Phillip F Knaack. All rights reserved.
.\"
.\" Copyright (c) 1992, 1993, 1994 Henry Spencer.
.\" Copyright (c) 1992, 1993, 1994
.\" The Regents of the University of California. All rights reserved.
.\"
.\" This code is derived from software contributed to Berkeley by
.\" Henry Spencer.
.\"
.\" Redistribution and use in source and binary forms, with or without
.\" modification, are permitted provided that the following conditions
.\" are met:
.\" 1. Redistributions of source code must retain the above copyright
.\" notice, this list of conditions and the following disclaimer.
.\" 2. Redistributions in binary form must reproduce the above copyright
.\" notice, this list of conditions and the following disclaimer in the
.\" documentation and/or other materials provided with the distribution.
.\" 3. Neither the name of the University nor the names of its contributors
.\" may be used to endorse or promote products derived from this software
.\" without specific prior written permission.
.\"
.\" THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
.\" ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
.\" IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
.\" ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
.\" FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
.\" DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
.\" OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
.\" HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
.\" LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
.\" OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
.\" SUCH DAMAGE.
.\"
.\" @(#)regex.3 8.4 (Berkeley) 3/20/94
.\"
.Dd $Mdocdate: May 31 2007 $
.Dt REGEX 3
.Os
.Sh NAME
.Nm regcomp ,
.Nm regexec ,
.Nm regerror ,
.Nm regfree
.Nd regular expression routines
.Sh SYNOPSIS
.Fd #include <sys/types.h>
.Fd #include <regex.h>
.Ft int
.Fn regcomp "regex_t *preg" "const char *pattern" "int cflags"
.Pp
.Ft int
.Fn regexec "const regex_t *preg" "const char *string" "size_t nmatch" \
"regmatch_t pmatch[]" "int eflags"
.Pp
.Ft size_t
.Fn regerror "int errcode" "const regex_t *preg" "char *errbuf" \
"size_t errbuf_size"
.Pp
.Ft void
.Fn regfree "regex_t *preg"
.Sh DESCRIPTION
These routines implement
.St -p1003.2
regular expressions
.Pq Dq REs ;
see
.Xr re_format 7 .
.Fn regcomp
compiles an RE written as a string into an internal form,
.Fn regexec
matches that internal form against a string and reports results,
.Fn regerror
transforms error codes from either into human-readable messages, and
.Fn regfree
frees any dynamically allocated storage used by the internal form
of an RE.
.Pp
The header
.Aq Pa regex.h
declares two structure types,
.Li regex_t
and
.Li regmatch_t ,
the former for compiled internal forms and the latter for match reporting.
It also declares the four functions,
a type
.Li regoff_t ,
and a number of constants with names starting with
.Dv REG_ .
.Pp
.Fn regcomp
compiles the regular expression contained in the
.Fa pattern
string,
subject to the flags in
.Fa cflags ,
and places the results in the
.Li regex_t
structure pointed to by
.Fa preg .
.Fa cflags
is the bitwise
.Tn OR
of zero or more of the following flags:
.Bl -tag -width XREG_EXTENDEDX
.It Dv REG_EXTENDED
Compile modern
.Pq Dq extended
REs,
rather than the obsolete
.Pq Dq basic
REs that are the default.
.It Dv REG_BASIC
This is a synonym for 0,
provided as a counterpart to
.Dv REG_EXTENDED
to improve readability.
.It Dv REG_NOSPEC
Compile with recognition of all special characters turned off.
All characters are thus considered ordinary,
so the RE is a literal string.
This is an extension,
compatible with but not specified by
.St -p1003.2 ,
and should be used with
caution in software intended to be portable to other systems.
.Dv REG_EXTENDED
and
.Dv REG_NOSPEC
may not be used in the same call to
.Fn regcomp .
.It Dv REG_ICASE
Compile for matching that ignores upper/lower case distinctions.
See
.Xr re_format 7 .
.It Dv REG_NOSUB
Compile for matching that need only report success or failure,
not what was matched.
.It Dv REG_NEWLINE
Compile for newline-sensitive matching.
By default, newline is a completely ordinary character with no special
meaning in either REs or strings.
With this flag,
.Ql \&[^
bracket expressions and
.Ql \&.
never match newline,
a
.Ql ^
anchor matches the null string after any newline in the string
in addition to its normal function,
and the
.Ql $
anchor matches the null string before any newline in the
string in addition to its normal function.
.It Dv REG_PEND
The regular expression ends,
not at the first NUL,
but just before the character pointed to by the
.Fa re_endp
member of the structure pointed to by
.Fa preg .
The
.Fa re_endp
member is of type
.Fa const\ char\ * .
This flag permits inclusion of NULs in the RE;
they are considered ordinary characters.
This is an extension,
compatible with but not specified by
.St -p1003.2 ,
and should be used with
caution in software intended to be portable to other systems.
.El
.Pp
When successful,
.Fn regcomp
returns 0 and fills in the structure pointed to by
.Fa preg .
One member of that structure
(other than
.Fa re_endp )
is publicized:
.Fa re_nsub ,
of type
.Fa size_t ,
contains the number of parenthesized subexpressions within the RE
(except that the value of this member is undefined if the
.Dv REG_NOSUB
flag was used).
If
.Fn regcomp
fails, it returns a non-zero error code;
see DIAGNOSTICS.
.Pp
.Fn regexec
matches the compiled RE pointed to by
.Fa preg
against the
.Fa string ,
subject to the flags in
.Fa eflags ,
and reports results using
.Fa nmatch ,
.Fa pmatch ,
and the returned value.
The RE must have been compiled by a previous invocation of
.Fn regcomp .
The compiled form is not altered during execution of
.Fn regexec ,
so a single compiled RE can be used simultaneously by multiple threads.
.Pp
By default,
the NUL-terminated string pointed to by
.Fa string
is considered to be the text of an entire line, minus any terminating
newline.
The
.Fa eflags
argument is the bitwise
.Tn OR
of zero or more of the following flags:
.Bl -tag -width XREG_STARTENDX
.It Dv REG_NOTBOL
The first character of
the string
is not the beginning of a line, so the
.Ql ^
anchor should not match before it.
This does not affect the behavior of newlines under
.Dv REG_NEWLINE .
.It Dv REG_NOTEOL
The NUL terminating
the string
does not end a line, so the
.Ql $
anchor should not match before it.
This does not affect the behavior of newlines under
.Dv REG_NEWLINE .
.It Dv REG_STARTEND
The string is considered to start at
\fIstring\fR\ + \fIpmatch\fR[0].\fIrm_so\fR
and to have a terminating NUL located at
\fIstring\fR\ + \fIpmatch\fR[0].\fIrm_eo\fR
(there need not actually be a NUL at that location),
regardless of the value of
.Fa nmatch .
See below for the definition of
.Fa pmatch
and
.Fa nmatch .
This is an extension,
compatible with but not specified by
.St -p1003.2 ,
and should be used with
caution in software intended to be portable to other systems.
Note that a non-zero \fIrm_so\fR does not imply
.Dv REG_NOTBOL ;
.Dv REG_STARTEND
affects only the location of the string,
not how it is matched.
.El
.Pp
See
.Xr re_format 7
for a discussion of what is matched in situations where an RE or a
portion thereof could match any of several substrings of
.Fa string .
.Pp
Normally,
.Fn regexec
returns 0 for success and the non-zero code
.Dv REG_NOMATCH
for failure.
Other non-zero error codes may be returned in exceptional situations;
see DIAGNOSTICS.
.Pp
If
.Dv REG_NOSUB
was specified in the compilation of the RE,
or if
.Fa nmatch
is 0,
.Fn regexec
ignores the
.Fa pmatch
argument (but see below for the case where
.Dv REG_STARTEND
is specified).
Otherwise,
.Fa pmatch
points to an array of
.Fa nmatch
structures of type
.Li regmatch_t .
Such a structure has at least the members
.Fa rm_so
and
.Fa rm_eo ,
both of type
.Fa regoff_t
(a signed arithmetic type at least as large as an
.Li off_t
and a
.Li ssize_t ) ,
containing respectively the offset of the first character of a substring
and the offset of the first character after the end of the substring.
Offsets are measured from the beginning of the
.Fa string
argument given to
.Fn regexec .
An empty substring is denoted by equal offsets,
both indicating the character following the empty substring.
.Pp
The 0th member of the
.Fa pmatch
array is filled in to indicate what substring of
.Fa string
was matched by the entire RE.
Remaining members report what substring was matched by parenthesized
subexpressions within the RE;
member
.Va i
reports subexpression
.Va i ,
with subexpressions counted (starting at 1) by the order of their opening
parentheses in the RE, left to right.
Unused entries in the array\(emcorresponding either to subexpressions that
did not participate in the match at all, or to subexpressions that do not
exist in the RE (that is, \fIi\fR\ > \fIpreg\fR\->\fIre_nsub\fR)\(emhave both
.Fa rm_so
and
.Fa rm_eo
set to \-1.
If a subexpression participated in the match several times,
the reported substring is the last one it matched.
(Note, as an example in particular, that when the RE
.Dq (b*)+
matches
.Dq bbb ,
the parenthesized subexpression matches each of the three
.Sq b Ns s
and then
an infinite number of empty strings following the last
.Sq b ,
so the reported substring is one of the empties.)
.Pp
If
.Dv REG_STARTEND
is specified,
.Fa pmatch
must point to at least one
.Li regmatch_t
(even if
.Fa nmatch
is 0 or
.Dv REG_NOSUB
was specified),
to hold the input offsets for
.Dv REG_STARTEND .
Use for output is still entirely controlled by
.Fa nmatch ;
if
.Fa nmatch
is 0 or
.Dv REG_NOSUB
was specified,
the value of
.Fa pmatch[0]
will not be changed by a successful
.Fn regexec .
.Pp
.Fn regerror
maps a non-zero
.Va errcode
from either
.Fn regcomp
or
.Fn regexec
to a human-readable, printable message.
If
.Fa preg
is non-NULL,
the error code should have arisen from use of
the
.Li regex_t
pointed to by
.Fa preg ,
and if the error code came from
.Fn regcomp ,
it should have been the result from the most recent
.Fn regcomp
using that
.Li regex_t .
.Pf ( Fn regerror
may be able to supply a more detailed message using information
from the
.Li regex_t . )
.Fn regerror
places the NUL-terminated message into the buffer pointed to by
.Fa errbuf ,
limiting the length (including the NUL) to at most
.Fa errbuf_size
bytes.
If the whole message won't fit,
as much of it as will fit before the terminating NUL is supplied.
In any case,
the returned value is the size of buffer needed to hold the whole
message (including the terminating NUL).
If
.Fa errbuf_size
is 0,
.Fa errbuf
is ignored but the return value is still correct.
.Pp
If the
.Fa errcode
given to
.Fn regerror
is first
.Tn OR Ns 'ed
with
.Dv REG_ITOA ,
the
.Dq message
that results is the printable name of the error code,
e.g.,
.Dq REG_NOMATCH ,
rather than an explanation thereof.
If
.Fa errcode
is
.Dv REG_ATOI ,
then
.Fa preg
shall be non-null and the
.Fa re_endp
member of the structure it points to
must point to the printable name of an error code;
in this case, the result in
.Fa errbuf
is the decimal digits of
the numeric value of the error code
(0 if the name is not recognized).
.Dv REG_ITOA
and
.Dv REG_ATOI
are intended primarily as debugging facilities;
they are extensions,
compatible with but not specified by
.St -p1003.2
and should be used with
caution in software intended to be portable to other systems.
Be warned also that they are considered experimental and changes are possible.
.Pp
.Fn regfree
frees any dynamically allocated storage associated with the compiled RE
pointed to by
.Fa preg .
The remaining
.Li regex_t
is no longer a valid compiled RE
and the effect of supplying it to
.Fn regexec
or
.Fn regerror
is undefined.
.Pp
None of these functions references global variables except for tables
of constants;
all are safe for use from multiple threads if the arguments are safe.
.Sh IMPLEMENTATION CHOICES
There are a number of decisions that
.St -p1003.2
leaves up to the implementor,
either by explicitly saying
.Dq undefined
or by virtue of them being
forbidden by the RE grammar.
This implementation treats them as follows.
.Pp
See
.Xr re_format 7
for a discussion of the definition of case-independent matching.
.Pp
There is no particular limit on the length of REs,
except insofar as memory is limited.
Memory usage is approximately linear in RE size, and largely insensitive
to RE complexity, except for bounded repetitions.
See
.Sx BUGS
for one short RE using them
that will run almost any system out of memory.
.Pp
A backslashed character other than one specifically given a magic meaning
by
.St -p1003.2
(such magic meanings occur only in obsolete REs)
is taken as an ordinary character.
.Pp
Any unmatched
.Ql \&[
is a
.Dv REG_EBRACK
error.
.Pp
Equivalence classes cannot begin or end bracket-expression ranges.
The endpoint of one range cannot begin another.
.Pp
RE_DUP_MAX, the limit on repetition counts in bounded repetitions, is 255.
.Pp
A repetition operator (?, *, +, or bounds) cannot follow another
repetition operator.
A repetition operator cannot begin an expression or subexpression
or follow
.Ql ^
or
.Ql | .
.Pp
A
.Ql |
cannot appear first or last in a (sub)expression, or after another
.Ql | ,
i.e., an operand of
.Ql |
cannot be an empty subexpression.
An empty parenthesized subexpression,
.Ql \&(\&) ,
is legal and matches an
empty (sub)string.
An empty string is not a legal RE.
.Pp
A
.Ql {
followed by a digit is considered the beginning of bounds for a
bounded repetition, which must then follow the syntax for bounds.
A
.Ql {
.Em not
followed by a digit is considered an ordinary character.
.Pp
.Ql ^
and
.Ql $
beginning and ending subexpressions in obsolete
.Pq Dq basic
REs are anchors, not ordinary characters.
.Sh DIAGNOSTICS
Non-zero error codes from
.Fn regcomp
and
.Fn regexec
include the following:
.Pp
.Bl -tag -compact -width XREG_ECOLLATEX
.It Er REG_NOMATCH
regexec() failed to match
.It Er REG_BADPAT
invalid regular expression
.It Er REG_ECOLLATE
invalid collating element
.It Er REG_ECTYPE
invalid character class
.It Er REG_EESCAPE
\e applied to unescapable character
.It Er REG_ESUBREG
invalid backreference number
.It Er REG_EBRACK
brackets [ ] not balanced
.It Er REG_EPAREN
parentheses ( ) not balanced
.It Er REG_EBRACE
braces { } not balanced
.It Er REG_BADBR
invalid repetition count(s) in { }
.It Er REG_ERANGE
invalid character range in [ ]
.It Er REG_ESPACE
ran out of memory
.It Er REG_BADRPT
?, *, or + operand invalid
.It Er REG_EMPTY
empty (sub)expression
.It Er REG_ASSERT
.Dq can't happen
\(emyou found a bug
.It Er REG_INVARG
invalid argument, e.g., negative-length string
.El
.Sh SEE ALSO
.Xr grep 1 ,
.Xr re_format 7
.Pp
.St -p1003.2 ,
sections 2.8 (Regular Expression Notation)
and
B.5 (C Binding for Regular Expression Matching).
.Sh HISTORY
Originally written by Henry Spencer.
Altered for inclusion in the
.Bx 4.4
distribution.
.Sh BUGS
This is an alpha release with known defects.
Please report problems.
.Pp
There is one known functionality bug.
The implementation of internationalization is incomplete:
the locale is always assumed to be the default one of
.St -p1003.2 ,
and only the collating elements etc. of that locale are available.
.Pp
The back-reference code is subtle and doubts linger about its correctness
in complex cases.
.Pp
.Fn regexec
performance is poor.
This will improve with later releases.
.Fa nmatch
exceeding 0 is expensive;
.Fa nmatch
exceeding 1 is worse.
.Fn regexec
is largely insensitive to RE complexity
.Em except
that back references are massively expensive.
RE length does matter; in particular, there is a strong speed bonus
for keeping RE length under about 30 characters,
with most special characters counting roughly double.
.Pp
.Fn regcomp
implements bounded repetitions by macro expansion,
which is costly in time and space if counts are large
or bounded repetitions are nested.
A RE like, say,
.Dq ((((a{1,100}){1,100}){1,100}){1,100}){1,100}
will (eventually) run almost any existing machine out of swap space.
.Pp
There are suspected problems with response to obscure error conditions.
Notably,
certain kinds of internal overflow,
produced only by truly enormous REs or by multiply nested bounded repetitions,
are probably not handled well.
.Pp
Due to a mistake in
.St -p1003.2 ,
things like
.Ql a)b
are legal REs because
.Ql \&)
is
a special character only in the presence of a previous unmatched
.Ql \&( .
This can't be fixed until the spec is fixed.
.Pp
The standard's definition of back references is vague.
For example, does
.Dq a\e(\e(b\e)*\e2\e)*d
match
.Dq abbbd ?
Until the standard is clarified,
behavior in such cases should not be relied on.
.Pp
The implementation of word-boundary matching is a bit of a kludge,
and bugs may lurk in combinations of word-boundary matching and anchoring.

158
libr/util/regex/regex2.h Normal file
View file

@ -0,0 +1,158 @@
/* $OpenBSD: regex2.h,v 1.7 2004/11/30 17:04:23 otto Exp $ */
/*-
* Copyright (c) 1992, 1993, 1994 Henry Spencer.
* Copyright (c) 1992, 1993, 1994
* The Regents of the University of California. All rights reserved.
*
* This code is derived from software contributed to Berkeley by
* Henry Spencer.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
* 3. Neither the name of the University nor the names of its contributors
* may be used to endorse or promote products derived from this software
* without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*
* @(#)regex2.h 8.4 (Berkeley) 3/20/94
*/
/*
* internals of regex_t
*/
#define MAGIC1 ((('r'^0200)<<8) | 'e')
/*
* The internal representation is a *strip*, a sequence of
* operators ending with an endmarker. (Some terminology etc. is a
* historical relic of earlier versions which used multiple strips.)
* Certain oddities in the representation are there to permit running
* the machinery backwards; in particular, any deviation from sequential
* flow must be marked at both its source and its destination. Some
* fine points:
*
* - OPLUS_ and O_PLUS are *inside* the loop they create.
* - OQUEST_ and O_QUEST are *outside* the bypass they create.
* - OCH_ and O_CH are *outside* the multi-way branch they create, while
* OOR1 and OOR2 are respectively the end and the beginning of one of
* the branches. Note that there is an implicit OOR2 following OCH_
* and an implicit OOR1 preceding O_CH.
*
* In state representations, an operator's bit is on to signify a state
* immediately *preceding* "execution" of that operator.
*/
typedef unsigned long sop; /* strip operator */
typedef long sopno;
#define OPRMASK 0xf8000000LU
#define OPDMASK 0x07ffffffLU
#define OPSHIFT ((unsigned)27)
#define OP(n) ((n)&OPRMASK)
#define OPND(n) ((n)&OPDMASK)
#define SOP(op, opnd) ((op)|(opnd))
/* operators meaning operand */
/* (back, fwd are offsets) */
#define OEND (1LU<<OPSHIFT) /* endmarker - */
#define OCHAR (2LU<<OPSHIFT) /* character unsigned char */
#define OBOL (3LU<<OPSHIFT) /* left anchor - */
#define OEOL (4LU<<OPSHIFT) /* right anchor - */
#define OANY (5LU<<OPSHIFT) /* . - */
#define OANYOF (6LU<<OPSHIFT) /* [...] set number */
#define OBACK_ (7LU<<OPSHIFT) /* begin \d paren number */
#define O_BACK (8LU<<OPSHIFT) /* end \d paren number */
#define OPLUS_ (9LU<<OPSHIFT) /* + prefix fwd to suffix */
#define O_PLUS (10LU<<OPSHIFT) /* + suffix back to prefix */
#define OQUEST_ (11LU<<OPSHIFT) /* ? prefix fwd to suffix */
#define O_QUEST (12LU<<OPSHIFT) /* ? suffix back to prefix */
#define OLPAREN (13LU<<OPSHIFT) /* ( fwd to ) */
#define ORPAREN (14LU<<OPSHIFT) /* ) back to ( */
#define OCH_ (15LU<<OPSHIFT) /* begin choice fwd to OOR2 */
#define OOR1 (16LU<<OPSHIFT) /* | pt. 1 back to OOR1 or OCH_ */
#define OOR2 (17LU<<OPSHIFT) /* | pt. 2 fwd to OOR2 or O_CH */
#define O_CH (18LU<<OPSHIFT) /* end choice back to OOR1 */
#define OBOW (19LU<<OPSHIFT) /* begin word - */
#define OEOW (20LU<<OPSHIFT) /* end word - */
/*
* Structure for [] character-set representation. Character sets are
* done as bit vectors, grouped 8 to a byte vector for compactness.
* The individual set therefore has both a pointer to the byte vector
* and a mask to pick out the relevant bit of each byte. A hash code
* simplifies testing whether two sets could be identical.
*
* This will get trickier for multicharacter collating elements. As
* preliminary hooks for dealing with such things, we also carry along
* a string of multi-character elements, and decide the size of the
* vectors at run time.
*/
typedef struct {
ut8 *ptr; /* -> ut8 [csetsize] */
ut8 mask; /* bit within array */
ut8 hash; /* hash code */
size_t smultis;
char *multis; /* -> char[smulti] ab\0cd\0ef\0\0 */
} cset;
/* note that CHadd and CHsub are unsafe, and CHIN doesn't yield 0/1 */
#define CHadd(cs, c) ((cs)->ptr[(ut8)(c)] |= (cs)->mask, (cs)->hash += (c))
#define CHsub(cs, c) ((cs)->ptr[(ut8)(c)] &= ~(cs)->mask, (cs)->hash -= (c))
#define CHIN(cs, c) ((cs)->ptr[(ut8)(c)] & (cs)->mask)
#define MCadd(p, cs, cp) mcadd(p, cs, cp) /* regcomp() internal fns */
#define MCsub(p, cs, cp) mcsub(p, cs, cp)
#define MCin(p, cs, cp) mcin(p, cs, cp)
/* stuff for character categories */
typedef unsigned char cat_t;
/*
* main compiled-expression structure
*/
struct re_guts {
int magic;
# define MAGIC2 ((('R'^0200)<<8)|'E')
sop *strip; /* malloced area for strip */
int csetsize; /* number of bits in a cset vector */
int ncsets; /* number of csets in use */
cset *sets; /* -> cset [ncsets] */
ut8 *setbits; /* -> ut8[csetsize][ncsets/CHAR_BIT] */
int cflags; /* copy of regcomp() cflags argument */
sopno nstates; /* = number of sops */
sopno firststate; /* the initial OEND (normally 0) */
sopno laststate; /* the final OEND */
int iflags; /* internal flags */
# define USEBOL 01 /* used ^ */
# define USEEOL 02 /* used $ */
# define BAD 04 /* something wrong */
int nbol; /* number of ^ used */
int neol; /* number of $ used */
int ncategories; /* how many character categories */
cat_t *categories; /* ->catspace[-CHAR_MIN] */
char *must; /* match must contain this string */
int mlen; /* length of must */
size_t nsub; /* copy of re_nsub */
int backrefs; /* does it use back references? */
sopno nplus; /* how deep does it nest +s? */
/* catspace must be last */
cat_t catspace[1]; /* actually [NC] */
};
/* misc utilities */
#undef OUT
#define OUT (CHAR_MAX+1) /* a non-character value */
#define ISWORD(c) (isalnum(c) || (c) == '_')

160
libr/util/regex/regexec.c Normal file
View file

@ -0,0 +1,160 @@
/* $OpenBSD: regexec.c,v 1.11 2005/08/05 13:03:00 espie Exp $ */
/*-
* Copyright (c) 1992, 1993, 1994 Henry Spencer.
* Copyright (c) 1992, 1993, 1994
* The Regents of the University of California. All rights reserved.
*
* This code is derived from software contributed to Berkeley by
* Henry Spencer.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
* 3. Neither the name of the University nor the names of its contributors
* may be used to endorse or promote products derived from this software
* without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*
* @(#)regexec.c 8.3 (Berkeley) 3/20/94
*/
/*
* the outer shell of regexec()
*
* This file includes engine.c *twice*, after muchos fiddling with the
* macros that code uses. This lets the same code operate on two different
* representations for state sets.
*/
#include <sys/types.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <limits.h>
#include <ctype.h>
#include <r_regex.h>
#include "utils.h"
#include "regex2.h"
/* macros for manipulating states, small version */
#define states long
#define states1 states /* for later use in regexec() decision */
#define CLEAR(v) ((v) = 0)
#define SET0(v, n) ((v) &= ~((unsigned long)1 << (n)))
#define SET1(v, n) ((v) |= (unsigned long)1 << (n))
#define ISSET(v, n) (((v) & ((unsigned long)1 << (n))) != 0)
#define ASSIGN(d, s) ((d) = (s))
#define EQ(a, b) ((a) == (b))
#define STATEVARS long dummy /* dummy version */
#define STATESETUP(m, n) /* nothing */
#define STATETEARDOWN(m) /* nothing */
#define SETUP(v) ((v) = 0)
#define onestate long
#define INIT(o, n) ((o) = (unsigned long)1 << (n))
#define INC(o) ((o) <<= 1)
#define ISSTATEIN(v, o) (((v) & (o)) != 0)
/* some abbreviations; note that some of these know variable names! */
/* do "if I'm here, I can also be there" etc without branches */
#define FWD(dst, src, n) ((dst) |= ((unsigned long)(src)&(here)) << (n))
#define BACK(dst, src, n) ((dst) |= ((unsigned long)(src)&(here)) >> (n))
#define ISSETBACK(v, n) (((v) & ((unsigned long)here >> (n))) != 0)
/* function names */
#define SNAMES /* engine.c looks after details */
#include "engine.c"
/* now undo things */
#undef states
#undef CLEAR
#undef SET0
#undef SET1
#undef ISSET
#undef ASSIGN
#undef EQ
#undef STATEVARS
#undef STATESETUP
#undef STATETEARDOWN
#undef SETUP
#undef onestate
#undef INIT
#undef INC
#undef ISSTATEIN
#undef FWD
#undef BACK
#undef ISSETBACK
#undef SNAMES
/* macros for manipulating states, large version */
#define states char *
#define CLEAR(v) memset(v, 0, m->g->nstates)
#define SET0(v, n) ((v)[n] = 0)
#define SET1(v, n) ((v)[n] = 1)
#define ISSET(v, n) ((v)[n])
#define ASSIGN(d, s) memcpy(d, s, m->g->nstates)
#define EQ(a, b) (memcmp(a, b, m->g->nstates) == 0)
#define STATEVARS long vn; char *space
#define STATESETUP(m, nv) { (m)->space = malloc((nv)*(m)->g->nstates); \
if ((m)->space == NULL) return(R_REGEX_ESPACE); \
(m)->vn = 0; }
#define STATETEARDOWN(m) { free((m)->space); }
#define SETUP(v) ((v) = &m->space[m->vn++ * m->g->nstates])
#define onestate long
#define INIT(o, n) ((o) = (n))
#define INC(o) ((o)++)
#define ISSTATEIN(v, o) ((v)[o])
/* some abbreviations; note that some of these know variable names! */
/* do "if I'm here, I can also be there" etc without branches */
#define FWD(dst, src, n) ((dst)[here+(n)] |= (src)[here])
#define BACK(dst, src, n) ((dst)[here-(n)] |= (src)[here])
#define ISSETBACK(v, n) ((v)[here - (n)])
/* function names */
#define LNAMES /* flag */
#include "engine.c"
/*
- regexec - interface for matching
*
* We put this here so we can exploit knowledge of the state representation
* when choosing which matcher to call. Also, by this point the matchers
* have been prototyped.
*/
int /* 0 success, R_REGEX_NOMATCH failure */
r_regex_exec(const RRegex *preg, const char *string, size_t nmatch,
RRegexMatch pmatch[], int eflags)
{
struct re_guts *g = preg->re_g;
#ifdef REDEBUG
# define GOODFLAGS(f) (f)
#else
# define GOODFLAGS(f) ((f)&(R_REGEX_NOTBOL|R_REGEX_NOTEOL|R_REGEX_STARTEND))
#endif
if (preg->re_magic != MAGIC1 || g->magic != MAGIC2)
return(R_REGEX_BADPAT);
assert(!(g->iflags&BAD));
if (g->iflags&BAD) /* backstop for no-debug case */
return(R_REGEX_BADPAT);
eflags = GOODFLAGS(eflags);
if (g->nstates <= CHAR_BIT*sizeof(states1) && !(eflags&R_REGEX_LARGE))
return(smatcher(g, (char *)string, nmatch, pmatch, eflags));
else
return(lmatcher(g, (char *)string, nmatch, pmatch, eflags));
}

31
libr/util/regex/test.c Normal file
View file

@ -0,0 +1,31 @@
#include <stdio.h>
#include <r_regex.h>
int _main() {
RRegex rx;
int rc = r_regex_comp (&rx, "^hi", R_REGEX_NOSUB);
if (rc) {
printf ("error\n");
} else {
rc = r_regex_exec (&rx, "patata", 0, 0, 0);
printf ("out = %d\n", rc);
rc = r_regex_exec (&rx, "hillow", 0, 0, 0);
printf ("out = %d\n", rc);
}
r_regex_free (&rx);
return 0;
}
int main() {
RRegex *rx = r_regex_new ("^hi", "");
if (rx) {
int res = r_regex_exec (rx, "patata", 0, 0, 0);
printf ("result (patata) = %d\n", res);
res = r_regex_exec (rx, "hillow", 0, 0, 0);
printf ("result (hillow) = %d\n", res);
r_regex_free (rx);
} else printf ("oops, cannot compile regexp\n");
return 0;
}

58
libr/util/regex/utils.h Normal file
View file

@ -0,0 +1,58 @@
/* $OpenBSD: utils.h,v 1.4 2003/06/02 20:18:36 millert Exp $ */
/*-
* Copyright (c) 1992, 1993, 1994 Henry Spencer.
* Copyright (c) 1992, 1993, 1994
* The Regents of the University of California. All rights reserved.
*
* This code is derived from software contributed to Berkeley by
* Henry Spencer.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
* 3. Neither the name of the University nor the names of its contributors
* may be used to endorse or promote products derived from this software
* without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*
* @(#)utils.h 8.3 (Berkeley) 3/20/94
*/
/* utility definitions */
#define DUPMAX 255
//_POSIX2_RE_DUP_MAX /* xxx is this right? */
#define INFINITY (DUPMAX + 1)
#define NC (CHAR_MAX - CHAR_MIN + 1)
#define strlcpy(x,y,z) strncpy((x),(y),(z));(x)[(z)]=0;
/* switch off assertions (if not already off) if no REDEBUG */
#ifndef REDEBUG
#ifndef NDEBUG
#define NDEBUG /* no assertions please */
#endif
#endif
#include <assert.h>
/* for old systems with bcopy() but no memmove() */
#ifdef USEBCOPY
#define memmove(d, s, c) bcopy(s, d, c)
#endif
#define ut8 unsigned char

View file

@ -10,3 +10,9 @@ stat-todo:
stat-make:
make 2>&1 | perl mk/stat-make.pl
stat-commiters:
@hg log -r tip:0|grep user:|cut -d @ -f 1 |cut -d '<' -f 1|sed -e 's,",,'|sed -e 's,\ *$$,,'|tr 'A-Z' 'a-z' | sort|uniq -c |sort -n|tail -r
stat-release:
@hg log -r 0.6:0.7|grep user:|cut -d @ -f 1 |cut -d '<' -f 1|sed -e 's,",,'|sed -e 's,\ *$$,,'|tr 'A-Z' 'a-z' | sort|uniq -c |sort -n|tail -r

12
sys/maemo.sh Normal file
View file

@ -0,0 +1,12 @@
#!/bin/sh
mad list >/dev/null 2>&1
if [ $? = 0 ]; then
make clean
echo './configure --without-ssl --prefix=/usr --with-little-endian' | mad sh
echo make | mad sh
cd maemo
make
else
echo "Cannot find 'mad'. Please install QtSDK or QtCreator"
exit 1
fi

View file

@ -5,7 +5,8 @@ if [ -x /usr/bin/pacman ]; then
./configure --without-gmp --with-compiler=i486-mingw32-gcc --with-ostype=windows --host=i486-unknown-windows --without-ssl && \
make -j 4 && \
make w32dist
elif [ `uname`= Darwin ]; then
elif [ `uname` = Darwin ]; then
make clean
./configure --without-gmp --with-compiler=i386-mingw32-gcc --with-ostype=windows --host=i386-unknown-windows --without-ssl && \
make -j 4 && \
make w32dist