Commits February 2021

commits@lists.geany.org

1 participants
28 discussions

[geany/geany] 7b7f0d: PO:(uk) Update translation
by nomadbyte 13 Feb '21

13 Feb '21

Branch: refs/heads/master Author: nomadbyte <nomadbyte(a)users.noreply.github.com> Committer: nomadbyte <nomadbyte(a)users.noreply.github.com> Date: Thu, 21 Jan 2021 23:33:37 UTC Commit: 7b7f0def021715a44ac1e8c3938becb28270b71f https://github.com/geany/geany/commit/7b7f0def021715a44ac1e8c3938becb28270b… Log Message: ----------- PO:(uk) Update translation Modified Paths: -------------- po/uk.po src/about.c Modified: po/uk.po 3865 lines changed, 1392 insertions(+), 2473 deletions(-) =================================================================== No diff available, check online Modified: src/about.c 5 lines changed, 3 insertions(+), 2 deletions(-) =================================================================== @@ -87,7 +87,7 @@ static const gchar *translators[][2] = { { "sv", "Tony Mattsson <superxorn(a)gmail.com>" }, { "sr", "Nikola Radovanovic <cobisimo(a)gmail.com>"}, { "tr", "Gürkan Gür <seqizz(a)gmail.com>"}, - { "uk", "Boris Dibrov <dibrov.bor(a)gmail.com>" }, + { "uk", "Artur Shepilko <nomadbyte(a)gmail.com>" }, { "vi_VN", "Clytie Siddall <clytie(a)riverland.net.au>" }, { "zh_CN", "Dormouse Young <mouselinux(a)163.com>,\nXhacker Liu <liu.dongyuan(a)gmail.com>" }, { "zh_TW", "KoViCH <kovich.ian(a)gmail.com>\nWei-Lun Chao <chaoweilun(a)gmail.com>" } @@ -98,7 +98,8 @@ static const gchar *prev_translators[][2] = { { "es", "Damián Viano <debian(a)damianv.com.ar>\nNacho Cabanes <ncabanes(a)gmail.com>" }, { "pl", "Jacek Wolszczak <shutdownrunner(a)o2.xn--pl>\njarosaw-ctc Foksa <jfoksa(a)gmail.com>" }, { "nl", "Kurt De Bree <kdebree(a)telenet.be>" }, - { "sk", "Tomáš Vadina <kyberdev(a)gmail.com>" } + { "sk", "Tomáš Vadina <kyberdev(a)gmail.com>" }, + { "uk", "Boris Dibrov <dibrov.bor(a)gmail.com>" } }; static const guint prev_translators_len = G_N_ELEMENTS(prev_translators); -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

[geany/geany] 776305: Merge pull request #2724 from nomadbyte/po/uk
by Frank Lanitz 13 Feb '21

13 Feb '21

Branch: refs/heads/master Author: Frank Lanitz <frank(a)frank.uvena.de> Committer: GitHub <noreply(a)github.com> Date: Sat, 13 Feb 2021 20:50:46 UTC Commit: 77630564ad5446df89e9973f74bd378a84fc817e https://github.com/geany/geany/commit/77630564ad5446df89e9973f74bd378a84fc8… Log Message: ----------- Merge pull request #2724 from nomadbyte/po/uk PO:(uk) Update translation Modified Paths: -------------- po/uk.po src/about.c Modified: po/uk.po 3865 lines changed, 1392 insertions(+), 2473 deletions(-) =================================================================== No diff available, check online Modified: src/about.c 5 lines changed, 3 insertions(+), 2 deletions(-) =================================================================== @@ -87,7 +87,7 @@ static const gchar *translators[][2] = { { "sv", "Tony Mattsson <superxorn(a)gmail.com>" }, { "sr", "Nikola Radovanovic <cobisimo(a)gmail.com>"}, { "tr", "Gürkan Gür <seqizz(a)gmail.com>"}, - { "uk", "Boris Dibrov <dibrov.bor(a)gmail.com>" }, + { "uk", "Artur Shepilko <nomadbyte(a)gmail.com>" }, { "vi_VN", "Clytie Siddall <clytie(a)riverland.net.au>" }, { "zh_CN", "Dormouse Young <mouselinux(a)163.com>,\nXhacker Liu <liu.dongyuan(a)gmail.com>" }, { "zh_TW", "KoViCH <kovich.ian(a)gmail.com>\nWei-Lun Chao <chaoweilun(a)gmail.com>" } @@ -98,7 +98,8 @@ static const gchar *prev_translators[][2] = { { "es", "Damián Viano <debian(a)damianv.com.ar>\nNacho Cabanes <ncabanes(a)gmail.com>" }, { "pl", "Jacek Wolszczak <shutdownrunner(a)o2.xn--pl>\njarosaw-ctc Foksa <jfoksa(a)gmail.com>" }, { "nl", "Kurt De Bree <kdebree(a)telenet.be>" }, - { "sk", "Tomáš Vadina <kyberdev(a)gmail.com>" } + { "sk", "Tomáš Vadina <kyberdev(a)gmail.com>" }, + { "uk", "Boris Dibrov <dibrov.bor(a)gmail.com>" } }; static const guint prev_translators_len = G_N_ELEMENTS(prev_translators); -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

[geany/geany] e027e2: Merge pull request #2747 from geany/startup-speed
by Colomban Wendling 07 Feb '21

07 Feb '21

Branch: refs/heads/master Author: Colomban Wendling <ban(a)herbesfolles.org> Committer: Colomban Wendling <ban(a)herbesfolles.org> Date: Sun, 07 Feb 2021 21:37:55 UTC Commit: e027e240c279040c2f6bc9ea0aaccc2cac2ca94e https://github.com/geany/geany/commit/e027e240c279040c2f6bc9ea0aaccc2cac2ca… Log Message: ----------- Merge pull request #2747 from geany/startup-speed Fix startup speed Modified Paths: -------------- src/editor.c src/sciwrappers.c Modified: src/editor.c 18 lines changed, 13 insertions(+), 5 deletions(-) =================================================================== @@ -4597,25 +4597,34 @@ void editor_ensure_final_newline(GeanyEditor *editor) } -void editor_set_font(GeanyEditor *editor, const gchar *font) +/* Similar to editor_set_font() but *only* sets the font, and doesn't take care + * of updating properties that might depend on the font */ +static void set_font(ScintillaObject *sci, const gchar *font) { gint style; gchar *font_name; PangoFontDescription *pfd; gdouble size; - g_return_if_fail(editor); + g_return_if_fail(sci); pfd = pango_font_description_from_string(font); size = pango_font_description_get_size(pfd) / (gdouble) PANGO_SCALE; font_name = g_strdup_printf("!%s", pango_font_description_get_family(pfd)); pango_font_description_free(pfd); for (style = 0; style <= STYLE_MAX; style++) - sci_set_font_fractional(editor->sci, style, font_name, size); + sci_set_font_fractional(sci, style, font_name, size); g_free(font_name); +} + +void editor_set_font(GeanyEditor *editor, const gchar *font) +{ + g_return_if_fail(editor); + + set_font(editor->sci, font); update_margins(editor->sci); /* zoom to 100% to prevent confusion */ sci_zoom_off(editor->sci); @@ -4926,7 +4935,6 @@ static ScintillaObject *create_new_sci(GeanyEditor *editor) setup_sci_keys(sci); - sci_set_symbol_margin(sci, editor_prefs.show_markers_margin); sci_set_lines_wrapped(sci, editor->line_wrapping); sci_set_caret_policy_x(sci, CARET_JUMPS | CARET_EVEN, 0); /* Y policy is set in editor_apply_update_prefs() */ @@ -5000,7 +5008,7 @@ ScintillaObject *editor_create_widget(GeanyEditor *editor) editor->sci = sci; editor_set_indent(editor, iprefs->type, iprefs->width); - editor_set_font(editor, interface_prefs.editor_font); + set_font(editor->sci, interface_prefs.editor_font); editor_apply_update_prefs(editor); /* if editor already had a widget, restore it */ Modified: src/sciwrappers.c 39 lines changed, 38 insertions(+), 1 deletions(-) =================================================================== @@ -144,10 +144,47 @@ void sci_set_mark_long_lines(ScintillaObject *sci, gint type, gint column, const } +/* Calls SCI_TEXTHEIGHT but tries very hard to cache the result as it's a very + * expensive operation */ +static gint sci_text_height_cached(ScintillaObject *sci) +{ + struct height_spec { + gchar *font; + gint size; + gint zoom; + gint extra; + }; + static struct height_spec cache = {0}; + static gint cache_value = 0; + struct height_spec current; + + current.font = sci_get_string(sci, SCI_STYLEGETFONT, 0); + current.size = SSM(sci, SCI_STYLEGETSIZEFRACTIONAL, 0, 0); + current.zoom = SSM(sci, SCI_GETZOOM, 0, 0); + current.extra = SSM(sci, SCI_GETEXTRAASCENT, 0, 0) + SSM(sci, SCI_GETEXTRADESCENT, 0, 0); + + if (g_strcmp0(current.font, cache.font) == 0 && + current.size == cache.size && + current.zoom == cache.zoom && + current.extra == cache.extra) + { + g_free(current.font); + } + else + { + g_free(cache.font); + cache = current; + + cache_value = SSM(sci, SCI_TEXTHEIGHT, 0, 0); + } + + return cache_value; +} + /* compute margin width based on ratio of line height */ static gint margin_width_from_line_height(ScintillaObject *sci, gdouble ratio, gint threshold) { - const gint line_height = SSM(sci, SCI_TEXTHEIGHT, 0, 0); + const gint line_height = sci_text_height_cached(sci); gint width; width = line_height * ratio; -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

[geany/geany] 33fc26: Add GNU regex bundle from upstream ctags and use it if missing
by Colomban Wendling 07 Feb '21

07 Feb '21

Branch: refs/heads/master Author: Colomban Wendling <ban(a)herbesfolles.org> Committer: Colomban Wendling <ban(a)herbesfolles.org> Date: Sun, 22 Nov 2020 10:45:30 UTC Commit: 33fc269ea8d4d03453489ca55a9ca208ec15aae2 https://github.com/geany/geany/commit/33fc269ea8d4d03453489ca55a9ca208ec15a… Log Message: ----------- Add GNU regex bundle from upstream ctags and use it if missing If we don't have `regcomp()` (e.g. on Windows), use the bundled GNU regex from upstream ctags. Modified Paths: -------------- configure.ac ctags/Makefile.am ctags/gnu_regex/README.txt ctags/gnu_regex/regcomp.c ctags/gnu_regex/regex.c ctags/gnu_regex/regex.h ctags/gnu_regex/regex_internal.c ctags/gnu_regex/regex_internal.h ctags/gnu_regex/regexec.c Modified: configure.ac 8 lines changed, 8 insertions(+), 0 deletions(-) =================================================================== @@ -44,6 +44,14 @@ AC_CHECK_HEADERS([fcntl.h glob.h stdlib.h sys/time.h errno.h limits.h]) # Checks for dependencies needed by ctags AC_CHECK_HEADERS([fnmatch.h direct.h io.h sys/dir.h]) AC_DEFINE([HAVE_STDBOOL_H], [1], [whether or not to use <stdbool.h>.]) +AC_CHECK_FUNC([regcomp], + [have_regcomp=yes], + [have_regcomp=no + dnl various stuff for ctags/gnu_regex/ + AC_CHECK_HEADERS([langinfo.h locale.h libintl.h wctype.h wchar.h]) + AC_CHECK_FUNCS([memcpy isblank wcrtomb mbrtowc wcscoll]) + AC_FUNC_ALLOCA]) +AM_CONDITIONAL([USE_BUNDLED_REGEX], [test "xno" = "x$have_regcomp"]) # Checks for typedefs, structures, and compiler characteristics. AC_TYPE_OFF_T Modified: ctags/Makefile.am 22 lines changed, 22 insertions(+), 0 deletions(-) =================================================================== @@ -176,3 +176,25 @@ libctags_la_SOURCES = \ main/xtag.h \ main/xtag_p.h \ $(parsers) + +# build bundled GNU regex if needed +if USE_BUNDLED_REGEX +noinst_LTLIBRARIES += libgnu_regex.la +libgnu_regex_la_SOURCES = \ + gnu_regex/regex.c \ + gnu_regex/regex.h +# regex.c includes other sources we have to distribute +EXTRA_libgnu_regex_la_SOURCES = \ + gnu_regex/regcomp.c \ + gnu_regex/regex.c \ + gnu_regex/regex.h \ + gnu_regex/regex_internal.c \ + gnu_regex/regex_internal.h \ + gnu_regex/regexec.c +libgnu_regex_la_CPPFLAGS = -D__USE_GNU +EXTRA_DIST = \ + gnu_regex/README.txt + +libctags_la_LIBADD = libgnu_regex.la +AM_CPPFLAGS += -I$(srcdir)/gnu_regex +endif Modified: ctags/gnu_regex/README.txt 5 lines changed, 5 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,5 @@ +These source files were taken from the GNU glibc-2.10.1 package. + + ftp://ftp.gnu.org/gnu/glibc/glibc-2.10.1.tar.bz2 + +Minor changes were made to eliminate compiler errors and warnings. Modified: ctags/gnu_regex/regcomp.c 3818 lines changed, 3818 insertions(+), 0 deletions(-) =================================================================== No diff available, check online Modified: ctags/gnu_regex/regex.c 74 lines changed, 74 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,74 @@ +/* Extended regular expression matching and search library. + Copyright (C) 2002, 2003, 2005 Free Software Foundation, Inc. + This file is part of the GNU C Library. + Contributed by Isamu Hasegawa <isamu(a)yamato.ibm.com>. + + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. + + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, write to the Free + Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA + 02111-1307 USA. */ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +/* Make sure no one compiles this code with a C++ compiler. */ +#ifdef __cplusplus +# error "This is C code, use a C compiler" +#endif + +#ifdef _LIBC +/* We have to keep the namespace clean. */ +# define regfree(preg) __regfree (preg) +# define regexec(pr, st, nm, pm, ef) __regexec (pr, st, nm, pm, ef) +# define regcomp(preg, pattern, cflags) __regcomp (preg, pattern, cflags) +# define regerror(errcode, preg, errbuf, errbuf_size) \ + __regerror(errcode, preg, errbuf, errbuf_size) +# define re_set_registers(bu, re, nu, st, en) \ + __re_set_registers (bu, re, nu, st, en) +# define re_match_2(bufp, string1, size1, string2, size2, pos, regs, stop) \ + __re_match_2 (bufp, string1, size1, string2, size2, pos, regs, stop) +# define re_match(bufp, string, size, pos, regs) \ + __re_match (bufp, string, size, pos, regs) +# define re_search(bufp, string, size, startpos, range, regs) \ + __re_search (bufp, string, size, startpos, range, regs) +# define re_compile_pattern(pattern, length, bufp) \ + __re_compile_pattern (pattern, length, bufp) +# define re_set_syntax(syntax) __re_set_syntax (syntax) +# define re_search_2(bufp, st1, s1, st2, s2, startpos, range, regs, stop) \ + __re_search_2 (bufp, st1, s1, st2, s2, startpos, range, regs, stop) +# define re_compile_fastmap(bufp) __re_compile_fastmap (bufp) + +# include "../locale/localeinfo.h" +#endif + +/* On some systems, limits.h sets RE_DUP_MAX to a lower value than + GNU regex allows. Include it before <regex.h>, which correctly + #undefs RE_DUP_MAX and sets it to the right value. */ +#include <limits.h> + +#include "regex.h" +#include "regex_internal.h" + +#include "regex_internal.c" +#include "regcomp.c" +#include "regexec.c" + +/* Binary backward compatibility. */ +#if _LIBC +# include <shlib-compat.h> +# if SHLIB_COMPAT (libc, GLIBC_2_0, GLIBC_2_3) +link_warning (re_max_failures, "the 're_max_failures' variable is obsolete and will go away.") +int re_max_failures = 2000; +# endif +#endif Modified: ctags/gnu_regex/regex.h 575 lines changed, 575 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,575 @@ +/* Definitions for data structures and routines for the regular + expression library. + Copyright (C) 1985,1989-93,1995-98,2000,2001,2002,2003,2005,2006,2008 + Free Software Foundation, Inc. + This file is part of the GNU C Library. + + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. + + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, write to the Free + Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA + 02111-1307 USA. */ + +#ifndef _REGEX_H +#define _REGEX_H 1 + +#include <sys/types.h> + +/* Allow the use in C++ code. */ +#ifdef __cplusplus +extern "C" { +#endif + +/* The following two types have to be signed and unsigned integer type + wide enough to hold a value of a pointer. For most ANSI compilers + ptrdiff_t and size_t should be likely OK. Still size of these two + types is 2 for Microsoft C. Ugh... */ +typedef long int s_reg_t; +typedef unsigned long int active_reg_t; + +/* The following bits are used to determine the regexp syntax we + recognize. The set/not-set meanings are chosen so that Emacs syntax + remains the value 0. The bits are given in alphabetical order, and + the definitions shifted by one from the previous bit; thus, when we + add or remove a bit, only one other definition need change. */ +typedef unsigned long int reg_syntax_t; + +#ifdef __USE_GNU +/* If this bit is not set, then \ inside a bracket expression is literal. + If set, then such a \ quotes the following character. */ +# define RE_BACKSLASH_ESCAPE_IN_LISTS ((unsigned long int) 1) + +/* If this bit is not set, then + and ? are operators, and \+ and \? are + literals. + If set, then \+ and \? are operators and + and ? are literals. */ +# define RE_BK_PLUS_QM (RE_BACKSLASH_ESCAPE_IN_LISTS << 1) + +/* If this bit is set, then character classes are supported. They are: + [:alpha:], [:upper:], [:lower:], [:digit:], [:alnum:], [:xdigit:], + [:space:], [:print:], [:punct:], [:graph:], and [:cntrl:]. + If not set, then character classes are not supported. */ +# define RE_CHAR_CLASSES (RE_BK_PLUS_QM << 1) + +/* If this bit is set, then ^ and $ are always anchors (outside bracket + expressions, of course). + If this bit is not set, then it depends: + ^ is an anchor if it is at the beginning of a regular + expression or after an open-group or an alternation operator; + $ is an anchor if it is at the end of a regular expression, or + before a close-group or an alternation operator. + + This bit could be (re)combined with RE_CONTEXT_INDEP_OPS, because + POSIX draft 11.2 says that * etc. in leading positions is undefined. + We already implemented a previous draft which made those constructs + invalid, though, so we haven't changed the code back. */ +# define RE_CONTEXT_INDEP_ANCHORS (RE_CHAR_CLASSES << 1) + +/* If this bit is set, then special characters are always special + regardless of where they are in the pattern. + If this bit is not set, then special characters are special only in + some contexts; otherwise they are ordinary. Specifically, + * + ? and intervals are only special when not after the beginning, + open-group, or alternation operator. */ +# define RE_CONTEXT_INDEP_OPS (RE_CONTEXT_INDEP_ANCHORS << 1) + +/* If this bit is set, then *, +, ?, and { cannot be first in an re or + immediately after an alternation or begin-group operator. */ +# define RE_CONTEXT_INVALID_OPS (RE_CONTEXT_INDEP_OPS << 1) + +/* If this bit is set, then . matches newline. + If not set, then it doesn't. */ +# define RE_DOT_NEWLINE (RE_CONTEXT_INVALID_OPS << 1) + +/* If this bit is set, then . doesn't match NUL. + If not set, then it does. */ +# define RE_DOT_NOT_NULL (RE_DOT_NEWLINE << 1) + +/* If this bit is set, nonmatching lists [^...] do not match newline. + If not set, they do. */ +# define RE_HAT_LISTS_NOT_NEWLINE (RE_DOT_NOT_NULL << 1) + +/* If this bit is set, either \{...\} or {...} defines an + interval, depending on RE_NO_BK_BRACES. + If not set, \{, \}, {, and } are literals. */ +# define RE_INTERVALS (RE_HAT_LISTS_NOT_NEWLINE << 1) + +/* If this bit is set, +, ? and | aren't recognized as operators. + If not set, they are. */ +# define RE_LIMITED_OPS (RE_INTERVALS << 1) + +/* If this bit is set, newline is an alternation operator. + If not set, newline is literal. */ +# define RE_NEWLINE_ALT (RE_LIMITED_OPS << 1) + +/* If this bit is set, then `{...}' defines an interval, and \{ and \} + are literals. + If not set, then `\{...\}' defines an interval. */ +# define RE_NO_BK_BRACES (RE_NEWLINE_ALT << 1) + +/* If this bit is set, (...) defines a group, and $ and $ are literals. + If not set, $...$ defines a group, and ( and ) are literals. */ +# define RE_NO_BK_PARENS (RE_NO_BK_BRACES << 1) + +/* If this bit is set, then \<digit> matches <digit>. + If not set, then \<digit> is a back-reference. */ +# define RE_NO_BK_REFS (RE_NO_BK_PARENS << 1) + +/* If this bit is set, then | is an alternation operator, and \| is literal. + If not set, then \| is an alternation operator, and | is literal. */ +# define RE_NO_BK_VBAR (RE_NO_BK_REFS << 1) + +/* If this bit is set, then an ending range point collating higher + than the starting range point, as in [z-a], is invalid. + If not set, then when ending range point collates higher than the + starting range point, the range is ignored. */ +# define RE_NO_EMPTY_RANGES (RE_NO_BK_VBAR << 1) + +/* If this bit is set, then an unmatched ) is ordinary. + If not set, then an unmatched ) is invalid. */ +# define RE_UNMATCHED_RIGHT_PAREN_ORD (RE_NO_EMPTY_RANGES << 1) + +/* If this bit is set, succeed as soon as we match the whole pattern, + without further backtracking. */ +# define RE_NO_POSIX_BACKTRACKING (RE_UNMATCHED_RIGHT_PAREN_ORD << 1) + +/* If this bit is set, do not process the GNU regex operators. + If not set, then the GNU regex operators are recognized. */ +# define RE_NO_GNU_OPS (RE_NO_POSIX_BACKTRACKING << 1) + +/* If this bit is set, turn on internal regex debugging. + If not set, and debugging was on, turn it off. + This only works if regex.c is compiled -DDEBUG. + We define this bit always, so that all that's needed to turn on + debugging is to recompile regex.c; the calling code can always have + this bit set, and it won't affect anything in the normal case. */ +# define RE_DEBUG (RE_NO_GNU_OPS << 1) + +/* If this bit is set, a syntactically invalid interval is treated as + a string of ordinary characters. For example, the ERE 'a{1' is + treated as 'a\{1'. */ +# define RE_INVALID_INTERVAL_ORD (RE_DEBUG << 1) + +/* If this bit is set, then ignore case when matching. + If not set, then case is significant. */ +# define RE_ICASE (RE_INVALID_INTERVAL_ORD << 1) + +/* This bit is used internally like RE_CONTEXT_INDEP_ANCHORS but only + for ^, because it is difficult to scan the regex backwards to find + whether ^ should be special. */ +# define RE_CARET_ANCHORS_HERE (RE_ICASE << 1) + +/* If this bit is set, then \{ cannot be first in an bre or + immediately after an alternation or begin-group operator. */ +# define RE_CONTEXT_INVALID_DUP (RE_CARET_ANCHORS_HERE << 1) + +/* If this bit is set, then no_sub will be set to 1 during + re_compile_pattern. */ +# define RE_NO_SUB (RE_CONTEXT_INVALID_DUP << 1) +#endif + +/* This global variable defines the particular regexp syntax to use (for + some interfaces). When a regexp is compiled, the syntax used is + stored in the pattern buffer, so changing this does not affect + already-compiled regexps. */ +extern reg_syntax_t re_syntax_options; + +#ifdef __USE_GNU +/* Define combinations of the above bits for the standard possibilities. + (The [[[ comments delimit what gets put into the Texinfo file, so + don't delete them!) */ +/* [[[begin syntaxes]]] */ +#define RE_SYNTAX_EMACS 0 + +#define RE_SYNTAX_AWK \ + (RE_BACKSLASH_ESCAPE_IN_LISTS | RE_DOT_NOT_NULL \ + | RE_NO_BK_PARENS | RE_NO_BK_REFS \ + | RE_NO_BK_VBAR | RE_NO_EMPTY_RANGES \ + | RE_DOT_NEWLINE | RE_CONTEXT_INDEP_ANCHORS \ + | RE_UNMATCHED_RIGHT_PAREN_ORD | RE_NO_GNU_OPS) + +#define RE_SYNTAX_GNU_AWK \ + ((RE_SYNTAX_POSIX_EXTENDED | RE_BACKSLASH_ESCAPE_IN_LISTS | RE_DEBUG) \ + & ~(RE_DOT_NOT_NULL | RE_INTERVALS | RE_CONTEXT_INDEP_OPS \ + | RE_CONTEXT_INVALID_OPS )) + +#define RE_SYNTAX_POSIX_AWK \ + (RE_SYNTAX_POSIX_EXTENDED | RE_BACKSLASH_ESCAPE_IN_LISTS \ + | RE_INTERVALS | RE_NO_GNU_OPS) + +#define RE_SYNTAX_GREP \ + (RE_BK_PLUS_QM | RE_CHAR_CLASSES \ + | RE_HAT_LISTS_NOT_NEWLINE | RE_INTERVALS \ + | RE_NEWLINE_ALT) + +#define RE_SYNTAX_EGREP \ + (RE_CHAR_CLASSES | RE_CONTEXT_INDEP_ANCHORS \ + | RE_CONTEXT_INDEP_OPS | RE_HAT_LISTS_NOT_NEWLINE \ + | RE_NEWLINE_ALT | RE_NO_BK_PARENS \ + | RE_NO_BK_VBAR) + +#define RE_SYNTAX_POSIX_EGREP \ + (RE_SYNTAX_EGREP | RE_INTERVALS | RE_NO_BK_BRACES \ + | RE_INVALID_INTERVAL_ORD) + +/* P1003.2/D11.2, section 4.20.7.1, lines 5078ff. */ +#define RE_SYNTAX_ED RE_SYNTAX_POSIX_BASIC + +#define RE_SYNTAX_SED RE_SYNTAX_POSIX_BASIC + +/* Syntax bits common to both basic and extended POSIX regex syntax. */ +#define _RE_SYNTAX_POSIX_COMMON \ + (RE_CHAR_CLASSES | RE_DOT_NEWLINE | RE_DOT_NOT_NULL \ + | RE_INTERVALS | RE_NO_EMPTY_RANGES) + +#define RE_SYNTAX_POSIX_BASIC \ + (_RE_SYNTAX_POSIX_COMMON | RE_BK_PLUS_QM | RE_CONTEXT_INVALID_DUP) + +/* Differs from ..._POSIX_BASIC only in that RE_BK_PLUS_QM becomes + RE_LIMITED_OPS, i.e., \? \+ \| are not recognized. Actually, this + isn't minimal, since other operators, such as \`, aren't disabled. */ +#define RE_SYNTAX_POSIX_MINIMAL_BASIC \ + (_RE_SYNTAX_POSIX_COMMON | RE_LIMITED_OPS) + +#define RE_SYNTAX_POSIX_EXTENDED \ + (_RE_SYNTAX_POSIX_COMMON | RE_CONTEXT_INDEP_ANCHORS \ + | RE_CONTEXT_INDEP_OPS | RE_NO_BK_BRACES \ + | RE_NO_BK_PARENS | RE_NO_BK_VBAR \ + | RE_CONTEXT_INVALID_OPS | RE_UNMATCHED_RIGHT_PAREN_ORD) + +/* Differs from ..._POSIX_EXTENDED in that RE_CONTEXT_INDEP_OPS is + removed and RE_NO_BK_REFS is added. */ +#define RE_SYNTAX_POSIX_MINIMAL_EXTENDED \ + (_RE_SYNTAX_POSIX_COMMON | RE_CONTEXT_INDEP_ANCHORS \ + | RE_CONTEXT_INVALID_OPS | RE_NO_BK_BRACES \ + | RE_NO_BK_PARENS | RE_NO_BK_REFS \ + | RE_NO_BK_VBAR | RE_UNMATCHED_RIGHT_PAREN_ORD) +/* [[[end syntaxes]]] */ + +/* Maximum number of duplicates an interval can allow. Some systems + (erroneously) define this in other header files, but we want our + value, so remove any previous define. */ +# ifdef RE_DUP_MAX +# undef RE_DUP_MAX +# endif +/* If sizeof(int) == 2, then ((1 << 15) - 1) overflows. */ +# define RE_DUP_MAX (0x7fff) +#endif + + +/* POSIX `cflags' bits (i.e., information for `regcomp'). */ + +/* If this bit is set, then use extended regular expression syntax. + If not set, then use basic regular expression syntax. */ +#define REG_EXTENDED 1 + +/* If this bit is set, then ignore case when matching. + If not set, then case is significant. */ +#define REG_ICASE (REG_EXTENDED << 1) + +/* If this bit is set, then anchors do not match at newline + characters in the string. + If not set, then anchors do match at newlines. */ +#define REG_NEWLINE (REG_ICASE << 1) + +/* If this bit is set, then report only success or fail in regexec. + If not set, then returns differ between not matching and errors. */ +#define REG_NOSUB (REG_NEWLINE << 1) + + +/* POSIX `eflags' bits (i.e., information for regexec). */ + +/* If this bit is set, then the beginning-of-line operator doesn't match + the beginning of the string (presumably because it's not the + beginning of a line). + If not set, then the beginning-of-line operator does match the + beginning of the string. */ +#define REG_NOTBOL 1 + +/* Like REG_NOTBOL, except for the end-of-line. */ +#define REG_NOTEOL (1 << 1) + +/* Use PMATCH[0] to delimit the start and end of the search in the + buffer. */ +#define REG_STARTEND (1 << 2) + + +/* If any error codes are removed, changed, or added, update the + `re_error_msg' table in regex.c. */ +typedef enum +{ +#if defined _XOPEN_SOURCE || defined __USE_XOPEN2K + REG_ENOSYS = -1, /* This will never happen for this implementation. */ +#endif + + REG_NOERROR = 0, /* Success. */ + REG_NOMATCH, /* Didn't find a match (for regexec). */ + + /* POSIX regcomp return error codes. (In the order listed in the + standard.) */ + REG_BADPAT, /* Invalid pattern. */ + REG_ECOLLATE, /* Inalid collating element. */ + REG_ECTYPE, /* Invalid character class name. */ + REG_EESCAPE, /* Trailing backslash. */ + REG_ESUBREG, /* Invalid back reference. */ + REG_EBRACK, /* Unmatched left bracket. */ + REG_EPAREN, /* Parenthesis imbalance. */ + REG_EBRACE, /* Unmatched \{. */ + REG_BADBR, /* Invalid contents of \{\}. */ + REG_ERANGE, /* Invalid range end. */ + REG_ESPACE, /* Ran out of memory. */ + REG_BADRPT, /* No preceding re for repetition op. */ + + /* Error codes we've added. */ + REG_EEND, /* Premature end. */ + REG_ESIZE, /* Compiled pattern bigger than 2^16 bytes. */ + REG_ERPAREN /* Unmatched ) or \); not returned from regcomp. */ +} reg_errcode_t; + +/* This data structure represents a compiled pattern. Before calling + the pattern compiler, the fields `buffer', `allocated', `fastmap', + `translate', and `no_sub' can be set. After the pattern has been + compiled, the `re_nsub' field is available. All other fields are + private to the regex routines. */ + +#ifndef RE_TRANSLATE_TYPE +# define __RE_TRANSLATE_TYPE unsigned char * +# ifdef __USE_GNU +# define RE_TRANSLATE_TYPE __RE_TRANSLATE_TYPE +# endif +#endif + +#ifdef __USE_GNU +# define __REPB_PREFIX(name) name +#else +# define __REPB_PREFIX(name) __##name +#endif + +struct re_pattern_buffer +{ + /* Space that holds the compiled pattern. It is declared as + `unsigned char *' because its elements are sometimes used as + array indexes. */ + unsigned char *__REPB_PREFIX(buffer); + + /* Number of bytes to which `buffer' points. */ + unsigned long int __REPB_PREFIX(allocated); + + /* Number of bytes actually used in `buffer'. */ + unsigned long int __REPB_PREFIX(used); + + /* Syntax setting with which the pattern was compiled. */ + reg_syntax_t __REPB_PREFIX(syntax); + + /* Pointer to a fastmap, if any, otherwise zero. re_search uses the + fastmap, if there is one, to skip over impossible starting points + for matches. */ + char *__REPB_PREFIX(fastmap); + + /* Either a translate table to apply to all characters before + comparing them, or zero for no translation. The translation is + applied to a pattern when it is compiled and to a string when it + is matched. */ + __RE_TRANSLATE_TYPE __REPB_PREFIX(translate); + + /* Number of subexpressions found by the compiler. */ + size_t re_nsub; + + /* Zero if this pattern cannot match the empty string, one else. + Well, in truth it's used only in `re_search_2', to see whether or + not we should use the fastmap, so we don't set this absolutely + perfectly; see `re_compile_fastmap' (the `duplicate' case). */ + unsigned __REPB_PREFIX(can_be_null) : 1; + + /* If REGS_UNALLOCATED, allocate space in the `regs' structure + for `max (RE_NREGS, re_nsub + 1)' groups. + If REGS_REALLOCATE, reallocate space if necessary. + If REGS_FIXED, use what's there. */ +#ifdef __USE_GNU +# define REGS_UNALLOCATED 0 +# define REGS_REALLOCATE 1 +# define REGS_FIXED 2 +#endif + unsigned __REPB_PREFIX(regs_allocated) : 2; + + /* Set to zero when `regex_compile' compiles a pattern; set to one + by `re_compile_fastmap' if it updates the fastmap. */ + unsigned __REPB_PREFIX(fastmap_accurate) : 1; + + /* If set, `re_match_2' does not return information about + subexpressions. */ + unsigned __REPB_PREFIX(no_sub) : 1; + + /* If set, a beginning-of-line anchor doesn't match at the beginning + of the string. */ + unsigned __REPB_PREFIX(not_bol) : 1; + + /* Similarly for an end-of-line anchor. */ + unsigned __REPB_PREFIX(not_eol) : 1; + + /* If true, an anchor at a newline matches. */ + unsigned __REPB_PREFIX(newline_anchor) : 1; +}; + +typedef struct re_pattern_buffer regex_t; + +/* Type for byte offsets within the string. POSIX mandates this. */ +typedef int regoff_t; + + +#ifdef __USE_GNU +/* This is the structure we store register match data in. See + regex.texinfo for a full description of what registers match. */ +struct re_registers +{ + unsigned num_regs; + regoff_t *start; + regoff_t *end; +}; + + +/* If `regs_allocated' is REGS_UNALLOCATED in the pattern buffer, + `re_match_2' returns information about at least this many registers + the first time a `regs' structure is passed. */ +# ifndef RE_NREGS +# define RE_NREGS 30 +# endif +#endif + + +/* POSIX specification for registers. Aside from the different names than + `re_registers', POSIX uses an array of structures, instead of a + structure of arrays. */ +typedef struct +{ + regoff_t rm_so; /* Byte offset from string's start to substring's start. */ + regoff_t rm_eo; /* Byte offset from string's start to substring's end. */ +} regmatch_t; + +/* Declarations for routines. */ + +#ifdef __USE_GNU +/* Sets the current default syntax to SYNTAX, and return the old syntax. + You can also simply assign to the `re_syntax_options' variable. */ +extern reg_syntax_t re_set_syntax (reg_syntax_t __syntax); + +/* Compile the regular expression PATTERN, with length LENGTH + and syntax given by the global `re_syntax_options', into the buffer + BUFFER. Return NULL if successful, and an error string if not. */ +extern const char *re_compile_pattern (const char *__pattern, size_t __length, + struct re_pattern_buffer *__buffer); + + +/* Compile a fastmap for the compiled pattern in BUFFER; used to + accelerate searches. Return 0 if successful and -2 if was an + internal error. */ +extern int re_compile_fastmap (struct re_pattern_buffer *__buffer); + + +/* Search in the string STRING (with length LENGTH) for the pattern + compiled into BUFFER. Start searching at position START, for RANGE + characters. Return the starting position of the match, -1 for no + match, or -2 for an internal error. Also return register + information in REGS (if REGS and BUFFER->no_sub are nonzero). */ +extern int re_search (struct re_pattern_buffer *__buffer, const char *__string, + int __length, int __start, int __range, + struct re_registers *__regs); + + +/* Like `re_search', but search in the concatenation of STRING1 and + STRING2. Also, stop searching at index START + STOP. */ +extern int re_search_2 (struct re_pattern_buffer *__buffer, + const char *__string1, int __length1, + const char *__string2, int __length2, int __start, + int __range, struct re_registers *__regs, int __stop); + + +/* Like `re_search', but return how many characters in STRING the regexp + in BUFFER matched, starting at position START. */ +extern int re_match (struct re_pattern_buffer *__buffer, const char *__string, + int __length, int __start, struct re_registers *__regs); + + +/* Relates to `re_match' as `re_search_2' relates to `re_search'. */ +extern int re_match_2 (struct re_pattern_buffer *__buffer, + const char *__string1, int __length1, + const char *__string2, int __length2, int __start, + struct re_registers *__regs, int __stop); + + +/* Set REGS to hold NUM_REGS registers, storing them in STARTS and + ENDS. Subsequent matches using BUFFER and REGS will use this memory + for recording register information. STARTS and ENDS must be + allocated with malloc, and must each be at least `NUM_REGS * sizeof + (regoff_t)' bytes long. + + If NUM_REGS == 0, then subsequent matches should allocate their own + register data. + + Unless this function is called, the first search or match using + PATTERN_BUFFER will allocate its own register data, without + freeing the old data. */ +extern void re_set_registers (struct re_pattern_buffer *__buffer, + struct re_registers *__regs, + unsigned int __num_regs, + regoff_t *__starts, regoff_t *__ends); +#endif /* Use GNU */ + +#if defined _REGEX_RE_COMP || (defined _LIBC && defined __USE_BSD) +# ifndef _CRAY +/* 4.2 bsd compatibility. */ +extern char *re_comp (const char *); +extern int re_exec (const char *); +# endif +#endif + +/* GCC 2.95 and later have "__restrict"; C99 compilers have + "restrict", and "configure" may have defined "restrict". */ +#ifndef __restrict +# if ! (2 < __GNUC__ || (2 == __GNUC__ && 95 <= __GNUC_MINOR__)) +# if defined restrict || 199901L <= __STDC_VERSION__ +# define __restrict restrict +# else +# define __restrict +# endif +# endif +#endif +/* gcc 3.1 and up support the [restrict] syntax. */ +#ifndef __restrict_arr +# if (__GNUC__ > 3 || (__GNUC__ == 3 && __GNUC_MINOR__ >= 1)) \ + && !defined __GNUG__ +# define __restrict_arr __restrict +# else +# define __restrict_arr +# endif +#endif + +/* POSIX compatibility. */ +extern int regcomp (regex_t *__restrict __preg, + const char *__restrict __pattern, + int __cflags); + +extern int regexec (const regex_t *__restrict __preg, + const char *__restrict __string, size_t __nmatch, + regmatch_t __pmatch[__restrict_arr], + int __eflags); + +extern size_t regerror (int __errcode, const regex_t *__restrict __preg, + char *__restrict __errbuf, size_t __errbuf_size); + +extern void regfree (regex_t *__preg); + + +#ifdef __cplusplus +} +#endif /* C++ */ + +#endif /* regex.h */ Modified: ctags/gnu_regex/regex_internal.c 1715 lines changed, 1715 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,1715 @@ +/* Extended regular expression matching and search library. + Copyright (C) 2002, 2003, 2004, 2005, 2006 Free Software Foundation, Inc. + This file is part of the GNU C Library. + Contributed by Isamu Hasegawa <isamu(a)yamato.ibm.com>. + + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. + + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, write to the Free + Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA + 02111-1307 USA. */ + +static void re_string_construct_common (const char *str, int len, + re_string_t *pstr, + RE_TRANSLATE_TYPE trans, int icase, + const re_dfa_t *dfa) internal_function; +static re_dfastate_t *create_ci_newstate (const re_dfa_t *dfa, + const re_node_set *nodes, + unsigned int hash) internal_function; +static re_dfastate_t *create_cd_newstate (const re_dfa_t *dfa, + const re_node_set *nodes, + unsigned int context, + unsigned int hash) internal_function; + +/* Functions for string operation. */ + +/* This function allocate the buffers. It is necessary to call + re_string_reconstruct before using the object. */ + +static reg_errcode_t +internal_function +re_string_allocate (re_string_t *pstr, const char *str, int len, int init_len, + RE_TRANSLATE_TYPE trans, int icase, const re_dfa_t *dfa) +{ + reg_errcode_t ret; + int init_buf_len; + + /* Ensure at least one character fits into the buffers. */ + if (init_len < dfa->mb_cur_max) + init_len = dfa->mb_cur_max; + init_buf_len = (len + 1 < init_len) ? len + 1: init_len; + re_string_construct_common (str, len, pstr, trans, icase, dfa); + + ret = re_string_realloc_buffers (pstr, init_buf_len); + if (BE (ret != REG_NOERROR, 0)) + return ret; + + pstr->word_char = dfa->word_char; + pstr->word_ops_used = dfa->word_ops_used; + pstr->mbs = pstr->mbs_allocated ? pstr->mbs : (unsigned char *) str; + pstr->valid_len = (pstr->mbs_allocated || dfa->mb_cur_max > 1) ? 0 : len; + pstr->valid_raw_len = pstr->valid_len; + return REG_NOERROR; +} + +/* This function allocate the buffers, and initialize them. */ + +static reg_errcode_t +internal_function +re_string_construct (re_string_t *pstr, const char *str, int len, + RE_TRANSLATE_TYPE trans, int icase, const re_dfa_t *dfa) +{ + reg_errcode_t ret; + memset (pstr, '\0', sizeof (re_string_t)); + re_string_construct_common (str, len, pstr, trans, icase, dfa); + + if (len > 0) + { + ret = re_string_realloc_buffers (pstr, len + 1); + if (BE (ret != REG_NOERROR, 0)) + return ret; + } + pstr->mbs = pstr->mbs_allocated ? pstr->mbs : (unsigned char *) str; + + if (icase) + { +#ifdef RE_ENABLE_I18N + if (dfa->mb_cur_max > 1) + { + while (1) + { + ret = build_wcs_upper_buffer (pstr); + if (BE (ret != REG_NOERROR, 0)) + return ret; + if (pstr->valid_raw_len >= len) + break; + if (pstr->bufs_len > pstr->valid_len + dfa->mb_cur_max) + break; + ret = re_string_realloc_buffers (pstr, pstr->bufs_len * 2); + if (BE (ret != REG_NOERROR, 0)) + return ret; + } + } + else +#endif /* RE_ENABLE_I18N */ + build_upper_buffer (pstr); + } + else + { +#ifdef RE_ENABLE_I18N + if (dfa->mb_cur_max > 1) + build_wcs_buffer (pstr); + else +#endif /* RE_ENABLE_I18N */ + { + if (trans != NULL) + re_string_translate_buffer (pstr); + else + { + pstr->valid_len = pstr->bufs_len; + pstr->valid_raw_len = pstr->bufs_len; + } + } + } + + return REG_NOERROR; +} + +/* Helper functions for re_string_allocate, and re_string_construct. */ + +static reg_errcode_t +internal_function +re_string_realloc_buffers (re_string_t *pstr, int new_buf_len) +{ +#ifdef RE_ENABLE_I18N + if (pstr->mb_cur_max > 1) + { + wint_t *new_wcs = re_realloc (pstr->wcs, wint_t, new_buf_len); + if (BE (new_wcs == NULL, 0)) + return REG_ESPACE; + pstr->wcs = new_wcs; + if (pstr->offsets != NULL) + { + int *new_offsets = re_realloc (pstr->offsets, int, new_buf_len); + if (BE (new_offsets == NULL, 0)) + return REG_ESPACE; + pstr->offsets = new_offsets; + } + } +#endif /* RE_ENABLE_I18N */ + if (pstr->mbs_allocated) + { + unsigned char *new_mbs = re_realloc (pstr->mbs, unsigned char, + new_buf_len); + if (BE (new_mbs == NULL, 0)) + return REG_ESPACE; + pstr->mbs = new_mbs; + } + pstr->bufs_len = new_buf_len; + return REG_NOERROR; +} + + +static void +internal_function +re_string_construct_common (const char *str, int len, re_string_t *pstr, + RE_TRANSLATE_TYPE trans, int icase, + const re_dfa_t *dfa) +{ + pstr->raw_mbs = (const unsigned char *) str; + pstr->len = len; + pstr->raw_len = len; + pstr->trans = trans; + pstr->icase = icase ? 1 : 0; + pstr->mbs_allocated = (trans != NULL || icase); + pstr->mb_cur_max = dfa->mb_cur_max; + pstr->is_utf8 = dfa->is_utf8; + pstr->map_notascii = dfa->map_notascii; + pstr->stop = pstr->len; + pstr->raw_stop = pstr->stop; +} + +#ifdef RE_ENABLE_I18N + +/* Build wide character buffer PSTR->WCS. + If the byte sequence of the string are: + <mb1>(0), <mb1>(1), <mb2>(0), <mb2>(1), <sb3> + Then wide character buffer will be: + <wc1> , WEOF , <wc2> , WEOF , <wc3> + We use WEOF for padding, they indicate that the position isn't + a first byte of a multibyte character. + + Note that this function assumes PSTR->VALID_LEN elements are already + built and starts from PSTR->VALID_LEN. */ + +static void +internal_function +build_wcs_buffer (re_string_t *pstr) +{ +#ifdef _LIBC + unsigned char buf[MB_LEN_MAX]; + assert (MB_LEN_MAX >= pstr->mb_cur_max); +#else + unsigned char buf[64]; +#endif + mbstate_t prev_st; + int byte_idx, end_idx, remain_len; + size_t mbclen; + + /* Build the buffers from pstr->valid_len to either pstr->len or + pstr->bufs_len. */ + end_idx = (pstr->bufs_len > pstr->len) ? pstr->len : pstr->bufs_len; + for (byte_idx = pstr->valid_len; byte_idx < end_idx;) + { + wchar_t wc; + const char *p; + + remain_len = end_idx - byte_idx; + prev_st = pstr->cur_state; + /* Apply the translation if we need. */ + if (BE (pstr->trans != NULL, 0)) + { + int i, ch; + + for (i = 0; i < pstr->mb_cur_max && i < remain_len; ++i) + { + ch = pstr->raw_mbs [pstr->raw_mbs_idx + byte_idx + i]; + buf[i] = pstr->mbs[byte_idx + i] = pstr->trans[ch]; + } + p = (const char *) buf; + } + else + p = (const char *) pstr->raw_mbs + pstr->raw_mbs_idx + byte_idx; + mbclen = __mbrtowc (&wc, p, remain_len, &pstr->cur_state); + if (BE (mbclen == (size_t) -2, 0)) + { + /* The buffer doesn't have enough space, finish to build. */ + pstr->cur_state = prev_st; + break; + } + else if (BE (mbclen == (size_t) -1 || mbclen == 0, 0)) + { + /* We treat these cases as a singlebyte character. */ + mbclen = 1; + wc = (wchar_t) pstr->raw_mbs[pstr->raw_mbs_idx + byte_idx]; + if (BE (pstr->trans != NULL, 0)) + wc = pstr->trans[wc]; + pstr->cur_state = prev_st; + } + + /* Write wide character and padding. */ + pstr->wcs[byte_idx++] = wc; + /* Write paddings. */ + for (remain_len = byte_idx + mbclen - 1; byte_idx < remain_len ;) + pstr->wcs[byte_idx++] = WEOF; + } + pstr->valid_len = byte_idx; + pstr->valid_raw_len = byte_idx; +} + +/* Build wide character buffer PSTR->WCS like build_wcs_buffer, + but for REG_ICASE. */ + +static reg_errcode_t +internal_function +build_wcs_upper_buffer (re_string_t *pstr) +{ + mbstate_t prev_st; + int src_idx, byte_idx, end_idx, remain_len; + size_t mbclen; +#ifdef _LIBC + char buf[MB_LEN_MAX]; + assert (MB_LEN_MAX >= pstr->mb_cur_max); +#else + char buf[64]; +#endif + + byte_idx = pstr->valid_len; + end_idx = (pstr->bufs_len > pstr->len) ? pstr->len : pstr->bufs_len; + + /* The following optimization assumes that ASCII characters can be + mapped to wide characters with a simple cast. */ + if (! pstr->map_notascii && pstr->trans == NULL && !pstr->offsets_needed) + { + while (byte_idx < end_idx) + { + wchar_t wc; + + if (isascii (pstr->raw_mbs[pstr->raw_mbs_idx + byte_idx]) + && mbsinit (&pstr->cur_state)) + { + /* In case of a singlebyte character. */ + pstr->mbs[byte_idx] + = toupper (pstr->raw_mbs[pstr->raw_mbs_idx + byte_idx]); + /* The next step uses the assumption that wchar_t is encoded + ASCII-safe: all ASCII values can be converted like this. */ + pstr->wcs[byte_idx] = (wchar_t) pstr->mbs[byte_idx]; + ++byte_idx; + continue; + } + + remain_len = end_idx - byte_idx; + prev_st = pstr->cur_state; + mbclen = __mbrtowc (&wc, + ((const char *) pstr->raw_mbs + pstr->raw_mbs_idx + + byte_idx), remain_len, &pstr->cur_state); + if (BE (mbclen + 2 > 2, 1)) + { + wchar_t wcu = wc; + if (iswlower (wc)) + { + size_t mbcdlen; + + wcu = towupper (wc); + mbcdlen = wcrtomb (buf, wcu, &prev_st); + if (BE (mbclen == mbcdlen, 1)) + memcpy (pstr->mbs + byte_idx, buf, mbclen); + else + { + src_idx = byte_idx; + goto offsets_needed; + } + } + else + memcpy (pstr->mbs + byte_idx, + pstr->raw_mbs + pstr->raw_mbs_idx + byte_idx, mbclen); + pstr->wcs[byte_idx++] = wcu; + /* Write paddings. */ + for (remain_len = byte_idx + mbclen - 1; byte_idx < remain_len ;) + pstr->wcs[byte_idx++] = WEOF; + } + else if (mbclen == (size_t) -1 || mbclen == 0) + { + /* It is an invalid character or '\0'. Just use the byte. */ + int ch = pstr->raw_mbs[pstr->raw_mbs_idx + byte_idx]; + pstr->mbs[byte_idx] = ch; + /* And also cast it to wide char. */ + pstr->wcs[byte_idx++] = (wchar_t) ch; + if (BE (mbclen == (size_t) -1, 0)) + pstr->cur_state = prev_st; + } + else + { + /* The buffer doesn't have enough space, finish to build. */ + pstr->cur_state = prev_st; + break; + } + } + pstr->valid_len = byte_idx; + pstr->valid_raw_len = byte_idx; + return REG_NOERROR; + } + else + for (src_idx = pstr->valid_raw_len; byte_idx < end_idx;) + { + wchar_t wc; + const char *p; + offsets_needed: + remain_len = end_idx - byte_idx; + prev_st = pstr->cur_state; + if (BE (pstr->trans != NULL, 0)) + { + int i, ch; + + for (i = 0; i < pstr->mb_cur_max && i < remain_len; ++i) + { + ch = pstr->raw_mbs [pstr->raw_mbs_idx + src_idx + i]; + buf[i] = pstr->trans[ch]; + } + p = (const char *) buf; + } + else + p = (const char *) pstr->raw_mbs + pstr->raw_mbs_idx + src_idx; + mbclen = __mbrtowc (&wc, p, remain_len, &pstr->cur_state); + if (BE (mbclen + 2 > 2, 1)) + { + wchar_t wcu = wc; + if (iswlower (wc)) + { + size_t mbcdlen; + + wcu = towupper (wc); + mbcdlen = wcrtomb ((char *) buf, wcu, &prev_st); + if (BE (mbclen == mbcdlen, 1)) + memcpy (pstr->mbs + byte_idx, buf, mbclen); + else if (mbcdlen != (size_t) -1) + { + size_t i; + + if (byte_idx + mbcdlen > pstr->bufs_len) + { + pstr->cur_state = prev_st; + break; + } + + if (pstr->offsets == NULL) + { + pstr->offsets = re_malloc (int, pstr->bufs_len); + + if (pstr->offsets == NULL) + return REG_ESPACE; + } + if (!pstr->offsets_needed) + { + for (i = 0; i < (size_t) byte_idx; ++i) + pstr->offsets[i] = i; + pstr->offsets_needed = 1; + } + + memcpy (pstr->mbs + byte_idx, buf, mbcdlen); + pstr->wcs[byte_idx] = wcu; + pstr->offsets[byte_idx] = src_idx; + for (i = 1; i < mbcdlen; ++i) + { + pstr->offsets[byte_idx + i] + = src_idx + (i < mbclen ? i : mbclen - 1); + pstr->wcs[byte_idx + i] = WEOF; + } + pstr->len += mbcdlen - mbclen; + if (pstr->raw_stop > src_idx) + pstr->stop += mbcdlen - mbclen; + end_idx = (pstr->bufs_len > pstr->len) + ? pstr->len : pstr->bufs_len; + byte_idx += mbcdlen; + src_idx += mbclen; + continue; + } + else + memcpy (pstr->mbs + byte_idx, p, mbclen); + } + else + memcpy (pstr->mbs + byte_idx, p, mbclen); + + if (BE (pstr->offsets_needed != 0, 0)) + { + size_t i; + for (i = 0; i < mbclen; ++i) + pstr->offsets[byte_idx + i] = src_idx + i; + } + src_idx += mbclen; + + pstr->wcs[byte_idx++] = wcu; + /* Write paddings. */ + for (remain_len = byte_idx + mbclen - 1; byte_idx < remain_len ;) + pstr->wcs[byte_idx++] = WEOF; + } + else if (mbclen == (size_t) -1 || mbclen == 0) + { + /* It is an invalid character or '\0'. Just use the byte. */ + int ch = pstr->raw_mbs[pstr->raw_mbs_idx + src_idx]; + + if (BE (pstr->trans != NULL, 0)) + ch = pstr->trans [ch]; + pstr->mbs[byte_idx] = ch; + + if (BE (pstr->offsets_needed != 0, 0)) + pstr->offsets[byte_idx] = src_idx; + ++src_idx; + + /* And also cast it to wide char. */ + pstr->wcs[byte_idx++] = (wchar_t) ch; + if (BE (mbclen == (size_t) -1, 0)) + pstr->cur_state = prev_st; + } + else + { + /* The buffer doesn't have enough space, finish to build. */ + pstr->cur_state = prev_st; + break; + } + } + pstr->valid_len = byte_idx; + pstr->valid_raw_len = src_idx; + return REG_NOERROR; +} + +/* Skip characters until the index becomes greater than NEW_RAW_IDX. + Return the index. */ + +static int +internal_function +re_string_skip_chars (re_string_t *pstr, int new_raw_idx, wint_t *last_wc) +{ + mbstate_t prev_st; + int rawbuf_idx; + size_t mbclen; + wchar_t wc = WEOF; + + /* Skip the characters which are not necessary to check. */ + for (rawbuf_idx = pstr->raw_mbs_idx + pstr->valid_raw_len; + rawbuf_idx < new_raw_idx;) + { + int remain_len; + remain_len = pstr->len - rawbuf_idx; + prev_st = pstr->cur_state; + mbclen = __mbrtowc (&wc, (const char *) pstr->raw_mbs + rawbuf_idx, + remain_len, &pstr->cur_state); + if (BE (mbclen == (size_t) -2 || mbclen == (size_t) -1 || mbclen == 0, 0)) + { + /* We treat these cases as a single byte character. */ + if (mbclen == 0 || remain_len == 0) + wc = L'\0'; + else + wc = *(unsigned char *) (pstr->raw_mbs + rawbuf_idx); + mbclen = 1; + pstr->cur_state = prev_st; + } + /* Then proceed the next character. */ + rawbuf_idx += mbclen; + } + *last_wc = (wint_t) wc; + return rawbuf_idx; +} +#endif /* RE_ENABLE_I18N */ + +/* Build the buffer PSTR->MBS, and apply the translation if we need. + This function is used in case of REG_ICASE. */ + +static void +internal_function +build_upper_buffer (re_string_t *pstr) +{ + int char_idx, end_idx; + end_idx = (pstr->bufs_len > pstr->len) ? pstr->len : pstr->bufs_len; + + for (char_idx = pstr->valid_len; char_idx < end_idx; ++char_idx) + { + int ch = pstr->raw_mbs[pstr->raw_mbs_idx + char_idx]; + if (BE (pstr->trans != NULL, 0)) + ch = pstr->trans[ch]; + if (islower (ch)) + pstr->mbs[char_idx] = toupper (ch); + else + pstr->mbs[char_idx] = ch; + } + pstr->valid_len = char_idx; + pstr->valid_raw_len = char_idx; +} + +/* Apply TRANS to the buffer in PSTR. */ + +static void +internal_function +re_string_translate_buffer (re_string_t *pstr) +{ + int buf_idx, end_idx; + end_idx = (pstr->bufs_len > pstr->len) ? pstr->len : pstr->bufs_len; + + for (buf_idx = pstr->valid_len; buf_idx < end_idx; ++buf_idx) + { + int ch = pstr->raw_mbs[pstr->raw_mbs_idx + buf_idx]; + pstr->mbs[buf_idx] = pstr->trans[ch]; + } + + pstr->valid_len = buf_idx; + pstr->valid_raw_len = buf_idx; +} + +/* This function re-construct the buffers. + Concretely, convert to wide character in case of pstr->mb_cur_max > 1, + convert to upper case in case of REG_ICASE, apply translation. */ + +static reg_errcode_t +internal_function +re_string_reconstruct (re_string_t *pstr, int idx, int eflags) +{ + int offset = idx - pstr->raw_mbs_idx; + if (BE (offset < 0, 0)) + { + /* Reset buffer. */ +#ifdef RE_ENABLE_I18N + if (pstr->mb_cur_max > 1) + memset (&pstr->cur_state, '\0', sizeof (mbstate_t)); +#endif /* RE_ENABLE_I18N */ + pstr->len = pstr->raw_len; + pstr->stop = pstr->raw_stop; + pstr->valid_len = 0; + pstr->raw_mbs_idx = 0; + pstr->valid_raw_len = 0; + pstr->offsets_needed = 0; + pstr->tip_context = ((eflags & REG_NOTBOL) ? CONTEXT_BEGBUF + : CONTEXT_NEWLINE | CONTEXT_BEGBUF); + if (!pstr->mbs_allocated) + pstr->mbs = (unsigned char *) pstr->raw_mbs; + offset = idx; + } + + if (BE (offset != 0, 1)) + { + /* Should the already checked characters be kept? */ + if (BE (offset < pstr->valid_raw_len, 1)) + { + /* Yes, move them to the front of the buffer. */ +#ifdef RE_ENABLE_I18N + if (BE (pstr->offsets_needed, 0)) + { + int low = 0, high = pstr->valid_len, mid; + do + { + mid = (high + low) / 2; + if (pstr->offsets[mid] > offset) + high = mid; + else if (pstr->offsets[mid] < offset) + low = mid + 1; + else + break; + } + while (low < high); + if (pstr->offsets[mid] < offset) + ++mid; + pstr->tip_context = re_string_context_at (pstr, mid - 1, + eflags); + /* This can be quite complicated, so handle specially + only the common and easy case where the character with + different length representation of lower and upper + case is present at or after offset. */ + if (pstr->valid_len > offset + && mid == offset && pstr->offsets[mid] == offset) + { + memmove (pstr->wcs, pstr->wcs + offset, + (pstr->valid_len - offset) * sizeof (wint_t)); + memmove (pstr->mbs, pstr->mbs + offset, pstr->valid_len - offset); + pstr->valid_len -= offset; + pstr->valid_raw_len -= offset; + for (low = 0; low < pstr->valid_len; low++) + pstr->offsets[low] = pstr->offsets[low + offset] - offset; + } + else + { + /* Otherwise, just find out how long the partial multibyte + character at offset is and fill it with WEOF/255. */ + pstr->len = pstr->raw_len - idx + offset; + pstr->stop = pstr->raw_stop - idx + offset; + pstr->offsets_needed = 0; + while (mid > 0 && pstr->offsets[mid - 1] == offset) + --mid; + while (mid < pstr->valid_len) + if (pstr->wcs[mid] != WEOF) + break; + else + ++mid; + if (mid == pstr->valid_len) + pstr->valid_len = 0; + else + { + pstr->valid_len = pstr->offsets[mid] - offset; + if (pstr->valid_len) + { + for (low = 0; low < pstr->valid_len; ++low) + pstr->wcs[low] = WEOF; + memset (pstr->mbs, 255, pstr->valid_len); + } + } + pstr->valid_raw_len = pstr->valid_len; + } + } + else +#endif + { + pstr->tip_context = re_string_context_at (pstr, offset - 1, + eflags); +#ifdef RE_ENABLE_I18N + if (pstr->mb_cur_max > 1) + memmove (pstr->wcs, pstr->wcs + offset, + (pstr->valid_len - offset) * sizeof (wint_t)); +#endif /* RE_ENABLE_I18N */ + if (BE (pstr->mbs_allocated, 0)) + memmove (pstr->mbs, pstr->mbs + offset, + pstr->valid_len - offset); + pstr->valid_len -= offset; + pstr->valid_raw_len -= offset; +#if DEBUG + assert (pstr->valid_len > 0); +#endif + } + } + else + { +#ifdef RE_ENABLE_I18N + /* No, skip all characters until IDX. */ + int prev_valid_len = pstr->valid_len; + + if (BE (pstr->offsets_needed, 0)) + { + pstr->len = pstr->raw_len - idx + offset; + pstr->stop = pstr->raw_stop - idx + offset; + pstr->offsets_needed = 0; + } +#endif + pstr->valid_len = 0; +#ifdef RE_ENABLE_I18N + if (pstr->mb_cur_max > 1) + { + int wcs_idx; + wint_t wc = WEOF; + + if (pstr->is_utf8) + { + const unsigned char *raw, *p, *q, *end; + + /* Special case UTF-8. Multi-byte chars start with any + byte other than 0x80 - 0xbf. */ + raw = pstr->raw_mbs + pstr->raw_mbs_idx; + end = raw + (offset - pstr->mb_cur_max); + if (end < pstr->raw_mbs) + end = pstr->raw_mbs; + p = raw + offset - 1; +#ifdef _LIBC + /* We know the wchar_t encoding is UCS4, so for the simple + case, ASCII characters, skip the conversion step. */ + if (isascii (*p) && BE (pstr->trans == NULL, 1)) + { + memset (&pstr->cur_state, '\0', sizeof (mbstate_t)); + /* pstr->valid_len = 0; */ + wc = (wchar_t) *p; + } + else +#endif + for (; p >= end; --p) + if ((*p & 0xc0) != 0x80) + { + mbstate_t cur_state; + wchar_t wc2; + int mlen = raw + pstr->len - p; + unsigned char buf[6]; + size_t mbclen; + + q = p; + if (BE (pstr->trans != NULL, 0)) + { + int i = mlen < 6 ? mlen : 6; + while (--i >= 0) + buf[i] = pstr->trans[p[i]]; + q = buf; + } + /* XXX Don't use mbrtowc, we know which conversion + to use (UTF-8 -> UCS4). */ + memset (&cur_state, 0, sizeof (cur_state)); + mbclen = __mbrtowc (&wc2, (const char *) p, mlen, + &cur_state); + if (raw + offset - p <= mbclen + && mbclen < (size_t) -2) + { + memset (&pstr->cur_state, '\0', + sizeof (mbstate_t)); + pstr->valid_len = mbclen - (raw + offset - p); + wc = wc2; + } + break; + } + } + + if (wc == WEOF) + pstr->valid_len = re_string_skip_chars (pstr, idx, &wc) - idx; + if (wc == WEOF) + pstr->tip_context + = re_string_context_at (pstr, prev_valid_len - 1, eflags); + else + pstr->tip_context = ((BE (pstr->word_ops_used != 0, 0) + && IS_WIDE_WORD_CHAR (wc)) + ? CONTEXT_WORD + : ((IS_WIDE_NEWLINE (wc) + && pstr->newline_anchor) + ? CONTEXT_NEWLINE : 0)); + if (BE (pstr->valid_len, 0)) + { + for (wcs_idx = 0; wcs_idx < pstr->valid_len; ++wcs_idx) + pstr->wcs[wcs_idx] = WEOF; + if (pstr->mbs_allocated) + memset (pstr->mbs, 255, pstr->valid_len); + } + pstr->valid_raw_len = pstr->valid_len; + } + else +#endif /* RE_ENABLE_I18N */ + { + int c = pstr->raw_mbs[pstr->raw_mbs_idx + offset - 1]; + pstr->valid_raw_len = 0; + if (pstr->trans) + c = pstr->trans[c]; + pstr->tip_context = (bitset_contain (pstr->word_char, c) + ? CONTEXT_WORD + : ((IS_NEWLINE (c) && pstr->newline_anchor) + ? CONTEXT_NEWLINE : 0)); + } + } + if (!BE (pstr->mbs_allocated, 0)) + pstr->mbs += offset; + } + pstr->raw_mbs_idx = idx; + pstr->len -= offset; + pstr->stop -= offset; + + /* Then build the buffers. */ +#ifdef RE_ENABLE_I18N + if (pstr->mb_cur_max > 1) + { + if (pstr->icase) + { + reg_errcode_t ret = build_wcs_upper_buffer (pstr); + if (BE (ret != REG_NOERROR, 0)) + return ret; + } + else + build_wcs_buffer (pstr); + } + else +#endif /* RE_ENABLE_I18N */ + if (BE (pstr->mbs_allocated, 0)) + { + if (pstr->icase) + build_upper_buffer (pstr); + else if (pstr->trans != NULL) + re_string_translate_buffer (pstr); + } + else + pstr->valid_len = pstr->len; + + pstr->cur_idx = 0; + return REG_NOERROR; +} + +static unsigned char +internal_function __attribute ((pure)) +re_string_peek_byte_case (const re_string_t *pstr, int idx) +{ + int ch, off; + + /* Handle the common (easiest) cases first. */ + if (BE (!pstr->mbs_allocated, 1)) + return re_string_peek_byte (pstr, idx); + +#ifdef RE_ENABLE_I18N + if (pstr->mb_cur_max > 1 + && ! re_string_is_single_byte_char (pstr, pstr->cur_idx + idx)) + return re_string_peek_byte (pstr, idx); +#endif + + off = pstr->cur_idx + idx; +#ifdef RE_ENABLE_I18N + if (pstr->offsets_needed) + off = pstr->offsets[off]; +#endif + + ch = pstr->raw_mbs[pstr->raw_mbs_idx + off]; + +#ifdef RE_ENABLE_I18N + /* Ensure that e.g. for tr_TR.UTF-8 BACKSLASH DOTLESS SMALL LETTER I + this function returns CAPITAL LETTER I instead of first byte of + DOTLESS SMALL LETTER I. The latter would confuse the parser, + since peek_byte_case doesn't advance cur_idx in any way. */ + if (pstr->offsets_needed && !isascii (ch)) + return re_string_peek_byte (pstr, idx); +#endif + + return ch; +} + +static unsigned char +internal_function __attribute ((pure)) +re_string_fetch_byte_case (re_string_t *pstr) +{ + if (BE (!pstr->mbs_allocated, 1)) + return re_string_fetch_byte (pstr); + +#ifdef RE_ENABLE_I18N + if (pstr->offsets_needed) + { + int off, ch; + + /* For tr_TR.UTF-8 [[:islower:]] there is + [[: CAPITAL LETTER I WITH DOT lower:]] in mbs. Skip + in that case the whole multi-byte character and return + the original letter. On the other side, with + [[: DOTLESS SMALL LETTER I return [[:I, as doing + anything else would complicate things too much. */ + + if (!re_string_first_byte (pstr, pstr->cur_idx)) + return re_string_fetch_byte (pstr); + + off = pstr->offsets[pstr->cur_idx]; + ch = pstr->raw_mbs[pstr->raw_mbs_idx + off]; + + if (! isascii (ch)) + return re_string_fetch_byte (pstr); + + re_string_skip_bytes (pstr, + re_string_char_size_at (pstr, pstr->cur_idx)); + return ch; + } +#endif + + return pstr->raw_mbs[pstr->raw_mbs_idx + pstr->cur_idx++]; +} + +static void +internal_function +re_string_destruct (re_string_t *pstr) +{ +#ifdef RE_ENABLE_I18N + re_free (pstr->wcs); + re_free (pstr->offsets); +#endif /* RE_ENABLE_I18N */ + if (pstr->mbs_allocated) + re_free (pstr->mbs); +} + +/* Return the context at IDX in INPUT. */ + +static unsigned int +internal_function +re_string_context_at (const re_string_t *input, int idx, int eflags) +{ + int c; + if (BE (idx < 0, 0)) + /* In this case, we use the value stored in input->tip_context, + since we can't know the character in input->mbs[-1] here. */ + return input->tip_context; + if (BE (idx == input->len, 0)) + return ((eflags & REG_NOTEOL) ? CONTEXT_ENDBUF + : CONTEXT_NEWLINE | CONTEXT_ENDBUF); +#ifdef RE_ENABLE_I18N + if (input->mb_cur_max > 1) + { + wint_t wc; + int wc_idx = idx; + while(input->wcs[wc_idx] == WEOF) + { +#ifdef DEBUG + /* It must not happen. */ + assert (wc_idx >= 0); +#endif + --wc_idx; + if (wc_idx < 0) + return input->tip_context; + } + wc = input->wcs[wc_idx]; + if (BE (input->word_ops_used != 0, 0) && IS_WIDE_WORD_CHAR (wc)) + return CONTEXT_WORD; + return (IS_WIDE_NEWLINE (wc) && input->newline_anchor + ? CONTEXT_NEWLINE : 0); + } + else +#endif + { + c = re_string_byte_at (input, idx); + if (bitset_contain (input->word_char, c)) + return CONTEXT_WORD; + return IS_NEWLINE (c) && input->newline_anchor ? CONTEXT_NEWLINE : 0; + } +} + +/* Functions for set operation. */ + +static reg_errcode_t +internal_function +re_node_set_alloc (re_node_set *set, int size) +{ + set->alloc = size; + set->nelem = 0; + set->elems = re_malloc (int, size); + if (BE (set->elems == NULL, 0)) + return REG_ESPACE; + return REG_NOERROR; +} + +static reg_errcode_t +internal_function +re_node_set_init_1 (re_node_set *set, int elem) +{ + set->alloc = 1; + set->nelem = 1; + set->elems = re_malloc (int, 1); + if (BE (set->elems == NULL, 0)) + { + set->alloc = set->nelem = 0; + return REG_ESPACE; + } + set->elems[0] = elem; + return REG_NOERROR; +} + +static reg_errcode_t +internal_function +re_node_set_init_2 (re_node_set *set, int elem1, int elem2) +{ + set->alloc = 2; + set->elems = re_malloc (int, 2); + if (BE (set->elems == NULL, 0)) + return REG_ESPACE; + if (elem1 == elem2) + { + set->nelem = 1; + set->elems[0] = elem1; + } + else + { + set->nelem = 2; + if (elem1 < elem2) + { + set->elems[0] = elem1; + set->elems[1] = elem2; + } + else + { + set->elems[0] = elem2; + set->elems[1] = elem1; + } + } + return REG_NOERROR; +} + +static reg_errcode_t +internal_function +re_node_set_init_copy (re_node_set *dest, const re_node_set *src) +{ + dest->nelem = src->nelem; + if (src->nelem > 0) + { + dest->alloc = dest->nelem; + dest->elems = re_malloc (int, dest->alloc); + if (BE (dest->elems == NULL, 0)) + { + dest->alloc = dest->nelem = 0; + return REG_ESPACE; + } + memcpy (dest->elems, src->elems, src->nelem * sizeof (int)); + } + else + re_node_set_init_empty (dest); + return REG_NOERROR; +} + +/* Calculate the intersection of the sets SRC1 and SRC2. And merge it to + DEST. Return value indicate the error code or REG_NOERROR if succeeded. + Note: We assume dest->elems is NULL, when dest->alloc is 0. */ + +static reg_errcode_t +internal_function +re_node_set_add_intersect (re_node_set *dest, const re_node_set *src1, + const re_node_set *src2) +{ + int i1, i2, is, id, delta, sbase; + if (src1->nelem == 0 || src2->nelem == 0) + return REG_NOERROR; + + /* We need dest->nelem + 2 * elems_in_intersection; this is a + conservative estimate. */ + if (src1->nelem + src2->nelem + dest->nelem > dest->alloc) + { + int new_alloc = src1->nelem + src2->nelem + dest->alloc; + int *new_elems = re_realloc (dest->elems, int, new_alloc); + if (BE (new_elems == NULL, 0)) + return REG_ESPACE; + dest->elems = new_elems; + dest->alloc = new_alloc; + } + + /* Find the items in the intersection of SRC1 and SRC2, and copy + into the top of DEST those that are not already in DEST itself. */ + sbase = dest->nelem + src1->nelem + src2->nelem; + i1 = src1->nelem - 1; + i2 = src2->nelem - 1; + id = dest->nelem - 1; + for (;;) + { + if (src1->elems[i1] == src2->elems[i2]) + { + /* Try to find the item in DEST. Maybe we could binary search? */ + while (id >= 0 && dest->elems[id] > src1->elems[i1]) + --id; + + if (id < 0 || dest->elems[id] != src1->elems[i1]) + dest->elems[--sbase] = src1->elems[i1]; + + if (--i1 < 0 || --i2 < 0) + break; + } + + /* Lower the highest of the two items. */ + else if (src1->elems[i1] < src2->elems[i2]) + { + if (--i2 < 0) + break; + } + else + { + if (--i1 < 0) + break; + } + } + + id = dest->nelem - 1; + is = dest->nelem + src1->nelem + src2->nelem - 1; + delta = is - sbase + 1; + + /* Now copy. When DELTA becomes zero, the remaining + DEST elements are already in place; this is more or + less the same loop that is in re_node_set_merge. */ + dest->nelem += delta; + if (delta > 0 && id >= 0) + for (;;) + { + if (dest->elems[is] > dest->elems[id]) + { + /* Copy from the top. */ + dest->elems[id + delta--] = dest->elems[is--]; + if (delta == 0) + break; + } + else + { + /* Slide from the bottom. */ + dest->elems[id + delta] = dest->elems[id]; + if (--id < 0) + break; + } + } + + /* Copy remaining SRC elements. */ + memcpy (dest->elems, dest->elems + sbase, delta * sizeof (int)); + + return REG_NOERROR; +} + +/* Calculate the union set of the sets SRC1 and SRC2. And store it to + DEST. Return value indicate the error code or REG_NOERROR if succeeded. */ + +static reg_errcode_t +internal_function +re_node_set_init_union (re_node_set *dest, const re_node_set *src1, + const re_node_set *src2) +{ + int i1, i2, id; + if (src1 != NULL && src1->nelem > 0 && src2 != NULL && src2->nelem > 0) + { + dest->alloc = src1->nelem + src2->nelem; + dest->elems = re_malloc (int, dest->alloc); + if (BE (dest->elems == NULL, 0)) + return REG_ESPACE; + } + else + { + if (src1 != NULL && src1->nelem > 0) + return re_node_set_init_copy (dest, src1); + else if (src2 != NULL && src2->nelem > 0) + return re_node_set_init_copy (dest, src2); + else + re_node_set_init_empty (dest); + return REG_NOERROR; + } + for (i1 = i2 = id = 0 ; i1 < src1->nelem && i2 < src2->nelem ;) + { + if (src1->elems[i1] > src2->elems[i2]) + { + dest->elems[id++] = src2->elems[i2++]; + continue; + } + if (src1->elems[i1] == src2->elems[i2]) + ++i2; + dest->elems[id++] = src1->elems[i1++]; + } + if (i1 < src1->nelem) + { + memcpy (dest->elems + id, src1->elems + i1, + (src1->nelem - i1) * sizeof (int)); + id += src1->nelem - i1; + } + else if (i2 < src2->nelem) + { + memcpy (dest->elems + id, src2->elems + i2, + (src2->nelem - i2) * sizeof (int)); + id += src2->nelem - i2; + } + dest->nelem = id; + return REG_NOERROR; +} + +/* Calculate the union set of the sets DEST and SRC. And store it to + DEST. Return value indicate the error code or REG_NOERROR if succeeded. */ + +static reg_errcode_t +internal_function +re_node_set_merge (re_node_set *dest, const re_node_set *src) +{ + int is, id, sbase, delta; + if (src == NULL || src->nelem == 0) + return REG_NOERROR; + if (dest->alloc < 2 * src->nelem + dest->nelem) + { + int new_alloc = 2 * (src->nelem + dest->alloc); + int *new_buffer = re_realloc (dest->elems, int, new_alloc); + if (BE (new_buffer == NULL, 0)) + return REG_ESPACE; + dest->elems = new_buffer; + dest->alloc = new_alloc; + } + + if (BE (dest->nelem == 0, 0)) + { + dest->nelem = src->nelem; + memcpy (dest->elems, src->elems, src->nelem * sizeof (int)); + return REG_NOERROR; + } + + /* Copy into the top of DEST the items of SRC that are not + found in DEST. Maybe we could binary search in DEST? */ + for (sbase = dest->nelem + 2 * src->nelem, + is = src->nelem - 1, id = dest->nelem - 1; is >= 0 && id >= 0; ) + { + if (dest->elems[id] == src->elems[is]) + is--, id--; + else if (dest->elems[id] < src->elems[is]) + dest->elems[--sbase] = src->elems[is--]; + else /* if (dest->elems[id] > src->elems[is]) */ + --id; + } + + if (is >= 0) + { + /* If DEST is exhausted, the remaining items of SRC must be unique. */ + sbase -= is + 1; + memcpy (dest->elems + sbase, src->elems, (is + 1) * sizeof (int)); + } + + id = dest->nelem - 1; + is = dest->nelem + 2 * src->nelem - 1; + delta = is - sbase + 1; + if (delta == 0) + return REG_NOERROR; + + /* Now copy. When DELTA becomes zero, the remaining + DEST elements are already in place. */ + dest->nelem += delta; + for (;;) + { + if (dest->elems[is] > dest->elems[id]) + { + /* Copy from the top. */ + dest->elems[id + delta--] = dest->elems[is--]; + if (delta == 0) + break; + } + else + { + /* Slide from the bottom. */ + dest->elems[id + delta] = dest->elems[id]; + if (--id < 0) + { + /* Copy remaining SRC elements. */ + memcpy (dest->elems, dest->elems + sbase, + delta * sizeof (int)); + break; + } + } + } + + return REG_NOERROR; +} + +/* Insert the new element ELEM to the re_node_set* SET. + SET should not already have ELEM. + return -1 if an error is occurred, return 1 otherwise. */ + +static int +internal_function +re_node_set_insert (re_node_set *set, int elem) +{ + int idx; + /* In case the set is empty. */ + if (set->alloc == 0) + { + if (BE (re_node_set_init_1 (set, elem) == REG_NOERROR, 1)) + return 1; + else + return -1; + } + + if (BE (set->nelem, 0) == 0) + { + /* We already guaranteed above that set->alloc != 0. */ + set->elems[0] = elem; + ++set->nelem; + return 1; + } + + /* Realloc if we need. */ + if (set->alloc == set->nelem) + { + int *new_elems; + set->alloc = set->alloc * 2; + new_elems = re_realloc (set->elems, int, set->alloc); + if (BE (new_elems == NULL, 0)) + return -1; + set->elems = new_elems; + } + + /* Move the elements which follows the new element. Test the + first element separately to skip a check in the inner loop. */ + if (elem < set->elems[0]) + { + idx = 0; + for (idx = set->nelem; idx > 0; idx--) + set->elems[idx] = set->elems[idx - 1]; + } + else + { + for (idx = set->nelem; set->elems[idx - 1] > elem; idx--) + set->elems[idx] = set->elems[idx - 1]; + } + + /* Insert the new element. */ + set->elems[idx] = elem; + ++set->nelem; + return 1; +} + +/* Insert the new element ELEM to the re_node_set* SET. + SET should not already have any element greater than or equal to ELEM. + Return -1 if an error is occurred, return 1 otherwise. */ + +static int +internal_function +re_node_set_insert_last (re_node_set *set, int elem) +{ + /* Realloc if we need. */ + if (set->alloc == set->nelem) + { + int *new_elems; + set->alloc = (set->alloc + 1) * 2; + new_elems = re_realloc (set->elems, int, set->alloc); + if (BE (new_elems == NULL, 0)) + return -1; + set->elems = new_elems; + } + + /* Insert the new element. */ + set->elems[set->nelem++] = elem; + return 1; +} + +/* Compare two node sets SET1 and SET2. + return 1 if SET1 and SET2 are equivalent, return 0 otherwise. */ + +static int +internal_function __attribute ((pure)) +re_node_set_compare (const re_node_set *set1, const re_node_set *set2) +{ + int i; + if (set1 == NULL || set2 == NULL || set1->nelem != set2->nelem) + return 0; + for (i = set1->nelem ; --i >= 0 ; ) + if (set1->elems[i] != set2->elems[i]) + return 0; + return 1; +} + +/* Return (idx + 1) if SET contains the element ELEM, return 0 otherwise. */ + +static int +internal_function __attribute ((pure)) +re_node_set_contains (const re_node_set *set, int elem) +{ + unsigned int idx, right, mid; + if (set->nelem <= 0) + return 0; + + /* Binary search the element. */ + idx = 0; + right = set->nelem - 1; + while (idx < right) + { + mid = (idx + right) / 2; + if (set->elems[mid] < elem) + idx = mid + 1; + else + right = mid; + } + return set->elems[idx] == elem ? idx + 1 : 0; +} + +static void +internal_function +re_node_set_remove_at (re_node_set *set, int idx) +{ + if (idx < 0 || idx >= set->nelem) + return; + --set->nelem; + for (; idx < set->nelem; idx++) + set->elems[idx] = set->elems[idx + 1]; +} + + +/* Add the token TOKEN to dfa->nodes, and return the index of the token. + Or return -1, if an error will be occurred. */ + +static int +internal_function +re_dfa_add_node (re_dfa_t *dfa, re_token_t token) +{ +#ifdef RE_ENABLE_I18N + int type = token.type; +#endif + if (BE (dfa->nodes_len >= dfa->nodes_alloc, 0)) + { + size_t new_nodes_alloc = dfa->nodes_alloc * 2; + int *new_nexts, *new_indices; + re_node_set *new_edests, *new_eclosures; + re_token_t *new_nodes; + + /* Avoid overflows. */ + if (BE (new_nodes_alloc < dfa->nodes_alloc, 0)) + return -1; + + new_nodes = re_realloc (dfa->nodes, re_token_t, new_nodes_alloc); + if (BE (new_nodes == NULL, 0)) + return -1; + dfa->nodes = new_nodes; + new_nexts = re_realloc (dfa->nexts, int, new_nodes_alloc); + new_indices = re_realloc (dfa->org_indices, int, new_nodes_alloc); + new_edests = re_realloc (dfa->edests, re_node_set, new_nodes_alloc); + new_eclosures = re_realloc (dfa->eclosures, re_node_set, new_nodes_alloc); + if (BE (new_nexts == NULL || new_indices == NULL + || new_edests == NULL || new_eclosures == NULL, 0)) + return -1; + dfa->nexts = new_nexts; + dfa->org_indices = new_indices; + dfa->edests = new_edests; + dfa->eclosures = new_eclosures; + dfa->nodes_alloc = new_nodes_alloc; + } + dfa->nodes[dfa->nodes_len] = token; + dfa->nodes[dfa->nodes_len].constraint = 0; +#ifdef RE_ENABLE_I18N + dfa->nodes[dfa->nodes_len].accept_mb = + (type == OP_PERIOD && dfa->mb_cur_max > 1) || type == COMPLEX_BRACKET; +#endif + dfa->nexts[dfa->nodes_len] = -1; + re_node_set_init_empty (dfa->edests + dfa->nodes_len); + re_node_set_init_empty (dfa->eclosures + dfa->nodes_len); + return dfa->nodes_len++; +} + +static inline unsigned int +internal_function +calc_state_hash (const re_node_set *nodes, unsigned int context) +{ + unsigned int hash = nodes->nelem + context; + int i; + for (i = 0 ; i < nodes->nelem ; i++) + hash += nodes->elems[i]; + return hash; +} + +/* Search for the state whose node_set is equivalent to NODES. + Return the pointer to the state, if we found it in the DFA. + Otherwise create the new one and return it. In case of an error + return NULL and set the error code in ERR. + Note: - We assume NULL as the invalid state, then it is possible that + return value is NULL and ERR is REG_NOERROR. + - We never return non-NULL value in case of any errors, it is for + optimization. */ + +static re_dfastate_t * +internal_function +re_acquire_state (reg_errcode_t *err, const re_dfa_t *dfa, + const re_node_set *nodes) +{ + unsigned int hash; + re_dfastate_t *new_state; + struct re_state_table_entry *spot; + int i; + if (BE (nodes->nelem == 0, 0)) + { + *err = REG_NOERROR; + return NULL; + } + hash = calc_state_hash (nodes, 0); + spot = dfa->state_table + (hash & dfa->state_hash_mask); + + for (i = 0 ; i < spot->num ; i++) + { + re_dfastate_t *state = spot->array[i]; + if (hash != state->hash) + continue; + if (re_node_set_compare (&state->nodes, nodes)) + return state; + } + + /* There are no appropriate state in the dfa, create the new one. */ + new_state = create_ci_newstate (dfa, nodes, hash); + if (BE (new_state == NULL, 0)) + *err = REG_ESPACE; + + return new_state; +} + +/* Search for the state whose node_set is equivalent to NODES and + whose context is equivalent to CONTEXT. + Return the pointer to the state, if we found it in the DFA. + Otherwise create the new one and return it. In case of an error + return NULL and set the error code in ERR. + Note: - We assume NULL as the invalid state, then it is possible that + return value is NULL and ERR is REG_NOERROR. + - We never return non-NULL value in case of any errors, it is for + optimization. */ + +static re_dfastate_t * +internal_function +re_acquire_state_context (reg_errcode_t *err, const re_dfa_t *dfa, + const re_node_set *nodes, unsigned int context) +{ + unsigned int hash; + re_dfastate_t *new_state; + struct re_state_table_entry *spot; + int i; + if (nodes->nelem == 0) + { + *err = REG_NOERROR; + return NULL; + } + hash = calc_state_hash (nodes, context); + spot = dfa->state_table + (hash & dfa->state_hash_mask); + + for (i = 0 ; i < spot->num ; i++) + { + re_dfastate_t *state = spot->array[i]; + if (state->hash == hash + && state->context == context + && re_node_set_compare (state->entrance_nodes, nodes)) + return state; + } + /* There are no appropriate state in `dfa', create the new one. */ + new_state = create_cd_newstate (dfa, nodes, context, hash); + if (BE (new_state == NULL, 0)) + *err = REG_ESPACE; + + return new_state; +} + +/* Finish initialization of the new state NEWSTATE, and using its hash value + HASH put in the appropriate bucket of DFA's state table. Return value + indicates the error code if failed. */ + +static reg_errcode_t +register_state (const re_dfa_t *dfa, re_dfastate_t *newstate, + unsigned int hash) +{ + struct re_state_table_entry *spot; + reg_errcode_t err; + int i; + + newstate->hash = hash; + err = re_node_set_alloc (&newstate->non_eps_nodes, newstate->nodes.nelem); + if (BE (err != REG_NOERROR, 0)) + return REG_ESPACE; + for (i = 0; i < newstate->nodes.nelem; i++) + { + int elem = newstate->nodes.elems[i]; + if (!IS_EPSILON_NODE (dfa->nodes[elem].type)) + re_node_set_insert_last (&newstate->non_eps_nodes, elem); + } + + spot = dfa->state_table + (hash & dfa->state_hash_mask); + if (BE (spot->alloc <= spot->num, 0)) + { + int new_alloc = 2 * spot->num + 2; + re_dfastate_t **new_array = re_realloc (spot->array, re_dfastate_t *, + new_alloc); + if (BE (new_array == NULL, 0)) + return REG_ESPACE; + spot->array = new_array; + spot->alloc = new_alloc; + } + spot->array[spot->num++] = newstate; + return REG_NOERROR; +} + +static void +free_state (re_dfastate_t *state) +{ + re_node_set_free (&state->non_eps_nodes); + re_node_set_free (&state->inveclosure); + if (state->entrance_nodes != &state->nodes) + { + re_node_set_free (state->entrance_nodes); + re_free (state->entrance_nodes); + } + re_node_set_free (&state->nodes); + re_free (state->word_trtable); + re_free (state->trtable); + re_free (state); +} + +/* Create the new state which is independ of contexts. + Return the new state if succeeded, otherwise return NULL. */ + +static re_dfastate_t * +internal_function +create_ci_newstate (const re_dfa_t *dfa, const re_node_set *nodes, + unsigned int hash) +{ + int i; + reg_errcode_t err; + re_dfastate_t *newstate; + + newstate = (re_dfastate_t *) calloc (sizeof (re_dfastate_t), 1); + if (BE (newstate == NULL, 0)) + return NULL; + err = re_node_set_init_copy (&newstate->nodes, nodes); + if (BE (err != REG_NOERROR, 0)) + { + re_free (newstate); + return NULL; + } + + newstate->entrance_nodes = &newstate->nodes; + for (i = 0 ; i < nodes->nelem ; i++) + { + re_token_t *node = dfa->nodes + nodes->elems[i]; + re_token_type_t type = node->type; + if (type == CHARACTER && !node->constraint) + continue; +#ifdef RE_ENABLE_I18N + newstate->accept_mb |= node->accept_mb; +#endif /* RE_ENABLE_I18N */ + + /* If the state has the halt node, the state is a halt state. */ + if (type == END_OF_RE) + newstate->halt = 1; + else if (type == OP_BACK_REF) + newstate->has_backref = 1; + else if (type == ANCHOR || node->constraint) + newstate->has_constraint = 1; + } + err = register_state (dfa, newstate, hash); + if (BE (err != REG_NOERROR, 0)) + { + free_state (newstate); + newstate = NULL; + } + return newstate; +} + +/* Create the new state which is depend on the context CONTEXT. + Return the new state if succeeded, otherwise return NULL. */ + +static re_dfastate_t * +internal_function +create_cd_newstate (const re_dfa_t *dfa, const re_node_set *nodes, + unsigned int context, unsigned int hash) +{ + int i, nctx_nodes = 0; + reg_errcode_t err; + re_dfastate_t *newstate; + + newstate = (re_dfastate_t *) calloc (sizeof (re_dfastate_t), 1); + if (BE (newstate == NULL, 0)) + return NULL; + err = re_node_set_init_copy (&newstate->nodes, nodes); + if (BE (err != REG_NOERROR, 0)) + { + re_free (newstate); + return NULL; + } + + newstate->context = context; + newstate->entrance_nodes = &newstate->nodes; + + for (i = 0 ; i < nodes->nelem ; i++) + { + re_token_t *node = dfa->nodes + nodes->elems[i]; + re_token_type_t type = node->type; + unsigned int constraint = node->constraint; + + if (type == CHARACTER && !constraint) + continue; +#ifdef RE_ENABLE_I18N + newstate->accept_mb |= node->accept_mb; +#endif /* RE_ENABLE_I18N */ + + /* If the state has the halt node, the state is a halt state. */ + if (type == END_OF_RE) + newstate->halt = 1; + else if (type == OP_BACK_REF) + newstate->has_backref = 1; + + if (constraint) + { + if (newstate->entrance_nodes == &newstate->nodes) + { + newstate->entrance_nodes = re_malloc (re_node_set, 1); + if (BE (newstate->entrance_nodes == NULL, 0)) + { + free_state (newstate); + return NULL; + } + re_node_set_init_copy (newstate->entrance_nodes, nodes); + nctx_nodes = 0; + newstate->has_constraint = 1; + } + + if (NOT_SATISFY_PREV_CONSTRAINT (constraint,context)) + { + re_node_set_remove_at (&newstate->nodes, i - nctx_nodes); + ++nctx_nodes; + } + } + } + err = register_state (dfa, newstate, hash); + if (BE (err != REG_NOERROR, 0)) + { + free_state (newstate); + newstate = NULL; + } + return newstate; +} Modified: ctags/gnu_regex/regex_internal.h 773 lines changed, 773 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,773 @@ +/* Extended regular expression matching and search library. + Copyright (C) 2002-2005, 2007, 2008 Free Software Foundation, Inc. + This file is part of the GNU C Library. + Contributed by Isamu Hasegawa <isamu(a)yamato.ibm.com>. + + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. + + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, write to the Free + Software Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA + 02111-1307 USA. */ + +#ifndef _REGEX_INTERNAL_H +#define _REGEX_INTERNAL_H 1 + +#include <assert.h> +#include <ctype.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> + +#if defined HAVE_LANGINFO_H || defined HAVE_LANGINFO_CODESET || defined _LIBC +# include <langinfo.h> +#endif +#if defined HAVE_LOCALE_H || defined _LIBC +# include <locale.h> +#endif +#if defined HAVE_WCHAR_H || defined _LIBC +# include <wchar.h> +#endif /* HAVE_WCHAR_H || _LIBC */ +#if defined HAVE_WCTYPE_H || defined _LIBC +# include <wctype.h> +#endif /* HAVE_WCTYPE_H || _LIBC */ +#if defined HAVE_STDBOOL_H || defined _LIBC +# include <stdbool.h> +#endif /* HAVE_STDBOOL_H || _LIBC */ +#if defined HAVE_STDINT_H || defined _LIBC +# include <stdint.h> +#endif /* HAVE_STDINT_H || _LIBC */ +#if defined _LIBC +# include <bits/libc-lock.h> +#else +# define __libc_lock_define(CLASS,NAME) +# define __libc_lock_init(NAME) do { } while (0) +# define __libc_lock_lock(NAME) do { } while (0) +# define __libc_lock_unlock(NAME) do { } while (0) +#endif + +/* In case that the system doesn't have isblank(). */ +#if !defined _LIBC && !defined HAVE_ISBLANK && !defined isblank +# define isblank(ch) ((ch) == ' ' || (ch) == '\t') +#endif + +#ifdef _LIBC +# ifndef _RE_DEFINE_LOCALE_FUNCTIONS +# define _RE_DEFINE_LOCALE_FUNCTIONS 1 +# include <locale/localeinfo.h> +# include <locale/elem-hash.h> +# include <locale/coll-lookup.h> +# endif +#endif + +/* This is for other GNU distributions with internationalized messages. */ +#if (HAVE_LIBINTL_H && ENABLE_NLS) || defined _LIBC +# include <libintl.h> +# ifdef _LIBC +# undef gettext +# define gettext(msgid) \ + INTUSE(__dcgettext) (_libc_intl_domainname, msgid, LC_MESSAGES) +# endif +#else +# define gettext(msgid) (msgid) +#endif + +#ifndef gettext_noop +/* This define is so xgettext can find the internationalizable + strings. */ +# define gettext_noop(String) String +#endif + +/* For loser systems without the definition. */ +#ifndef SIZE_MAX +# define SIZE_MAX ((size_t) -1) +#endif + +#if (defined MB_CUR_MAX && HAVE_LOCALE_H && HAVE_WCTYPE_H && HAVE_WCHAR_H && HAVE_WCRTOMB && HAVE_MBRTOWC && HAVE_WCSCOLL) || _LIBC +# define RE_ENABLE_I18N +#endif + +#if __GNUC__ >= 3 +# define BE(expr, val) __builtin_expect (expr, val) +#else +# define BE(expr, val) (expr) +# define inline +#endif + +/* Number of single byte character. */ +#define SBC_MAX 256 + +#define COLL_ELEM_LEN_MAX 8 + +/* The character which represents newline. */ +#define NEWLINE_CHAR '\n' +#define WIDE_NEWLINE_CHAR L'\n' + +/* Rename to standard API for using out of glibc. */ +#ifndef _LIBC +# define __wctype wctype +# define __iswctype iswctype +# define __btowc btowc +# define __mbrtowc mbrtowc +# define __mempcpy mempcpy +# define __wcrtomb wcrtomb +# define __regfree regfree +# define attribute_hidden +#endif /* not _LIBC */ + +#ifdef __GNUC__ +# define __attribute(arg) __attribute__ (arg) +#else +# define __attribute(arg) +#endif + +extern const char __re_error_msgid[] attribute_hidden; +extern const size_t __re_error_msgid_idx[] attribute_hidden; + +/* An integer used to represent a set of bits. It must be unsigned, + and must be at least as wide as unsigned int. */ +typedef unsigned long int bitset_word_t; +/* All bits set in a bitset_word_t. */ +#define BITSET_WORD_MAX ULONG_MAX +/* Number of bits in a bitset_word_t. */ +#define BITSET_WORD_BITS (sizeof (bitset_word_t) * CHAR_BIT) +/* Number of bitset_word_t in a bit_set. */ +#define BITSET_WORDS (SBC_MAX / BITSET_WORD_BITS) +typedef bitset_word_t bitset_t[BITSET_WORDS]; +typedef bitset_word_t *re_bitset_ptr_t; +typedef const bitset_word_t *re_const_bitset_ptr_t; + +#define bitset_set(set,i) \ + (set[i / BITSET_WORD_BITS] |= (bitset_word_t) 1 << i % BITSET_WORD_BITS) +#define bitset_clear(set,i) \ + (set[i / BITSET_WORD_BITS] &= ~((bitset_word_t) 1 << i % BITSET_WORD_BITS)) +#define bitset_contain(set,i) \ + (set[i / BITSET_WORD_BITS] & ((bitset_word_t) 1 << i % BITSET_WORD_BITS)) +#define bitset_empty(set) memset (set, '\0', sizeof (bitset_t)) +#define bitset_set_all(set) memset (set, '\xff', sizeof (bitset_t)) +#define bitset_copy(dest,src) memcpy (dest, src, sizeof (bitset_t)) + +#define PREV_WORD_CONSTRAINT 0x0001 +#define PREV_NOTWORD_CONSTRAINT 0x0002 +#define NEXT_WORD_CONSTRAINT 0x0004 +#define NEXT_NOTWORD_CONSTRAINT 0x0008 +#define PREV_NEWLINE_CONSTRAINT 0x0010 +#define NEXT_NEWLINE_CONSTRAINT 0x0020 +#define PREV_BEGBUF_CONSTRAINT 0x0040 +#define NEXT_ENDBUF_CONSTRAINT 0x0080 +#define WORD_DELIM_CONSTRAINT 0x0100 +#define NOT_WORD_DELIM_CONSTRAINT 0x0200 + +typedef enum +{ + INSIDE_WORD = PREV_WORD_CONSTRAINT | NEXT_WORD_CONSTRAINT, + WORD_FIRST = PREV_NOTWORD_CONSTRAINT | NEXT_WORD_CONSTRAINT, + WORD_LAST = PREV_WORD_CONSTRAINT | NEXT_NOTWORD_CONSTRAINT, + INSIDE_NOTWORD = PREV_NOTWORD_CONSTRAINT | NEXT_NOTWORD_CONSTRAINT, + LINE_FIRST = PREV_NEWLINE_CONSTRAINT, + LINE_LAST = NEXT_NEWLINE_CONSTRAINT, + BUF_FIRST = PREV_BEGBUF_CONSTRAINT, + BUF_LAST = NEXT_ENDBUF_CONSTRAINT, + WORD_DELIM = WORD_DELIM_CONSTRAINT, + NOT_WORD_DELIM = NOT_WORD_DELIM_CONSTRAINT +} re_context_type; + +typedef struct +{ + int alloc; + int nelem; + int *elems; +} re_node_set; + +typedef enum +{ + NON_TYPE = 0, + + /* Node type, These are used by token, node, tree. */ + CHARACTER = 1, + END_OF_RE = 2, + SIMPLE_BRACKET = 3, + OP_BACK_REF = 4, + OP_PERIOD = 5, +#ifdef RE_ENABLE_I18N + COMPLEX_BRACKET = 6, + OP_UTF8_PERIOD = 7, +#endif /* RE_ENABLE_I18N */ + + /* We define EPSILON_BIT as a macro so that OP_OPEN_SUBEXP is used + when the debugger shows values of this enum type. */ +#define EPSILON_BIT 8 + OP_OPEN_SUBEXP = EPSILON_BIT | 0, + OP_CLOSE_SUBEXP = EPSILON_BIT | 1, + OP_ALT = EPSILON_BIT | 2, + OP_DUP_ASTERISK = EPSILON_BIT | 3, + ANCHOR = EPSILON_BIT | 4, + + /* Tree type, these are used only by tree. */ + CONCAT = 16, + SUBEXP = 17, + + /* Token type, these are used only by token. */ + OP_DUP_PLUS = 18, + OP_DUP_QUESTION, + OP_OPEN_BRACKET, + OP_CLOSE_BRACKET, + OP_CHARSET_RANGE, + OP_OPEN_DUP_NUM, + OP_CLOSE_DUP_NUM, + OP_NON_MATCH_LIST, + OP_OPEN_COLL_ELEM, + OP_CLOSE_COLL_ELEM, + OP_OPEN_EQUIV_CLASS, + OP_CLOSE_EQUIV_CLASS, + OP_OPEN_CHAR_CLASS, + OP_CLOSE_CHAR_CLASS, + OP_WORD, + OP_NOTWORD, + OP_SPACE, + OP_NOTSPACE, + BACK_SLASH + +} re_token_type_t; + +#ifdef RE_ENABLE_I18N +typedef struct +{ + /* Multibyte characters. */ + wchar_t *mbchars; + + /* Collating symbols. */ +# ifdef _LIBC + int32_t *coll_syms; +# endif + + /* Equivalence classes. */ +# ifdef _LIBC + int32_t *equiv_classes; +# endif + + /* Range expressions. */ +# ifdef _LIBC + uint32_t *range_starts; + uint32_t *range_ends; +# else /* not _LIBC */ + wchar_t *range_starts; + wchar_t *range_ends; +# endif /* not _LIBC */ + + /* Character classes. */ + wctype_t *char_classes; + + /* If this character set is the non-matching list. */ + unsigned int non_match : 1; + + /* # of multibyte characters. */ + int nmbchars; + + /* # of collating symbols. */ + int ncoll_syms; + + /* # of equivalence classes. */ + int nequiv_classes; + + /* # of range expressions. */ + int nranges; + + /* # of character classes. */ + int nchar_classes; +} re_charset_t; +#endif /* RE_ENABLE_I18N */ + +typedef struct +{ + union + { + unsigned char c; /* for CHARACTER */ + re_bitset_ptr_t sbcset; /* for SIMPLE_BRACKET */ +#ifdef RE_ENABLE_I18N + re_charset_t *mbcset; /* for COMPLEX_BRACKET */ +#endif /* RE_ENABLE_I18N */ + int idx; /* for BACK_REF */ + re_context_type ctx_type; /* for ANCHOR */ + } opr; +#if __GNUC__ >= 2 + re_token_type_t type : 8; +#else + re_token_type_t type; +#endif + unsigned int constraint : 10; /* context constraint */ + unsigned int duplicated : 1; + unsigned int opt_subexp : 1; +#ifdef RE_ENABLE_I18N + unsigned int accept_mb : 1; + /* These 2 bits can be moved into the union if needed (e.g. if running out + of bits; move opr.c to opr.c.c and move the flags to opr.c.flags). */ + unsigned int mb_partial : 1; +#endif + unsigned int word_char : 1; +} re_token_t; + +#define IS_EPSILON_NODE(type) ((type) & EPSILON_BIT) + +struct re_string_t +{ + /* Indicate the raw buffer which is the original string passed as an + argument of regexec(), re_search(), etc.. */ + const unsigned char *raw_mbs; + /* Store the multibyte string. In case of "case insensitive mode" like + REG_ICASE, upper cases of the string are stored, otherwise MBS points + the same address that RAW_MBS points. */ + unsigned char *mbs; +#ifdef RE_ENABLE_I18N + /* Store the wide character string which is corresponding to MBS. */ + wint_t *wcs; + int *offsets; + mbstate_t cur_state; +#endif + /* Index in RAW_MBS. Each character mbs[i] corresponds to + raw_mbs[raw_mbs_idx + i]. */ + int raw_mbs_idx; + /* The length of the valid characters in the buffers. */ + int valid_len; + /* The corresponding number of bytes in raw_mbs array. */ + int valid_raw_len; + /* The length of the buffers MBS and WCS. */ + int bufs_len; + /* The index in MBS, which is updated by re_string_fetch_byte. */ + int cur_idx; + /* length of RAW_MBS array. */ + int raw_len; + /* This is RAW_LEN - RAW_MBS_IDX + VALID_LEN - VALID_RAW_LEN. */ + int len; + /* End of the buffer may be shorter than its length in the cases such + as re_match_2, re_search_2. Then, we use STOP for end of the buffer + instead of LEN. */ + int raw_stop; + /* This is RAW_STOP - RAW_MBS_IDX adjusted through OFFSETS. */ + int stop; + + /* The context of mbs[0]. We store the context independently, since + the context of mbs[0] may be different from raw_mbs[0], which is + the beginning of the input string. */ + unsigned int tip_context; + /* The translation passed as a part of an argument of re_compile_pattern. */ + RE_TRANSLATE_TYPE trans; + /* Copy of re_dfa_t's word_char. */ + re_const_bitset_ptr_t word_char; + /* 1 if REG_ICASE. */ + unsigned char icase; + unsigned char is_utf8; + unsigned char map_notascii; + unsigned char mbs_allocated; + unsigned char offsets_needed; + unsigned char newline_anchor; + unsigned char word_ops_used; + int mb_cur_max; +}; +typedef struct re_string_t re_string_t; + + +struct re_dfa_t; +typedef struct re_dfa_t re_dfa_t; + +#ifndef _LIBC +# ifdef __i386__ +# define internal_function __attribute ((regparm (3), stdcall)) +# else +# define internal_function +# endif +#endif + +#ifndef NOT_IN_libc +static reg_errcode_t re_string_realloc_buffers (re_string_t *pstr, + int new_buf_len) + internal_function; +# ifdef RE_ENABLE_I18N +static void build_wcs_buffer (re_string_t *pstr) internal_function; +static reg_errcode_t build_wcs_upper_buffer (re_string_t *pstr) + internal_function; +# endif /* RE_ENABLE_I18N */ +static void build_upper_buffer (re_string_t *pstr) internal_function; +static void re_string_translate_buffer (re_string_t *pstr) internal_function; +static unsigned int re_string_context_at (const re_string_t *input, int idx, + int eflags) + internal_function __attribute ((pure)); +#endif +#define re_string_peek_byte(pstr, offset) \ + ((pstr)->mbs[(pstr)->cur_idx + offset]) +#define re_string_fetch_byte(pstr) \ + ((pstr)->mbs[(pstr)->cur_idx++]) +#define re_string_first_byte(pstr, idx) \ + ((idx) == (pstr)->valid_len || (pstr)->wcs[idx] != WEOF) +#define re_string_is_single_byte_char(pstr, idx) \ + ((pstr)->wcs[idx] != WEOF && ((pstr)->valid_len == (idx) + 1 \ + || (pstr)->wcs[(idx) + 1] != WEOF)) +#define re_string_eoi(pstr) ((pstr)->stop <= (pstr)->cur_idx) +#define re_string_cur_idx(pstr) ((pstr)->cur_idx) +#define re_string_get_buffer(pstr) ((pstr)->mbs) +#define re_string_length(pstr) ((pstr)->len) +#define re_string_byte_at(pstr,idx) ((pstr)->mbs[idx]) +#define re_string_skip_bytes(pstr,idx) ((pstr)->cur_idx += (idx)) +#define re_string_set_index(pstr,idx) ((pstr)->cur_idx = (idx)) + +#ifdef WIN32 +# include <malloc.h> +#else +# include <alloca.h> +#endif + +#ifndef _LIBC +# if HAVE_ALLOCA +/* The OS usually guarantees only one guard page at the bottom of the stack, + and a page size can be as small as 4096 bytes. So we cannot safely + allocate anything larger than 4096 bytes. Also care for the possibility + of a few compiler-allocated temporary stack slots. */ +# define __libc_use_alloca(n) ((n) < 4032) +# else +/* alloca is implemented with malloc, so just use malloc. */ +# define __libc_use_alloca(n) 0 +# endif +#endif + +#define re_malloc(t,n) ((t *) malloc ((n) * sizeof (t))) +#define re_realloc(p,t,n) ((t *) realloc (p, (n) * sizeof (t))) +#define re_free(p) free (p) + +struct bin_tree_t +{ + struct bin_tree_t *parent; + struct bin_tree_t *left; + struct bin_tree_t *right; + struct bin_tree_t *first; + struct bin_tree_t *next; + + re_token_t token; + + /* `node_idx' is the index in dfa->nodes, if `type' == 0. + Otherwise `type' indicate the type of this node. */ + int node_idx; +}; +typedef struct bin_tree_t bin_tree_t; + +#define BIN_TREE_STORAGE_SIZE \ + ((1024 - sizeof (void *)) / sizeof (bin_tree_t)) + +struct bin_tree_storage_t +{ + struct bin_tree_storage_t *next; + bin_tree_t data[BIN_TREE_STORAGE_SIZE]; +}; +typedef struct bin_tree_storage_t bin_tree_storage_t; + +#define CONTEXT_WORD 1 +#define CONTEXT_NEWLINE (CONTEXT_WORD << 1) +#define CONTEXT_BEGBUF (CONTEXT_NEWLINE << 1) +#define CONTEXT_ENDBUF (CONTEXT_BEGBUF << 1) + +#define IS_WORD_CONTEXT(c) ((c) & CONTEXT_WORD) +#define IS_NEWLINE_CONTEXT(c) ((c) & CONTEXT_NEWLINE) +#define IS_BEGBUF_CONTEXT(c) ((c) & CONTEXT_BEGBUF) +#define IS_ENDBUF_CONTEXT(c) ((c) & CONTEXT_ENDBUF) +#define IS_ORDINARY_CONTEXT(c) ((c) == 0) + +#define IS_WORD_CHAR(ch) (isalnum (ch) || (ch) == '_') +#define IS_NEWLINE(ch) ((ch) == NEWLINE_CHAR) +#define IS_WIDE_WORD_CHAR(ch) (iswalnum (ch) || (ch) == L'_') +#define IS_WIDE_NEWLINE(ch) ((ch) == WIDE_NEWLINE_CHAR) + +#define NOT_SATISFY_PREV_CONSTRAINT(constraint,context) \ + ((((constraint) & PREV_WORD_CONSTRAINT) && !IS_WORD_CONTEXT (context)) \ + || ((constraint & PREV_NOTWORD_CONSTRAINT) && IS_WORD_CONTEXT (context)) \ + || ((constraint & PREV_NEWLINE_CONSTRAINT) && !IS_NEWLINE_CONTEXT (context))\ + || ((constraint & PREV_BEGBUF_CONSTRAINT) && !IS_BEGBUF_CONTEXT (context))) + +#define NOT_SATISFY_NEXT_CONSTRAINT(constraint,context) \ + ((((constraint) & NEXT_WORD_CONSTRAINT) && !IS_WORD_CONTEXT (context)) \ + || (((constraint) & NEXT_NOTWORD_CONSTRAINT) && IS_WORD_CONTEXT (context)) \ + || (((constraint) & NEXT_NEWLINE_CONSTRAINT) && !IS_NEWLINE_CONTEXT (context)) \ + || (((constraint) & NEXT_ENDBUF_CONSTRAINT) && !IS_ENDBUF_CONTEXT (context))) + +struct re_dfastate_t +{ + unsigned int hash; + re_node_set nodes; + re_node_set non_eps_nodes; + re_node_set inveclosure; + re_node_set *entrance_nodes; + struct re_dfastate_t **trtable, **word_trtable; + unsigned int context : 4; + unsigned int halt : 1; + /* If this state can accept `multi byte'. + Note that we refer to multibyte characters, and multi character + collating elements as `multi byte'. */ + unsigned int accept_mb : 1; + /* If this state has backreference node(s). */ + unsigned int has_backref : 1; + unsigned int has_constraint : 1; +}; +typedef struct re_dfastate_t re_dfastate_t; + +struct re_state_table_entry +{ + int num; + int alloc; + re_dfastate_t **array; +}; + +/* Array type used in re_sub_match_last_t and re_sub_match_top_t. */ + +typedef struct +{ + int next_idx; + int alloc; + re_dfastate_t **array; +} state_array_t; + +/* Store information about the node NODE whose type is OP_CLOSE_SUBEXP. */ + +typedef struct +{ + int node; + int str_idx; /* The position NODE match at. */ + state_array_t path; +} re_sub_match_last_t; + +/* Store information about the node NODE whose type is OP_OPEN_SUBEXP. + And information about the node, whose type is OP_CLOSE_SUBEXP, + corresponding to NODE is stored in LASTS. */ + +typedef struct +{ + int str_idx; + int node; + state_array_t *path; + int alasts; /* Allocation size of LASTS. */ + int nlasts; /* The number of LASTS. */ + re_sub_match_last_t **lasts; +} re_sub_match_top_t; + +struct re_backref_cache_entry +{ + int node; + int str_idx; + int subexp_from; + int subexp_to; + char more; + char unused; + unsigned short int eps_reachable_subexps_map; +}; + +typedef struct +{ + /* The string object corresponding to the input string. */ + re_string_t input; +#if defined _LIBC || (defined __STDC_VERSION__ && __STDC_VERSION__ >= 199901L) + const re_dfa_t *const dfa; +#else + const re_dfa_t *dfa; +#endif + /* EFLAGS of the argument of regexec. */ + int eflags; + /* Where the matching ends. */ + int match_last; + int last_node; + /* The state log used by the matcher. */ + re_dfastate_t **state_log; + int state_log_top; + /* Back reference cache. */ + int nbkref_ents; + int abkref_ents; + struct re_backref_cache_entry *bkref_ents; + int max_mb_elem_len; + int nsub_tops; + int asub_tops; + re_sub_match_top_t **sub_tops; +} re_match_context_t; + +typedef struct +{ + re_dfastate_t **sifted_states; + re_dfastate_t **limited_states; + int last_node; + int last_str_idx; + re_node_set limits; +} re_sift_context_t; + +struct re_fail_stack_ent_t +{ + int idx; + int node; + regmatch_t *regs; + re_node_set eps_via_nodes; +}; + +struct re_fail_stack_t +{ + int num; + int alloc; + struct re_fail_stack_ent_t *stack; +}; + +struct re_dfa_t +{ + re_token_t *nodes; + size_t nodes_alloc; + size_t nodes_len; + int *nexts; + int *org_indices; + re_node_set *edests; + re_node_set *eclosures; + re_node_set *inveclosures; + struct re_state_table_entry *state_table; + re_dfastate_t *init_state; + re_dfastate_t *init_state_word; + re_dfastate_t *init_state_nl; + re_dfastate_t *init_state_begbuf; + bin_tree_t *str_tree; + bin_tree_storage_t *str_tree_storage; + re_bitset_ptr_t sb_char; + int str_tree_storage_idx; + + /* number of subexpressions `re_nsub' is in regex_t. */ + unsigned int state_hash_mask; + int init_node; + int nbackref; /* The number of backreference in this dfa. */ + + /* Bitmap expressing which backreference is used. */ + bitset_word_t used_bkref_map; + bitset_word_t completed_bkref_map; + + unsigned int has_plural_match : 1; + /* If this dfa has "multibyte node", which is a backreference or + a node which can accept multibyte character or multi character + collating element. */ + unsigned int has_mb_node : 1; + unsigned int is_utf8 : 1; + unsigned int map_notascii : 1; + unsigned int word_ops_used : 1; + int mb_cur_max; + bitset_t word_char; + reg_syntax_t syntax; + int *subexp_map; +#ifdef DEBUG + char* re_str; +#endif + __libc_lock_define (, lock) +}; + +#define re_node_set_init_empty(set) memset (set, '\0', sizeof (re_node_set)) +#define re_node_set_remove(set,id) \ + (re_node_set_remove_at (set, re_node_set_contains (set, id) - 1)) +#define re_node_set_empty(p) ((p)->nelem = 0) +#define re_node_set_free(set) re_free ((set)->elems) + + +typedef enum +{ + SB_CHAR, + MB_CHAR, + EQUIV_CLASS, + COLL_SYM, + CHAR_CLASS +} bracket_elem_type; + +typedef struct +{ + bracket_elem_type type; + union + { + unsigned char ch; + unsigned char *name; + wchar_t wch; + } opr; +} bracket_elem_t; + + +/* Inline functions for bitset operation. */ +static inline void +bitset_not (bitset_t set) +{ + int bitset_i; + for (bitset_i = 0; bitset_i < BITSET_WORDS; ++bitset_i) + set[bitset_i] = ~set[bitset_i]; +} + +static inline void +bitset_merge (bitset_t dest, const bitset_t src) +{ + int bitset_i; + for (bitset_i = 0; bitset_i < BITSET_WORDS; ++bitset_i) + dest[bitset_i] |= src[bitset_i]; +} + +static inline void +bitset_mask (bitset_t dest, const bitset_t src) +{ + int bitset_i; + for (bitset_i = 0; bitset_i < BITSET_WORDS; ++bitset_i) + dest[bitset_i] &= src[bitset_i]; +} + +#ifdef RE_ENABLE_I18N +/* Inline functions for re_string. */ +static inline int +internal_function __attribute ((pure)) +re_string_char_size_at (const re_string_t *pstr, int idx) +{ + int byte_idx; + if (pstr->mb_cur_max == 1) + return 1; + for (byte_idx = 1; idx + byte_idx < pstr->valid_len; ++byte_idx) + if (pstr->wcs[idx + byte_idx] != WEOF) + break; + return byte_idx; +} + +static inline wint_t +internal_function __attribute ((pure)) +re_string_wchar_at (const re_string_t *pstr, int idx) +{ + if (pstr->mb_cur_max == 1) + return (wint_t) pstr->mbs[idx]; + return (wint_t) pstr->wcs[idx]; +} + +# ifndef NOT_IN_libc +static int +internal_function __attribute ((pure)) +re_string_elem_size_at (const re_string_t *pstr, int idx) +{ +# ifdef _LIBC + const unsigned char *p, *extra; + const int32_t *table, *indirect; + int32_t tmp; +# include <locale/weight.h> + uint_fast32_t nrules = _NL_CURRENT_WORD (LC_COLLATE, _NL_COLLATE_NRULES); + + if (nrules != 0) + { + table = (const int32_t *) _NL_CURRENT (LC_COLLATE, _NL_COLLATE_TABLEMB); + extra = (const unsigned char *) + _NL_CURRENT (LC_COLLATE, _NL_COLLATE_EXTRAMB); + indirect = (const int32_t *) _NL_CURRENT (LC_COLLATE, + _NL_COLLATE_INDIRECTMB); + p = pstr->mbs + idx; + tmp = findidx (&p); + return p - pstr->mbs - idx; + } + else +# endif /* _LIBC */ + return 1; +} +# endif +#endif /* RE_ENABLE_I18N */ + +#endif /* _REGEX_INTERNAL_H */ Modified: ctags/gnu_regex/regexec.c 4344 lines changed, 4344 insertions(+), 0 deletions(-) =================================================================== No diff available, check online -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

[geany/geany] 92382d: Update to latest ctags main
by Jiří Techet 07 Feb '21

07 Feb '21

Branch: refs/heads/master Author: Jiří Techet <techet(a)gmail.com> Committer: Jiří Techet <techet(a)gmail.com> Date: Wed, 18 Nov 2020 21:22:41 UTC Commit: 92382dcdbcccb7f531569dc291c33e275a2a9e44 https://github.com/geany/geany/commit/92382dcdbcccb7f531569dc291c33e275a2a9… Log Message: ----------- Update to latest ctags main See commit 2aa034a30dc54f5db18db2c9140821f64426283e upstream. Modified Paths: -------------- configure.ac ctags/Makefile.am ctags/main/args.c ctags/main/args_p.h ctags/main/cmd.c ctags/main/colprint.c ctags/main/colprint_p.h ctags/main/ctags-api.c ctags/main/ctags-api.h ctags/main/ctags.h ctags/main/debug.c ctags/main/debug.h ctags/main/dependency.c ctags/main/dependency.h ctags/main/dependency_p.h ctags/main/e_msoft.h ctags/main/entry.c ctags/main/entry.h ctags/main/entry_p.h ctags/main/entry_private.c ctags/main/error.c ctags/main/error_p.h ctags/main/field.c ctags/main/field.h ctags/main/field_p.h ctags/main/flags.c ctags/main/flags_p.h ctags/main/fmt.c ctags/main/fmt_p.h ctags/main/gcc-attr.h ctags/main/general.h ctags/main/gvars.h ctags/main/htable.c ctags/main/htable.h ctags/main/interactive_p.h ctags/main/keyword.c ctags/main/keyword.h ctags/main/keyword_p.h ctags/main/kind.c ctags/main/kind.h ctags/main/kind_p.h ctags/main/lcpp.c ctags/main/lcpp.h ctags/main/lregex.c ctags/main/lregex.h ctags/main/lregex_p.h ctags/main/lxcmd.c ctags/main/lxpath.c ctags/main/lxpath.h ctags/main/lxpath_p.h ctags/main/main.c ctags/main/main.h ctags/main/main_p.h ctags/main/mbcs.c ctags/main/mbcs.h ctags/main/mbcs_p.h ctags/main/mini-geany.c ctags/main/mio.c ctags/main/mio.h ctags/main/nestlevel.c ctags/main/nestlevel.h ctags/main/numarray.c ctags/main/numarray.h ctags/main/objpool.c ctags/main/options.c ctags/main/options.h ctags/main/options_p.h ctags/main/output-ctags.c ctags/main/output.h ctags/main/param.c ctags/main/param.h ctags/main/param_p.h ctags/main/parse.c ctags/main/parse.h ctags/main/parse_p.h ctags/main/parsers.h ctags/main/parsers_p.h ctags/main/pcoproc.c ctags/main/pcoproc.h ctags/main/portable-dirent_p.h ctags/main/portable-scandir.c ctags/main/promise.c ctags/main/promise.h ctags/main/promise_p.h ctags/main/ptag.c ctags/main/ptag_p.h ctags/main/ptrarray.c ctags/main/ptrarray.h ctags/main/rbtree.c ctags/main/rbtree.h ctags/main/read.c ctags/main/read.h ctags/main/read_p.h ctags/main/repoinfo.h ctags/main/routines.c ctags/main/routines.h ctags/main/routines_p.h ctags/main/seccomp.c ctags/main/selectors.c ctags/main/selectors.h ctags/main/sort.c ctags/main/sort_p.h ctags/main/stats.c ctags/main/stats_p.h ctags/main/strlist.c ctags/main/strlist.h ctags/main/subparser.h ctags/main/subparser_p.h ctags/main/tokeninfo.c ctags/main/tokeninfo.h ctags/main/trace.c ctags/main/trace.h ctags/main/trashbox.h ctags/main/trashbox_p.h ctags/main/types.h ctags/main/unwindi.c ctags/main/unwindi.h ctags/main/vstring.c ctags/main/vstring.h ctags/main/writer-ctags.c ctags/main/writer-etags.c ctags/main/writer-json.c ctags/main/writer-xref.c ctags/main/writer.c ctags/main/writer_p.h ctags/main/xtag.c ctags/main/xtag.h ctags/main/xtag_p.h Modified: configure.ac 3 lines changed, 1 insertions(+), 2 deletions(-) =================================================================== @@ -43,8 +43,7 @@ AC_CHECK_HEADERS([fcntl.h glob.h stdlib.h sys/time.h errno.h limits.h]) # Checks for dependencies needed by ctags AC_CHECK_HEADERS([fnmatch.h direct.h io.h sys/dir.h]) -AC_DEFINE([USE_STDBOOL_H], [1], [whether or not to use <stdbool.h>.]) -AC_DEFINE([CTAGS_LIB], [1], [compile ctags as a library.]) +AC_DEFINE([HAVE_STDBOOL_H], [1], [whether or not to use <stdbool.h>.]) # Checks for typedefs, structures, and compiler characteristics. AC_TYPE_OFF_T Modified: ctags/Makefile.am 81 lines changed, 61 insertions(+), 20 deletions(-) =================================================================== @@ -1,6 +1,7 @@ AM_CPPFLAGS = \ -I$(srcdir)/main \ -I$(srcdir)/parsers \ + -DEXTERNAL_PARSER_LIST_FILE=\"$(top_srcdir)/src/tagmanager/tm_parsers.h\" \ -DG_LOG_DOMAIN=\"CTags\" AM_CFLAGS = \ $(GTK_CFLAGS) \ @@ -30,6 +31,8 @@ parsers = \ parsers/html.c \ parsers/jscript.c \ parsers/json.c \ + parsers/lcpp.c \ + parsers/lcpp.h \ parsers/lua.c \ parsers/make.c \ parsers/markdown.c \ @@ -53,44 +56,56 @@ parsers = \ parsers/verilog.c \ parsers/vhdl.c +# skip cmd.c and mini-geany.c which define main() libctags_la_SOURCES = \ main/args.c \ - main/args.h \ + main/args_p.h \ + main/colprint.c \ + main/colprint_p.h \ main/ctags.h \ - main/ctags-api.c \ - main/ctags-api.h \ - main/debug.h \ main/debug.c \ - main/dependency.h \ + main/debug.h \ main/dependency.c \ + main/dependency.h \ + main/dependency_p.h \ main/e_msoft.h \ main/entry.c \ main/entry.h \ + main/entry_p.h \ + main/entry_private.c \ main/error.c \ - main/error.h \ + main/error_p.h \ main/field.c \ main/field.h \ + main/field_p.h \ main/flags.c \ - main/flags.h \ + main/flags_p.h \ main/fmt.c \ - main/fmt.h \ + main/fmt_p.h \ main/gcc-attr.h \ main/general.h \ + main/gvars.h \ main/htable.c \ main/htable.h \ main/inline.h \ + main/interactive_p.h \ main/keyword.c \ main/keyword.h \ + main/keyword_p.h \ main/kind.c \ main/kind.h \ - main/lcpp.c \ - main/lcpp.h \ + main/kind_p.h \ main/lregex.c \ - main/lxcmd.c \ + main/lregex.h \ + main/lregex_p.h \ main/lxpath.c \ + main/lxpath.h \ + main/lxpath_p.h \ main/main.c \ - main/main.h \ + main/main_p.h \ + main/mbcs.c \ main/mbcs.h \ + main/mbcs_p.h \ main/mio.c \ main/mio.h \ main/nestlevel.c \ @@ -101,37 +116,63 @@ libctags_la_SOURCES = \ main/objpool.h \ main/options.c \ main/options.h \ - main/output-ctags.c \ - main/output.h \ + main/options_p.h \ + main/param.c \ + main/param.h \ + main/param_p.h \ main/parse.c \ main/parse.h \ - main/parsers.h \ - main/pcoproc.c \ - main/pcoproc.h \ + main/parse_p.h \ + main/parsers_p.h \ + main/portable-dirent_p.h \ + main/portable-scandir.c \ main/promise.c \ main/promise.h \ + main/promise_p.h \ main/ptag.c \ - main/ptag.h \ + main/ptag_p.h \ main/ptrarray.c \ main/ptrarray.h \ + main/rbtree.c \ + main/rbtree.h \ main/read.c \ main/read.h \ + main/read_p.h \ main/repoinfo.c \ main/repoinfo.h \ main/routines.c \ main/routines.h \ + main/routines_p.h \ + main/seccomp.c \ main/selectors.c \ main/selectors.h \ main/sort.c \ - main/sort.h \ + main/sort_p.h \ + main/stats.c \ + main/stats_p.h \ main/strlist.c \ main/strlist.h \ + main/subparser.h \ + main/subparser_p.h \ + main/tokeninfo.c \ + main/tokeninfo.h \ + main/trace.c \ main/trace.h \ main/trashbox.c \ main/trashbox.h \ + main/trashbox_p.h \ main/types.h \ + main/unwindi.c \ + main/unwindi.h \ main/vstring.c \ main/vstring.h \ - main/xtag.h \ + main/writer-ctags.c \ + main/writer-etags.c \ + main/writer-json.c \ + main/writer-xref.c \ + main/writer.c \ + main/writer_p.h \ main/xtag.c \ + main/xtag.h \ + main/xtag_p.h \ $(parsers) Modified: ctags/main/args.c 6 lines changed, 4 insertions(+), 2 deletions(-) =================================================================== @@ -16,9 +16,10 @@ #include <string.h> #include <ctype.h> -#include "args.h" +#include "args_p.h" #include "debug.h" #include "routines.h" +#include "vstring.h" /* * FUNCTION DEFINITIONS @@ -281,7 +282,8 @@ extern void argForth (Arguments* const current) extern void argDelete (Arguments* const current) { Assert (current != NULL); - if (current->type == ARG_STRING && current->item != NULL) + if ((current->type == ARG_STRING + || current->type == ARG_FILE) && current->item != NULL) eFree (current->item); memset (current, 0, sizeof (Arguments)); eFree (current); Modified: ctags/main/args_p.h 6 lines changed, 3 insertions(+), 3 deletions(-) =================================================================== @@ -6,8 +6,8 @@ * * Defines external interface to command line argument reading. */ -#ifndef CTAGS_MAIN_ARGS_H -#define CTAGS_MAIN_ARGS_H +#ifndef CTAGS_MAIN_ARGS_PRIVATE_H +#define CTAGS_MAIN_ARGS_PRIVATE_H /* * INCLUDE FILES @@ -54,4 +54,4 @@ extern void argSetLineMode (Arguments* const current); extern void argForth (Arguments* const current); extern void argDelete (Arguments* const current); -#endif /* CTAGS_MAIN_ARGS_H */ +#endif /* CTAGS_MAIN_ARGS_PRIVATE_H */ Modified: ctags/main/cmd.c 22 lines changed, 22 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,22 @@ +/* +* Copyright (c) 1998-2002, Darren Hiebert +* +* This source code is released for free distribution under the terms of the +* GNU General Public License version 2 or (at your option) any later version. +*/ + +/* +* INCLUDE FILES +*/ +#include "general.h" /* must always come first */ + +#include "main_p.h" + + +/* +* FUNCTION DEFINITIONS +*/ +int main(int argc, char **argv) +{ + return ctags_cli_main (argc, argv); +} Modified: ctags/main/colprint.c 295 lines changed, 295 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,295 @@ +/* +* Copyright (c) 2017 Masatake YAMATO +* +* This source code is released for free distribution under the terms of the +* GNU General Public License version 2 or (at your option) any later version. +* +*/ +#include "general.h" /* must always come first */ + +#include "colprint_p.h" +#include "ptrarray.h" +#include "routines.h" +#include "strlist.h" +#include "vstring.h" + +#include <stdarg.h> +#include <stdio.h> +#include <string.h> + + +enum colprintJustification { + COLPRINT_LEFT, /* L:... */ + COLPRINT_RIGHT, /* R:... */ + COLPRINT_LAST, +}; + +struct colprintHeaderColumn { + vString *value; + enum colprintJustification justification; + unsigned int maxWidth; + bool needPrefix; +}; + +struct colprintTable { + ptrArray *header; + ptrArray *lines; +}; + +static void fillWithWhitespaces (int i, FILE *fp) +{ + while (i-- > 0) + { + fputc(' ', fp); + } +} + +static struct colprintHeaderColumn * colprintHeaderColumnNew (const char* spec) +{ + int offset = 2; + struct colprintHeaderColumn *headerCol = xCalloc (1, struct colprintHeaderColumn); + + if (strstr(spec, "L:") == spec) + headerCol->justification = COLPRINT_LEFT; + else if (strstr(spec, "R:") == spec) + headerCol->justification = COLPRINT_RIGHT; + else + { + headerCol->justification = COLPRINT_LEFT; + offset = 0; + } + + headerCol->value = vStringNewInit(spec + offset); + headerCol->maxWidth = vStringLength(headerCol->value); + return headerCol; +} + +static void colprintHeaderColumnDelete (struct colprintHeaderColumn * headerCol) +{ + vStringDelete (headerCol->value); + eFree (headerCol); +} + +struct colprintTable *colprintTableNew (const char* columnHeader, ... /* NULL TERMINATED */) +{ + char *tmp; + va_list ap; + struct colprintTable *table; + struct colprintHeaderColumn *headerCol; + + + table = xCalloc (1, struct colprintTable); + table->header = ptrArrayNew ((ptrArrayDeleteFunc)colprintHeaderColumnDelete); + table->lines = ptrArrayNew ((ptrArrayDeleteFunc)stringListDelete); + + headerCol = colprintHeaderColumnNew(columnHeader); + ptrArrayAdd (table->header, headerCol); + + va_start(ap, columnHeader); + while (1) + { + tmp = va_arg(ap, char*); + if (tmp) + { + headerCol = colprintHeaderColumnNew(tmp); + ptrArrayAdd (table->header, headerCol); + } + else + break; + } + va_end(ap); + + struct colprintHeaderColumn *last_col = ptrArrayLast (table->header); + if (last_col) + last_col->justification = COLPRINT_LAST; + + return table; +} + +void colprintTableDelete (struct colprintTable *table) +{ + ptrArrayDelete(table->header); + table->header = NULL; + + ptrArrayDelete(table->lines); + table->header = NULL; + + eFree (table); +} + +static void colprintColumnPrintGeneric (vString *column, struct colprintHeaderColumn *spec, bool machinable, FILE *fp) +{ + int maxWidth = spec->maxWidth + (spec->needPrefix? 1: 0); + + if ((column == spec->value) && (spec->needPrefix)) + { + fputc('#', fp); + maxWidth--; + } + + if (machinable) + { + fputs (vStringValue (column), fp); + if (spec->justification != COLPRINT_LAST) + fputc ('\t', fp); + } + else + { + int padLen = maxWidth - vStringLength (column); + if (spec->justification == COLPRINT_LEFT + || spec->justification == COLPRINT_LAST) + { + fputs (vStringValue (column), fp); + if (spec->justification != COLPRINT_LAST) + { + fillWithWhitespaces (padLen, fp); + fputc (' ', fp); + } + } + else + { + fillWithWhitespaces (padLen, fp); + fputs (vStringValue (column), fp); + fputc (' ', fp); + } + } +} + +static void colprintHeaderColumnPrint (struct colprintHeaderColumn *headerCol, bool machinable, FILE* fp) +{ + colprintColumnPrintGeneric (headerCol->value, headerCol, machinable, fp); +} + +static void colprintHeaderPrint (ptrArray *header, unsigned int startFrom, bool withHeader, bool machinable, FILE *fp) +{ + unsigned int i; + + if (!withHeader) + return; + + for (i = startFrom; i < ptrArrayCount(header); i++) + { + struct colprintHeaderColumn *headerCol = ptrArrayItem (header, i); + colprintHeaderColumnPrint (headerCol, machinable, fp); + } + fputc('\n', fp); +} + +static void colprintLinePrint (stringList *line, unsigned int startFrom, ptrArray *header, bool machinable, FILE *fp) +{ + unsigned int i; + + for (i = startFrom; i < stringListCount (line); i++) + { + vString *value = stringListItem(line, i); + struct colprintHeaderColumn *spec = ptrArrayItem (header, i); + colprintColumnPrintGeneric(value, spec, machinable, fp); + } +} +static void colprintLinesPrint (ptrArray *lines, unsigned int startFrom, ptrArray *header, bool machinable, FILE *fp) +{ + unsigned int i; + + for (i = 0; i < ptrArrayCount (lines); i++) + { + stringList *line = ptrArrayItem (lines, i); + colprintLinePrint (line, startFrom, header, machinable, fp); + fputc('\n', fp); + } +} + +static void colprintUpdateMaxWidths (ptrArray *header, ptrArray *lines, unsigned int startFrom) +{ + for (unsigned int c = 0; c < ptrArrayCount(header); c++) + { + struct colprintHeaderColumn *spec = ptrArrayItem (header, c); + + if (c == startFrom) + spec->needPrefix = true; + else + spec->needPrefix = false; + } + + for (unsigned int c = 0; c < ptrArrayCount(header); c++) + { + struct colprintHeaderColumn *spec = ptrArrayItem (header, c); + + for (unsigned int l = 0; l < ptrArrayCount(lines); l++) + { + struct colprintLine *line = ptrArrayItem(lines, l); + vString *column = ptrArrayItem((ptrArray *)line, c); + if (spec->maxWidth < vStringLength(column)) + spec->maxWidth = vStringLength(column); + } + } +} + +void colprintTablePrint (struct colprintTable *table, unsigned int startFrom, bool withHeader, bool machinable, FILE *fp) +{ + colprintUpdateMaxWidths (table->header, table->lines, startFrom); + + colprintHeaderPrint (table->header, startFrom, withHeader, machinable, fp); + colprintLinesPrint (table->lines, startFrom, table->header, machinable, fp); +} + +void colprintTableSort (struct colprintTable *table, int (* compareFn) (struct colprintLine *, struct colprintLine *)) +{ + ptrArraySort (table->lines, (int (*) (const void *, const void *))compareFn); +} + +struct colprintLine *colprintTableGetNewLine (struct colprintTable *table) +{ + stringList *line = stringListNew (); + + ptrArrayAdd (table->lines, line); + return (struct colprintLine *)line; +} + +static void colprintLineAppendColumn (struct colprintLine *line, vString *column) +{ + stringList *slist = (stringList *)line; + stringListAdd (slist, column); +} + +void colprintLineAppendColumnCString (struct colprintLine *line, const char *column) +{ + vString* vcol = vStringNewInit (column? column: ""); + colprintLineAppendColumn (line, vcol); +} + +void colprintLineAppendColumnVString (struct colprintLine *line, vString* column) +{ + colprintLineAppendColumnCString(line, vStringValue (column)); +} + +void colprintLineAppendColumnChar (struct colprintLine *line, char column) +{ + vString* vcol = vStringNew (); + vStringPut (vcol, column); + colprintLineAppendColumn (line, vcol); +} + +void colprintLineAppendColumnInt (struct colprintLine *line, unsigned int column) +{ + char buf[12]; + + snprintf(buf, 12, "%u", column); + colprintLineAppendColumnCString (line, buf); +} + +void colprintLineAppendColumnBool (struct colprintLine *line, bool column) +{ + colprintLineAppendColumnCString (line, column? "yes": "no"); +} + +const char *colprintLineGetColumn (struct colprintLine *line, unsigned int column) +{ + stringList *slist = (stringList *)line; + if (column <= stringListCount(slist)) + { + vString *vstr = stringListItem (slist, column); + return vStringValue (vstr); + } + else + return NULL; +} Modified: ctags/main/colprint_p.h 37 lines changed, 37 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,37 @@ +/* +* Copyright (c) 2017 Masatake YAMATO +* +* This source code is released for free distribution under the terms of the +* GNU General Public License version 2 or (at your option) any later version. +* +*/ +#ifndef CTAGS_MAIN_COLPRINT_PRIVATE_H +#define CTAGS_MAIN_COLPRINT_PRIVATE_H + +#include "general.h" + +#include "vstring.h" +#include <stdio.h> + +struct colprintTable; +struct colprintLine; + +/* Each column must have a prefix for specifying justification: "L:" or "R:". */ +struct colprintTable *colprintTableNew (const char* columnHeader, ... /* NULL TERMINATED */); +void colprintTableDelete (struct colprintTable *table); +void colprintTablePrint (struct colprintTable *table, unsigned int startFrom, bool withHeader, bool machinable, FILE *fp); +void colprintTableSort (struct colprintTable *table, int (* compareFn) (struct colprintLine *, struct colprintLine *)); + +struct colprintLine *colprintTableGetNewLine (struct colprintTable *table); + +void colprintLineAppendColumnCString (struct colprintLine *line, const char* column); +void colprintLineAppendColumnVString (struct colprintLine *line, vString* column); +void colprintLineAppendColumnChar (struct colprintLine *line, char column); +void colprintLineAppendColumnInt (struct colprintLine *line, unsigned int column); + +/* Appends "yes" or "no". */ +void colprintLineAppendColumnBool (struct colprintLine *line, bool column); + +const char *colprintLineGetColumn (struct colprintLine *line, unsigned int column); + +#endif /* CTAGS_MAIN_COLPRINT_PRIVATE_H */ Modified: ctags/main/ctags-api.c 144 lines changed, 0 insertions(+), 144 deletions(-) =================================================================== @@ -1,144 +0,0 @@ -/* -* Copyright (c) 2016, Jiri Techet -* -* This source code is released for free distribution under the terms of the -* GNU General Public License version 2 or (at your option) any later version. -* -* Defines ctags API when compiled as a library. -*/ - -#include "general.h" /* must always come first */ - -#ifdef CTAGS_LIB - -#include "ctags-api.h" -#include "types.h" -#include "routines.h" -#include "error.h" -#include "output.h" -#include "parse.h" -#include "options.h" -#include "trashbox.h" - -#include <stdio.h> -#include <string.h> -#include <errno.h> - -static bool nofatalErrorPrinter (const errorSelection selection, - const char *const format, - va_list ap, void *data CTAGS_ATTR_UNUSED) -{ - fprintf (stderr, "%s: ", (selection & WARNING) ? "Warning: " : "Error"); - vfprintf (stderr, format, ap); - if (selection & PERROR) -#ifdef HAVE_STRERROR - fprintf (stderr, " : %s", strerror (errno)); -#else - perror (" "); -#endif - fputs ("\n", stderr); - - return false; -} - -extern void ctagsInit(void) -{ - setErrorPrinter (nofatalErrorPrinter, NULL); - setTagWriter (&ctagsWriter); - - checkRegex (); - initFieldDescs (); - - initializeParsing (); - initOptions (); - - initDefaultTrashBox (); - - /* make sure all parsers are initialized */ - initializeParser (LANG_AUTO); -} - - - -extern void ctagsParse(unsigned char *buffer, size_t bufferSize, - const char *fileName, const langType language, - tagEntryFunction tagCallback, passStartCallback passCallback, - void *userData) -{ - if (buffer == NULL && fileName == NULL) - { - error(FATAL, "Neither buffer nor file provided to ctagsParse()"); - return; - } - - createTagsWithFallback(buffer, bufferSize, fileName, language, - tagCallback, passCallback, userData); -} - - -extern const char *ctagsGetLangName(int lang) -{ - return getLanguageName(lang); -} - - -extern int ctagsGetNamedLang(const char *name) -{ - return getNamedLanguage(name, 0); -} - - -extern const char *ctagsGetLangKinds(int lang) -{ - const parserDefinition *def = getParserDefinition(lang); - unsigned int i; - static char kinds[257]; - - for (i = 0; i < def->kindCount; i++) - kinds[i] = def->kindTable[i].letter; - kinds[i] = '\0'; - - return kinds; -} - - -extern const char *ctagsGetKindName(char kind, int lang) -{ - const parserDefinition *def = getParserDefinition(lang); - unsigned int i; - - for (i = 0; i < def->kindCount; i++) - { - if (def->kindTable[i].letter == kind) - return def->kindTable[i].name; - } - return "unknown"; -} - - -extern char ctagsGetKindFromName(const char *name, int lang) -{ - const parserDefinition *def = getParserDefinition(lang); - unsigned int i; - - for (i = 0; i < def->kindCount; i++) - { - if (strcmp(def->kindTable[i].name, name) == 0) - return def->kindTable[i].letter; - } - return '-'; -} - - -extern bool ctagsIsUsingRegexParser(int lang) -{ - return getParserDefinition(lang)->method & METHOD_REGEX; -} - - -extern unsigned int ctagsGetLangCount(void) -{ - return countParsers(); -} - -#endif /* CTAGS_LIB */ Modified: ctags/main/ctags-api.h 57 lines changed, 0 insertions(+), 57 deletions(-) =================================================================== @@ -1,57 +0,0 @@ -/* -* Copyright (c) 2016, Jiri Techet -* -* This source code is released for free distribution under the terms of the -* GNU General Public License version 2 or (at your option) any later version. -* -* Defines ctags API when compiled as a library. -*/ -#ifndef CTAGS_API_H -#define CTAGS_API_H - -#include "general.h" /* must always come first */ - -#ifdef CTAGS_LIB - -#include <stdlib.h> -#include <stdbool.h> - -typedef struct { - const char *name; - const char *signature; - const char *scopeName; - const char *inheritance; - const char *varType; - const char *access; - const char *implementation; - char kindLetter; - bool isFileScope; - unsigned long lineNumber; - int lang; -} ctagsTag; - -/* Callback invoked for every tag found by the parser. The return value is - * currently unused. */ -typedef bool (*tagEntryFunction) (const ctagsTag *const tag, void *userData); - -/* Callback invoked at the beginning of every parsing pass. The return value is - * currently unused */ -typedef bool (*passStartCallback) (void *userData); - - -extern void ctagsInit(void); -extern void ctagsParse(unsigned char *buffer, size_t bufferSize, - const char *fileName, const int language, - tagEntryFunction tagCallback, passStartCallback passCallback, - void *userData); -extern const char *ctagsGetLangName(int lang); -extern int ctagsGetNamedLang(const char *name); -extern const char *ctagsGetLangKinds(int lang); -extern const char *ctagsGetKindName(char kind, int lang); -extern char ctagsGetKindFromName(const char *name, int lang); -extern bool ctagsIsUsingRegexParser(int lang); -extern unsigned int ctagsGetLangCount(void); - -#endif /* CTAGS_LIB */ - -#endif /* CTAGS_API_H */ Modified: ctags/main/ctags.h 10 lines changed, 8 insertions(+), 2 deletions(-) =================================================================== @@ -17,7 +17,7 @@ #if defined (HAVE_CONFIG_H) # define PROGRAM_VERSION PACKAGE_VERSION #else -# define PROGRAM_VERSION "0.0.0" +# define PROGRAM_VERSION "5.9.0" #endif #define PROGRAM_NAME "Universal Ctags" #define PROGRAM_URL "https://ctags.io/" @@ -30,5 +30,11 @@ extern const char* ctags_repoinfo; #define CTAGS_FIELD_PREFIX "UCTAGS" - +/* + * Reserved words + */ +#define RSV_LANGMAP_DEFAULT "default" +#define RSV_LANG_ALL "all" +#define RSV_LANG_AUTO "auto" +#define RSV_NONE "NONE" #endif /* CTAGS_MAIN_CTAGS_H */ Modified: ctags/main/debug.c 103 lines changed, 99 insertions(+), 4 deletions(-) =================================================================== @@ -16,16 +16,22 @@ #include <stdlib.h> #include <stdio.h> #include <stdarg.h> +#include <string.h> #include "debug.h" +#include "entry_p.h" #include "options.h" +#include "parse_p.h" #include "read.h" +#include "read_p.h" /* * FUNCTION DEFINITIONS */ #ifdef DEBUG +#include "htable.h" + extern void lineBreak (void) {} /* provides a line-specified break point */ @@ -74,11 +80,16 @@ extern void debugEntry (const tagEntryInfo *const tag) if (debug (DEBUG_PARSE)) { - printf ("<#%s%s:%s", scope, tag->kind->name, tag->name); - - if (tag->extensionFields.scopeKind != NULL && + langType lang = (tag->extensionFields.scopeLangType == LANG_AUTO) + ? tag->langType + : tag->extensionFields.scopeLangType; + kindDefinition *scopeKindDef = getLanguageKind(lang, + tag->extensionFields.scopeKindIndex); + printf ("<#%s%s:%s", scope, getTagKindName(tag), tag->name); + + if (tag->extensionFields.scopeKindIndex != KIND_GHOST_INDEX && tag->extensionFields.scopeName != NULL) - printf (" [%s:%s]", tag->extensionFields.scopeKind->name, + printf (" [%s:%s]", scopeKindDef->name, tag->extensionFields.scopeName); if (isFieldEnabled (FIELD_INHERITANCE) && @@ -125,4 +136,88 @@ extern void debugAssert (const char *assertion, const char *file, unsigned int l abort(); } +static int debugScopeDepth; +#define DEBUG_INDENT_UNIT 4 + +static char debugPrefix[DEBUG_INDENT_UNIT + 1]; + +extern void debugInit (void) +{ + memset(debugPrefix, ' ', DEBUG_INDENT_UNIT); + debugPrefix[DEBUG_INDENT_UNIT] = '\0'; +} + +extern void debugIndent(void) +{ + for(int i=0;i< debugScopeDepth;i++) + fputs(debugPrefix, stderr); +} + +extern void debugInc(void) +{ + debugScopeDepth++; +} + +extern void debugDec(void) +{ + debugScopeDepth--; + if(debugScopeDepth < 0) + debugScopeDepth = 0; +} + + + +struct circularRefChecker { + hashTable *visitTable; + int counter; +}; + +extern void circularRefCheckerDestroy (struct circularRefChecker * checker) +{ + hashTableDelete (checker->visitTable); + checker->visitTable = NULL; + eFree (checker); +} + +extern struct circularRefChecker * circularRefCheckerNew (void) +{ + Assert (sizeof(void *) >= sizeof(int)); + + struct circularRefChecker *c = xMalloc (1, struct circularRefChecker); + + c->visitTable = hashTableNew (17, hashPtrhash, hashPtreq, NULL, NULL); + c->counter = 0; + + return c; +} + +extern int circularRefCheckerCheck (struct circularRefChecker *c, void *ptr) +{ + union conv { + int i; + void *ptr; + } v; + + v.ptr = hashTableGetItem(c->visitTable, ptr); + if (v.ptr) + return v.i; + else + { + v.i = ++c->counter; + hashTablePutItem (c->visitTable, ptr, v.ptr); + return 0; + } +} + +extern int circularRefCheckerGetCurrent (struct circularRefChecker *c) +{ + return c->counter; +} + +extern void circularRefCheckClear (struct circularRefChecker *c) +{ + hashTableClear (c->visitTable); + c->counter = 0; +} + #endif Modified: ctags/main/debug.h 32 lines changed, 26 insertions(+), 6 deletions(-) =================================================================== @@ -14,27 +14,26 @@ */ #include "general.h" /* must always come first */ +#include "gvars.h" +#include "types.h" #ifdef DEBUG # include <assert.h> #endif -#include "entry.h" /* * Macros */ #ifdef DEBUG -# define debug(level) ((Option.debugLevel & (long)(level)) != 0) +# define debug(level) ((ctags_debugLevel & (long)(level)) != 0) # define DebugStatement(x) x # define PrintStatus(x) if (debug(DEBUG_STATUS)) printf x; # ifdef NDEBUG # define Assert(c) do {} while(0) # define AssertNotReached() do {} while(0) # else - /* based on glibc's assert.h __ASSERT_FUNCTION */ -# if defined (__GNUC__) && (__GNUC__ > 2 || (__GNUC__ == 2 && __GNUC_MINOR__ >= 4)) -# define ASSERT_FUNCTION __PRETTY_FUNCTION__ -# elif defined __STDC_VERSION__ && __STDC_VERSION__ >= 199901L + /* We expect cc supports c99 standard. */ +# if defined __STDC_VERSION__ && __STDC_VERSION__ >= 199901L # define ASSERT_FUNCTION __func__ # else # define ASSERT_FUNCTION ((const char*)0) @@ -52,6 +51,10 @@ # endif #endif +#ifdef DEBUG +/* This makes valgrind report an error earlier. */ +#define DISABLE_OBJPOOL +#endif /* * Data declarations */ @@ -79,4 +82,21 @@ extern void debugCppIgnore (const bool ignore); extern void debugEntry (const tagEntryInfo *const tag); extern void debugAssert (const char *assertion, const char *file, unsigned int line, const char *function) attr__noreturn; +#ifdef DEBUG +#define DEBUG_INIT() debugInit() +extern void debugInit (void); +extern void debugIndent(void); +extern void debugInc(void); +extern void debugDec(void); + +struct circularRefChecker; +extern struct circularRefChecker * circularRefCheckerNew (void); +extern void circularRefCheckerDestroy (struct circularRefChecker * checker); +extern int circularRefCheckerCheck (struct circularRefChecker *c, void *ptr); +extern int circularRefCheckerGetCurrent (struct circularRefChecker *c); +extern void circularRefCheckClear (struct circularRefChecker *c); + +#else +#define DEBUG_INIT() do { } while(0) +#endif /* DEBUG */ #endif /* CTAGS_MAIN_DEBUG_H */ Modified: ctags/main/dependency.c 353 lines changed, 303 insertions(+), 50 deletions(-) =================================================================== @@ -12,91 +12,344 @@ #include "general.h" /* must always come first */ +#include "debug.h" #include "dependency.h" -#include "parse.h" +#include "options.h" +#include "parse_p.h" +#include "read.h" +#include "read_p.h" +#include "routines.h" +#include "subparser.h" +#include "subparser_p.h" +#include "xtag.h" #include <string.h> +struct slaveControlBlock { + slaveParser *slaveParsers; /* The parsers on this list must be initialized when + this parser is initialized. */ + subparser *subparsersDefault; + subparser *subparsersInUse; + langType owner; +}; -static void linkKinds (kindDefinition *masterKind, kindDefinition *slaveKind) +extern void linkDependencyAtInitializeParsing (depType dtype, + parserDefinition *const master, + struct slaveControlBlock *masterSCB, + struct kindControlBlock *masterKCB, + parserDefinition *const slave, + struct kindControlBlock *slaveKCB, + void *data) { - kindDefinition *tail; + if (dtype == DEPTYPE_KIND_OWNER) + linkKindDependency (masterKCB, slaveKCB); + else if (dtype == DEPTYPE_SUBPARSER) + { + slaveParser *s = xMalloc (1, slaveParser); - slaveKind->master = masterKind; + s->type = dtype; + s->id = slave->id; + s->data = data; - tail = slaveKind; - while (tail->slave) - { - tail->enabled = masterKind->enabled; - tail = tail->slave; + s->next = masterSCB->slaveParsers; + masterSCB->slaveParsers = s; } +} - tail->slave = masterKind->slave; - masterKind->slave = slaveKind; +static void attachSubparser (struct slaveControlBlock *base_sb, subparser *subparser) +{ + subparser->next = base_sb->subparsersDefault; + base_sb->subparsersDefault = subparser; } -static void linkKindDependency (parserDefinition *const masterParser, - parserDefinition *const slaveParser) + +extern struct slaveControlBlock *allocSlaveControlBlock (parserDefinition *parser) { - unsigned int k_slave, k_master; - kindDefinition *kind_slave, *kind_master; + struct slaveControlBlock *cb; - for (k_slave = 0; k_slave < slaveParser->kindCount; k_slave++) + cb = xMalloc (1, struct slaveControlBlock); + cb->slaveParsers = NULL; + cb->subparsersDefault = NULL; + cb->subparsersInUse = NULL; + cb->owner = parser->id; + + return cb; +} + +extern void freeSlaveControlBlock (struct slaveControlBlock *cb) +{ + eFree (cb); +} + +extern void initializeDependencies (parserDefinition *parser, + struct slaveControlBlock *cb) +{ + unsigned int i; + slaveParser *sp; + + /* Initialize slaves */ + sp = cb->slaveParsers; + while (sp != NULL) { - if (slaveParser->kindTable [k_slave].syncWith == LANG_AUTO) + if (sp->type == DEPTYPE_SUBPARSER) { - kind_slave = slaveParser->kindTable + k_slave; - for (k_master = 0; k_master < masterParser->kindCount; k_master++) + subparser *sub; + + sub = (subparser *)sp->data; + sub->slaveParser = sp; + } + + if (sp->type == DEPTYPE_KIND_OWNER + || (sp->type == DEPTYPE_SUBPARSER && + (((subparser *)sp->data)->direction & SUBPARSER_BASE_RUNS_SUB))) + { + initializeParser (sp->id); + if (sp->type == DEPTYPE_SUBPARSER + && isXtagEnabled (XTAG_SUBPARSER)) { - kind_master = masterParser->kindTable + k_master; - if ((kind_slave->letter == kind_master->letter) - && (strcmp (kind_slave->name, kind_master->name) == 0)) - { - linkKinds (kind_master, kind_slave); - kind_slave->syncWith = masterParser->id; - kind_master->syncWith = masterParser->id; - break; - } + subparser *subparser = sp->data; + attachSubparser (cb, subparser); } } + sp = sp->next; + } + + /* Initialize masters that act as base parsers. */ + for (i = 0; i < parser->dependencyCount; i++) + { + parserDependency *d = parser->dependencies + i; + if (d->type == DEPTYPE_SUBPARSER && + ((subparser *)(d->data))->direction & SUBPARSER_SUB_RUNS_BASE) + { + langType baseParser; + baseParser = getNamedLanguage (d->upperParser, 0); + Assert (baseParser != LANG_IGNORE); + initializeParser (baseParser); + } } } -extern void linkDependencyAtInitializeParsing (depType dtype, - parserDefinition *const masterParser, - parserDefinition *const slaveParser) +extern void finalizeDependencies (parserDefinition *parser, + struct slaveControlBlock *cb) { - if (dtype == DEPTYPE_KIND_OWNER) - linkKindDependency (masterParser, slaveParser); - else if (dtype == DEPTYPE_SUBPARSER) + while (cb->slaveParsers) { - subparser *s = xMalloc (1, subparser); + slaveParser *sp = cb->slaveParsers; + cb->slaveParsers = sp->next; + sp->next = NULL; + eFree (sp); + } +} + +extern void notifyInputStart (void) +{ + subparser *s; - s->id = slaveParser->id; - s->next = masterParser->subparsers; - masterParser->subparsers = s; + foreachSubparser(s, false) + { + langType lang = getSubparserLanguage (s); + notifyLanguageRegexInputStart (lang); + + if (s->inputStart) + { + enterSubparser(s); + s->inputStart (s); + leaveSubparser(); + } } } -extern void initializeSubparsers (const parserDefinition *parser) +extern void notifyInputEnd (void) { - subparser *sp; + subparser *s; + + foreachSubparser(s, false) + { + if (s->inputEnd) + { + enterSubparser(s); + s->inputEnd (s); + leaveSubparser(); + } + + langType lang = getSubparserLanguage (s); + notifyLanguageRegexInputEnd (lang); + } +} - for (sp = parser->subparsers; sp; sp = sp->next) - initializeParser (sp->id); +extern void notifyMakeTagEntry (const tagEntryInfo *tag, int corkIndex) +{ + subparser *s; + + foreachSubparser(s, false) + { + if (s->makeTagEntryNotify) + { + enterSubparser(s); + s->makeTagEntryNotify (s, tag, corkIndex); + leaveSubparser(); + } + } +} + +extern langType getSubparserLanguage (subparser *s) +{ + return s->slaveParser->id; +} + +extern void chooseExclusiveSubparser (subparser *s, void *data) +{ + if (s->exclusiveSubparserChosenNotify) + { + s->chosenAsExclusiveSubparser = true; + enterSubparser(s); + s->exclusiveSubparserChosenNotify (s, data); + verbose ("%s is chosen as exclusive subparser\n", + getLanguageName (getSubparserLanguage (s))); + leaveSubparser(); + } } -extern void finalizeSubparsers (parserDefinition *parser) +extern subparser *getFirstSubparser(struct slaveControlBlock *controlBlock) +{ + if (controlBlock) + return controlBlock->subparsersInUse; + return NULL; +} + +extern void useDefaultSubparsers (struct slaveControlBlock *controlBlock) +{ + controlBlock->subparsersInUse = controlBlock->subparsersDefault; +} + +extern void useSpecifiedSubparser (struct slaveControlBlock *controlBlock, subparser *s) +{ + s->schedulingBaseparserExplicitly = true; + controlBlock->subparsersInUse = s; +} + +extern void setupSubparsersInUse (struct slaveControlBlock *controlBlock) +{ + if (!controlBlock->subparsersInUse) + useDefaultSubparsers(controlBlock); +} + +extern subparser* teardownSubparsersInUse (struct slaveControlBlock *controlBlock) { - subparser *sp; subparser *tmp; + subparser *s = NULL; - for (sp = parser->subparsers; sp;) + tmp = controlBlock->subparsersInUse; + controlBlock->subparsersInUse = NULL; + + if (tmp && tmp->schedulingBaseparserExplicitly) { - tmp = sp; - sp = sp->next; - tmp->next = NULL; - eFree (tmp); + tmp->schedulingBaseparserExplicitly = false; + s = tmp; } - parser->subparsers = NULL; + + if (s) + return s; + + while (tmp) + { + if (tmp->chosenAsExclusiveSubparser) + { + s = tmp; + } + tmp = tmp->next; + } + + return s; +} + + +static int subparserDepth; + +extern void enterSubparser(subparser *subparser) +{ + subparserDepth++; + pushLanguage (getSubparserLanguage (subparser)); +} + +extern void leaveSubparser(void) +{ + popLanguage (); + subparserDepth--; +} + +extern bool doesSubparserRun (void) +{ + if (getLanguageForBaseParser () == getInputLanguage()) + return false; + return subparserDepth; +} + +extern slaveParser *getFirstSlaveParser (struct slaveControlBlock *scb) +{ + if (scb) + return scb->slaveParsers; + return NULL; +} + +extern struct colprintTable * subparserColprintTableNew (void) +{ + return colprintTableNew ("L:NAME", "L:BASEPARSER", "L:DIRECTIONS", NULL); +} + +extern void subparserColprintAddSubparsers (struct colprintTable *table, + struct slaveControlBlock *scb) +{ + slaveParser *tmp; + + pushLanguage (scb->owner); + foreachSlaveParser(tmp) + { + struct colprintLine *line = colprintTableGetNewLine(table); + + colprintLineAppendColumnCString (line, getLanguageName (tmp->id)); + colprintLineAppendColumnCString (line, getLanguageName (scb->owner)); + + const char *direction; + switch (((subparser *)tmp->data)->direction) + { + case SUBPARSER_BASE_RUNS_SUB: + direction = "base => sub {shared}"; + break; + case SUBPARSER_SUB_RUNS_BASE: + direction = "base <= sub {dedicated}"; + break; + case SUBPARSER_BI_DIRECTION: + direction = "base <> sub {bidirectional}"; + break; + default: + direction = "UNKNOWN(INTERNAL BUG)"; + break; + } + colprintLineAppendColumnCString (line, direction); + } + popLanguage (); +} + +static int subparserColprintCompareLines (struct colprintLine *a , struct colprintLine *b) +{ + const char *a_name = colprintLineGetColumn (a, 0); + const char *b_name = colprintLineGetColumn (b, 0); + + int r; + r = strcmp (a_name, b_name); + if (r != 0) + return r; + + const char *a_baseparser = colprintLineGetColumn (a, 1); + const char *b_baseparser = colprintLineGetColumn (b, 1); + + return strcmp(a_baseparser, b_baseparser); +} + +extern void subparserColprintTablePrint (struct colprintTable *table, + bool withListHeader, bool machinable, FILE *fp) +{ + colprintTableSort (table, subparserColprintCompareLines); + colprintTablePrint (table, 0, withListHeader, machinable, fp); } Modified: ctags/main/dependency.h 26 lines changed, 13 insertions(+), 13 deletions(-) =================================================================== @@ -12,34 +12,34 @@ #ifndef CTAGS_MAIN_DEPENDENCY_H #define CTAGS_MAIN_DEPENDENCY_H -#include "general.h" +/* +* INCLUDE FILES +*/ +#include "general.h" /* must always come first */ #include "types.h" +/* +* DATA DECLARATIONS +*/ typedef enum eDepType { DEPTYPE_KIND_OWNER, DEPTYPE_SUBPARSER, COUNT_DEPTYPES, } depType; -typedef struct sParserDependency { +struct sParserDependency { depType type; const char *upperParser; void *data; -} parserDependency; - -extern void linkDependencyAtInitializeParsing (depType dtype, - parserDefinition *const masterParser, - parserDefinition *const slaveParser); +}; -typedef struct sSubparser subparser; -struct sSubparser { +struct sSlaveParser { + depType type; langType id; - subparser *next; + void *data; + slaveParser *next; }; -extern void initializeSubparsers (const parserDefinition *parser); -extern void finalizeSubparsers (parserDefinition *parser); - #endif /* CTAGS_MAIN_DEPENDENCY_H */ Modified: ctags/main/dependency_p.h 58 lines changed, 58 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,58 @@ +/* + * + * Copyright (c) 2016, Red Hat, Inc. + * Copyright (c) 2016, Masatake YAMATO + * + * Author: Masatake YAMATO <yamato(a)redhat.com> + * + * This source code is released for free distribution under the terms of the + * GNU General Public License version 2 or (at your option) any later version. + * + */ +#ifndef CTAGS_MAIN_DEPENDENCY_PRIVATE_H +#define CTAGS_MAIN_DEPENDENCY_PRIVATE_H + +/* +* INCLUDE FILES +*/ +#include "general.h" /* must always come first */ + +#include "dependency.h" +#include "kind.h" +#include "types.h" + +/* +* MACROS +*/ +#define foreachSlaveParser(VAR) \ + VAR = NULL; \ + while ((VAR = getNextSlaveParser (VAR)) != NULL) + + +/* +* DATA DECLARATIONS +*/ +struct slaveControlBlock; /* Opaque data type for parse.c */ + +/* +* FUNCTION PROTOTYPES +*/ +extern void linkDependencyAtInitializeParsing (depType dtype, + parserDefinition *const master, + struct slaveControlBlock *masterSCB, + struct kindControlBlock *masterKCB, + parserDefinition *const slave, + struct kindControlBlock *slaveKCB, + void *data); + +extern struct slaveControlBlock *allocSlaveControlBlock (parserDefinition *parser); +extern void freeSlaveControlBlock (struct slaveControlBlock *cb); +extern void initializeDependencies (parserDefinition *parser, + struct slaveControlBlock *cb); +extern void finalizeDependencies (parserDefinition *parser, + struct slaveControlBlock *cb); + +extern slaveParser *getFirstSlaveParser(struct slaveControlBlock *controlBlock); +extern slaveParser *getNextSlaveParser(slaveParser *last); + +#endif /* CTAGS_MAIN_DEPENDENCY_PRIVATE_H */ Modified: ctags/main/e_msoft.h 23 lines changed, 10 insertions(+), 13 deletions(-) =================================================================== @@ -14,34 +14,30 @@ #define MSDOS_STYLE_PATH 1 #define HAVE_FCNTL_H 1 #define HAVE_IO_H 1 -#define HAVE_LIMITS_H 1 -#define HAVE_STDLIB_H 1 #define HAVE_SYS_STAT_H 1 #define HAVE_SYS_TYPES_H 1 -#define HAVE_TIME_H 1 -#define HAVE_CLOCK 1 #define HAVE_CHSIZE 1 -#define HAVE_FGETPOS 1 +#define HAVE_DIRECT_H 1 #define HAVE_STRICMP 1 #define HAVE_STRNICMP 1 #define HAVE_STRSTR 1 #define HAVE_STRERROR 1 +#define HAVE__FINDFIRST 1 #define HAVE_FINDNEXT 1 -#define HAVE_TEMPNAM 1 +#define findfirst_t intptr_t +#define HAVE_MKSTEMP 1 #define HAVE_FNMATCH 1 #define HAVE_FNMATCH_H 1 #define HAVE_PUTENV 1 -#define tempnam(dir,pfx) _tempnam(dir,pfx) #define TMPDIR "\\" +int mkstemp (char *template_name); + #ifdef _MSC_VER -# define HAVE__FINDFIRST 1 -# define HAVE_DIRECT_H 1 # if _MSC_VER < 1900 # define snprintf _snprintf # endif -# define findfirst_t intptr_t #if (_MSC_VER >= 1800) // Visual Studio 2013 or newer #define HAVE_STDBOOL_H 1 @@ -58,14 +54,15 @@ typedef enum { false, true } bool; # include <_mingw.h> # define HAVE_STDBOOL_H 1 -# define HAVE_DIR_H 1 # define HAVE_DIRENT_H 1 -# define HAVE__FINDFIRST 1 -# define findfirst_t long # define ffblk _finddata_t # define FA_DIREC _A_SUBDIR # define ff_name name +# if defined(__USE_MINGW_ANSI_STDIO) && defined(__MINGW64_VERSION_MAJOR) +# define HAVE_ASPRINTF 1 +# endif + #endif #endif Modified: ctags/main/entry.c 1190 lines changed, 880 insertions(+), 310 deletions(-) =================================================================== @@ -36,20 +36,30 @@ # include <io.h> #endif +#include <stdint.h> +#include <limits.h> /* to define INT_MAX */ + #include "debug.h" -#include "entry.h" +#include "entry_p.h" #include "field.h" -#include "fmt.h" +#include "fmt_p.h" #include "kind.h" -#include "main.h" -#include "options.h" -#include "output.h" -#include "ptag.h" +#include "nestlevel.h" +#include "options_p.h" +#include "ptag_p.h" +#include "rbtree.h" #include "read.h" +#include "read_p.h" #include "routines.h" -#include "sort.h" +#include "routines_p.h" +#include "parse_p.h" +#include "ptrarray.h" +#include "sort_p.h" #include "strlist.h" -#include "xtag.h" +#include "subparser_p.h" +#include "trashbox.h" +#include "writer_p.h" +#include "xtag_p.h" /* * MACROS @@ -82,43 +92,38 @@ typedef struct eTagFile { struct sMax { size_t line, tag; } max; vString *vLine; - unsigned int cork; - struct sCorkQueue { - struct sTagEntryInfo* queue; - unsigned int length; - unsigned int count; - } corkQueue; + int cork; + unsigned int corkFlags; + ptrArray *corkQueue; bool patternCacheValid; } tagFile; +typedef struct sTagEntryInfoX { + tagEntryInfo slot; + int corkIndex; + struct rb_root symtab; + struct rb_node symnode; +} tagEntryInfoX; + /* * DATA DEFINITIONS */ -tagFile TagFile = { +static tagFile TagFile = { NULL, /* tag file name */ NULL, /* tag file directory (absolute) */ NULL, /* file pointer */ { 0, 0 }, /* numTags */ { 0, 0 }, /* max */ NULL, /* vLine */ .cork = false, - .corkQueue = { - .queue = NULL, - .length = 0, - .count = 0 - }, + .corkQueue = NULL, .patternCacheValid = false, }; static bool TagsToStdout = false; -#ifdef CTAGS_LIB -static tagEntryFunction TagEntryFunction = NULL; -static void *TagEntryUserData = NULL; -#endif - /* * FUNCTION PROTOTYPES */ @@ -152,10 +157,8 @@ extern const char *tagFileName (void) extern void abort_if_ferror(MIO *const mio) { -#ifndef CTAGS_LIB - if (mio_error (mio)) + if (mio != NULL && mio_error (mio)) error (FATAL | PERROR, "cannot write tag file"); -#endif } static void rememberMaxLengths (const size_t nameLength, const size_t lineLength) @@ -169,52 +172,38 @@ static void rememberMaxLengths (const size_t nameLength, const size_t lineLength static void addCommonPseudoTags (void) { - int i; - - for (i = 0; i < PTAG_COUNT; i++) + for (int i = 0; i < PTAG_COUNT; i++) { if (isPtagCommonInParsers (i)) - makePtagIfEnabled (i, NULL); + makePtagIfEnabled (i, LANG_IGNORE, &Option); } } extern void makeFileTag (const char *const fileName) { - xtagType xtag = XTAG_UNKNOWN; - - if (isXtagEnabled(XTAG_FILE_NAMES)) - xtag = XTAG_FILE_NAMES; - - if (xtag != XTAG_UNKNOWN) - { - tagEntryInfo tag; - kindDefinition *kind; - - kind = getInputLanguageFileKind(); - Assert (kind); - kind->enabled = isXtagEnabled(XTAG_FILE_NAMES); - - /* TODO: you can return here if enabled == false. */ + tagEntryInfo tag; - initTagEntry (&tag, baseFilename (fileName), KIND_FILE_INDEX); + if (!isXtagEnabled(XTAG_FILE_NAMES)) + return; - tag.isFileEntry = true; - tag.lineNumberEntry = true; - markTagExtraBit (&tag, xtag); + initTagEntry (&tag, baseFilename (fileName), KIND_FILE_INDEX); - tag.lineNumber = 1; - if (isFieldEnabled (FIELD_END)) - { - /* isFieldEnabled is called again in the rendering - stage. However, it is called here for avoiding - unnecessary read line loop. */ - while (readLineFromInputFile () != NULL) - ; /* Do nothing */ - tag.extensionFields.endLine = getInputLineNumber (); - } + tag.isFileEntry = true; + tag.lineNumberEntry = true; + markTagExtraBit (&tag, XTAG_FILE_NAMES); - makeTagEntry (&tag); + tag.lineNumber = 1; + if (isFieldEnabled (FIELD_END_LINE)) + { + /* isFieldEnabled is called again in the rendering + stage. However, it is called here for avoiding + unnecessary read line loop. */ + while (readLineFromInputFile () != NULL) + ; /* Do nothing */ + tag.extensionFields.endLine = getInputLineNumber (); } + + makeTagEntry (&tag); } static void updateSortedFlag ( @@ -382,7 +371,7 @@ static bool isTagFile (const char *const filename) ok = true; else ok = (bool) (isCtagsLine (line) || isEtagsLine (line)); - mio_free (mio); + mio_unref (mio); } return ok; } @@ -399,9 +388,13 @@ extern void openTagFile (void) */ if (TagsToStdout) { - /* Open a tempfile with read and write mode. Read mode is used when - * write the result to stdout. */ - TagFile.mio = tempFile ("w+", &TagFile.name); + if (Option.interactive == INTERACTIVE_SANDBOX) + { + TagFile.mio = mio_new_memory (NULL, 0, eRealloc, eFreeNoNullCheck); + TagFile.name = NULL; + } + else + TagFile.mio = tempFile ("w+", &TagFile.name); if (isXtagEnabled (XTAG_PSEUDO_TAGS)) addCommonPseudoTags (); } @@ -431,7 +424,7 @@ extern void openTagFile (void) if (TagFile.mio != NULL) { TagFile.numTags.prev = updatePseudoTags (TagFile.mio); - mio_free (TagFile.mio); + mio_unref (TagFile.mio); TagFile.mio = mio_new_file (TagFile.name, "a+"); } } @@ -445,10 +438,14 @@ extern void openTagFile (void) if (TagFile.mio == NULL) error (FATAL | PERROR, "cannot open tag file"); } - if (TagsToStdout) - TagFile.directory = eStrdup (CurrentDirectory); - else - TagFile.directory = absoluteDirname (TagFile.name); + + if (TagFile.directory == NULL) + { + if (TagsToStdout) + TagFile.directory = eStrdup (CurrentDirectory); + else + TagFile.directory = absoluteDirname (TagFile.name); + } } #ifdef USE_REPLACEMENT_TRUNCATE @@ -485,19 +482,20 @@ static void copyFile (const char *const from, const char *const to, const long s else { copyBytes (fromMio, toMio, size); - mio_free (toMio); + mio_unref (toMio); } - mio_free (fromMio); + mio_unref (fromMio); } } /* Replacement for missing library function. */ static int replacementTruncate (const char *const name, const long size) { +#define WHOLE_FILE -1L char *tempName = NULL; MIO *mio = tempFile ("w", &tempName); - mio_free (mio); + mio_unref (mio); copyFile (name, tempName, size); copyFile (tempName, name, WHOLE_FILE); remove (tempName); @@ -532,7 +530,7 @@ static void internalSortTagFile (void) TagFile.numTags.added + TagFile.numTags.prev); if (! TagsToStdout) - mio_free (mio); + mio_unref (mio); } #endif @@ -558,6 +556,12 @@ static void resizeTagFile (const long newSize) { int result; + if (!TagFile.name) + { + mio_try_resize (TagFile.mio, newSize); + return; + } + #ifdef USE_REPLACEMENT_TRUNCATE result = replacementTruncate (TagFile.name, newSize); #else @@ -605,30 +609,35 @@ extern void closeTagFile (const bool resize) if (Option.etags) writeEtagsIncludes (TagFile.mio); mio_flush (TagFile.mio); + abort_if_ferror (TagFile.mio); desiredSize = mio_tell (TagFile.mio); mio_seek (TagFile.mio, 0L, SEEK_END); size = mio_tell (TagFile.mio); if (! TagsToStdout) /* The tag file should be closed before resizing. */ - if (mio_free (TagFile.mio) != 0) + if (mio_unref (TagFile.mio) != 0) error (FATAL | PERROR, "cannot close tag file"); if (resize && desiredSize < size) { DebugStatement ( debugPrintf (DEBUG_STATUS, "shrinking %s from %ld to %ld bytes\n", - TagFile.name, size, desiredSize); ) + TagFile.name? TagFile.name: "<mio>", size, desiredSize); ) resizeTagFile (desiredSize); } sortTagFile (); if (TagsToStdout) { - if (mio_free (TagFile.mio) != 0) + if (mio_unref (TagFile.mio) != 0) error (FATAL | PERROR, "cannot close tag file"); - remove (tagFileName ()); /* remove temporary file */ + if (TagFile.name) + remove (TagFile.name); /* remove temporary file */ } - eFree (TagFile.name); + + TagFile.mio = NULL; + if (TagFile.name) + eFree (TagFile.name); TagFile.name = NULL; } @@ -641,10 +650,13 @@ extern void closeTagFile (const bool resize) * are doubled and a leading '^' or trailing '$' is also quoted. End of line * characters (line feed or carriage return) are dropped. */ -static size_t appendInputLine (int putc_func (char , void *), const char *const line, void * data, bool *omitted) +static size_t appendInputLine (int putc_func (char , void *), const char *const line, + unsigned int patternLengthLimit, + void * data, bool *omitted) { size_t length = 0; const char *p; + int extraLength = 0; /* Write everything up to, but not including, a line end character. */ @@ -657,7 +669,11 @@ static size_t appendInputLine (int putc_func (char , void *), const char *const if (c == CRETURN || c == NEWLINE) break; - if (Option.patternLengthLimit != 0 && length >= Option.patternLengthLimit) + if (patternLengthLimit != 0 && length >= patternLengthLimit && + /* Do not cut inside a multi-byte UTF-8 character, but safe-guard it not to + * allow more than one extra valid UTF-8 character in case it's not actually + * UTF-8. To do that, limit to an extra 3 UTF-8 sub-bytes (0b10xxxxxx). */ + ((((unsigned char) c) & 0xc0) != 0x80 || ++extraLength > 3)) { *omitted = true; break; @@ -687,11 +703,12 @@ static int vstring_putc (char c, void *data) static int vstring_puts (const char* s, void *data) { vString *str = data; - int len = vStringLength (str); + size_t len = vStringLength (str); vStringCatS (str, s); - return vStringLength (str) - len; + return (int) (vStringLength (str) - len); } +#ifdef DEBUG static bool isPosSet(MIOPos pos) { char * p = (char *)&pos; @@ -702,26 +719,23 @@ static bool isPosSet(MIOPos pos) r |= p[i]; return r; } +#endif -extern char *readLineFromBypassAnyway (vString *const vLine, const tagEntryInfo *const tag, +extern char *readLineFromBypassForTag (vString *const vLine, const tagEntryInfo *const tag, long *const pSeekValue) { - char * line; - - if (isPosSet (tag->filePosition) || (tag->pattern == NULL)) - line = readLineFromBypass (vLine, tag->filePosition, pSeekValue); - else - line = readLineFromBypassSlow (vLine, tag->lineNumber, tag->pattern, pSeekValue); - - return line; + Assert (isPosSet (tag->filePosition) || (tag->pattern == NULL)); + return readLineFromBypass (vLine, tag->filePosition, pSeekValue); } /* Truncates the text line containing the tag at the character following the * tag, providing a character which designates the end of the tag. + * Returns the length of the truncated line (or 0 if it doesn't truncate). */ -extern void truncateTagLine ( +extern size_t truncateTagLineAfterTag ( char *const line, const char *const token, const bool discardNewline) { + size_t len = 0; char *p = strstr (line, token); if (p != NULL) @@ -730,7 +744,10 @@ extern void truncateTagLine ( if (*p != '\0' && ! (*p == '\n' && discardNewline)) ++p; /* skip past character terminating character */ *p = '\0'; + len = p - line; } + + return len; } static char* getFullQualifiedScopeNameFromCorkQueue (const tagEntryInfo * inner_scope) @@ -739,6 +756,7 @@ static char* getFullQualifiedScopeNameFromCorkQueue (const tagEntryInfo * inner_ int kindIndex = KIND_GHOST_INDEX; langType lang; const tagEntryInfo *scope = inner_scope; + const tagEntryInfo *root_scope = NULL; stringList *queue = stringListNew (); vString *v; vString *n; @@ -760,11 +778,16 @@ static char* getFullQualifiedScopeNameFromCorkQueue (const tagEntryInfo * inner_ stringListAdd (queue, v); kindIndex = scope->kindIndex; lang = scope->langType; + root_scope = scope; } scope = getEntryInCorkQueue (scope->extensionFields.scopeIndex); } n = vStringNew (); + sep = root_scope? scopeSeparatorFor (root_scope->langType, root_scope->kindIndex, KIND_GHOST_INDEX): NULL; + if (sep) + vStringCatS(n, sep); + while ((c = stringListCount (queue)) > 0) { v = stringListLast (queue); @@ -785,16 +808,13 @@ extern void getTagScopeInformation (tagEntryInfo *const tag, if (name) *name = NULL; + const tagEntryInfo * scope = getEntryInCorkQueue (tag->extensionFields.scopeIndex); if (tag->extensionFields.scopeKindIndex == KIND_GHOST_INDEX && tag->extensionFields.scopeName == NULL - && tag->extensionFields.scopeIndex != CORK_NIL - && TagFile.corkQueue.count > 0) + && scope + && ptrArrayCount (TagFile.corkQueue) > 0) { - const tagEntryInfo * scope = NULL; - char *full_qualified_scope_name = NULL; - - scope = getEntryInCorkQueue (tag->extensionFields.scopeIndex); - full_qualified_scope_name = getFullQualifiedScopeNameFromCorkQueue(scope); + char *full_qualified_scope_name = getFullQualifiedScopeNameFromCorkQueue(scope); Assert (full_qualified_scope_name); /* Make the information reusable to generate full qualified entry, and xformat output*/ @@ -821,9 +841,9 @@ extern void getTagScopeInformation (tagEntryInfo *const tag, } -extern int makePatternStringCommon (const tagEntryInfo *const tag, - int putc_func (char , void *), - int puts_func (const char* , void *), +static int makePatternStringCommon (const tagEntryInfo *const tag, + int (* putc_func) (char , void *), + int (* puts_func) (const char* , void *), void *output) { int length = 0; @@ -845,20 +865,33 @@ extern int makePatternStringCommon (const tagEntryInfo *const tag, && (memcmp (&tag->filePosition, &cached_location, sizeof(MIOPos)) == 0)) return puts_func (vStringValue (cached_pattern), output); - line = readLineFromBypass (TagFile.vLine, tag->filePosition, NULL); + line = readLineFromBypassForTag (TagFile.vLine, tag, NULL); if (line == NULL) - error (FATAL, "could not read tag line from %s at line %lu", getInputFileName (),tag->lineNumber); + { + /* This can be occurs if the size of input file is zero, and + an empty regex pattern (//) matches to the input. */ + line = ""; + line_len = 0; + } + else + line_len = vStringLength (TagFile.vLine); + if (tag->truncateLineAfterTag) - truncateTagLine (line, tag->name, false); + { + size_t truncted_len; + + truncted_len = truncateTagLineAfterTag (line, tag->name, false); + if (truncted_len > 0) + line_len = truncted_len; + } - line_len = strlen (line); searchChar = Option.backward ? '?' : '/'; - terminator = (bool) (line [line_len - 1] == '\n') ? "$": ""; + terminator = (line_len > 0 && (line [line_len - 1] == '\n')) ? "$": ""; if (!tag->truncateLineAfterTag) { making_cache = true; - cached_pattern = vStringNewOrClear (cached_pattern); + cached_pattern = vStringNewOrClearWithAutoRelease (cached_pattern); puts_o_func = puts_func; o_output = output; @@ -870,7 +903,8 @@ extern int makePatternStringCommon (const tagEntryInfo *const tag, length += putc_func(searchChar, output); if ((tag->boundaryInfo & BOUNDARY_START) == 0) length += putc_func('^', output); - length += appendInputLine (putc_func, line, output, &omitted); + length += appendInputLine (putc_func, line, Option.patternLengthLimit, + output, &omitted); length += puts_func (omitted? "": terminator, output); length += putc_func (searchChar, output); @@ -891,30 +925,97 @@ extern char* makePatternString (const tagEntryInfo *const tag) return vStringDeleteUnwrap (pattern); } -extern void attachParserField (tagEntryInfo *const tag, fieldType ftype, const char * value) +static tagField * tagFieldNew(fieldType ftype, const char *value, bool valueOwner) { - Assert (tag->usedParserFields < PRE_ALLOCATED_PARSER_FIELDS); + tagField *f = xMalloc (1, tagField); - tag->parserFields [tag->usedParserFields].ftype = ftype; - tag->parserFields [tag->usedParserFields].value = value; - tag->usedParserFields++; + f->ftype = ftype; + f->value = value; + f->valueOwner = valueOwner; + return f; } -extern void attachParserFieldToCorkEntry (int index, - fieldType type, - const char *value) +static void tagFieldDelete (tagField * f) { - tagEntryInfo * tag; - const char * v; + if (f->valueOwner) + eFree((void *)f->value); + eFree (f); +} - if (index == CORK_NIL) - return; +static void attachParserFieldGeneric (tagEntryInfo *const tag, fieldType ftype, const char * value, + bool valueOwner) +{ + if (tag->usedParserFields < PRE_ALLOCATED_PARSER_FIELDS) + { + tag->parserFields [tag->usedParserFields].ftype = ftype; + tag->parserFields [tag->usedParserFields].value = value; + tag->parserFields [tag->usedParserFields].valueOwner = valueOwner; + tag->usedParserFields++; + } + else if (tag->parserFieldsDynamic == NULL) + { + tag->parserFieldsDynamic = ptrArrayNew((ptrArrayDeleteFunc)tagFieldDelete); + PARSER_TRASH_BOX(tag->parserFieldsDynamic, ptrArrayDelete); + attachParserFieldGeneric (tag, ftype, value, valueOwner); + } + else + { + tagField *f = tagFieldNew (ftype, value, valueOwner); + ptrArrayAdd(tag->parserFieldsDynamic, f); + tag->usedParserFields++; + } +} - tag = getEntryInCorkQueue(index); +extern void attachParserField (tagEntryInfo *const tag, bool inCorkQueue, fieldType ftype, const char * value) +{ Assert (tag != NULL); - v = eStrdup (value); - attachParserField (tag, type, v); + if (inCorkQueue) + { + const char * v; + v = eStrdup (value); + + bool dynfields_allocated = tag->parserFieldsDynamic? true: false; + attachParserFieldGeneric (tag, ftype, v, true); + if (!dynfields_allocated && tag->parserFieldsDynamic) + PARSER_TRASH_BOX_TAKE_BACK(tag->parserFieldsDynamic); + } + else + attachParserFieldGeneric (tag, ftype, value, false); +} + +extern void attachParserFieldToCorkEntry (int index, + fieldType ftype, + const char *value) +{ + tagEntryInfo * tag = getEntryInCorkQueue (index); + if (tag) + attachParserField (tag, true, ftype, value); +} + +extern const tagField* getParserFieldForIndex (const tagEntryInfo * tag, int index) +{ + if (index < 0 + || tag->usedParserFields <= ((unsigned int)index) ) + return NULL; + else if (index < PRE_ALLOCATED_PARSER_FIELDS) + return tag->parserFields + index; + else + { + unsigned int n = index - PRE_ALLOCATED_PARSER_FIELDS; + return ptrArrayItem(tag->parserFieldsDynamic, n); + } +} + +extern const char* getParserFieldValueForType (tagEntryInfo *const tag, fieldType ftype) +{ + for (int i = 0; i < tag->usedParserFields; i++) + { + const tagField *f = getParserFieldForIndex (tag, i); + if (f && f->ftype == ftype) + return f->value; + } + return NULL; } static void copyParserFields (const tagEntryInfo *const tag, tagEntryInfo* slot) @@ -924,24 +1025,42 @@ static void copyParserFields (const tagEntryInfo *const tag, tagEntryInfo* slot) for (i = 0; i < tag->usedParserFields; i++) { - value = tag->parserFields [i].value; + const tagField *f = getParserFieldForIndex (tag, i); + Assert(f); + + value = f->value; if (value) value = eStrdup (value); - attachParserField (slot, - tag->parserFields [i].ftype, - value); + attachParserFieldGeneric (slot, + f->ftype, + value, + true); } + } -static void recordTagEntryInQueue (const tagEntryInfo *const tag, tagEntryInfo* slot) +static tagEntryInfo *newNilTagEntry (unsigned int corkFlags) { + tagEntryInfoX *x = xCalloc (1, tagEntryInfoX); + x->corkIndex = CORK_NIL; + x->symtab = RB_ROOT; + x->slot.kindIndex = KIND_FILE_INDEX; + return &(x->slot); +} + +static tagEntryInfoX *copyTagEntry (const tagEntryInfo *const tag, + unsigned int corkFlags) +{ + tagEntryInfoX *x = xMalloc (1, tagEntryInfoX); + x->symtab = RB_ROOT; + x->corkIndex = CORK_NIL; + tagEntryInfo *slot = (tagEntryInfo *)x; + *slot = *tag; if (slot->pattern) slot->pattern = eStrdup (slot->pattern); - else if (!slot->lineNumberEntry) - slot->pattern = makePatternString (slot); slot->inputFileName = eStrdup (slot->inputFileName); slot->name = eStrdup (slot->name); @@ -961,39 +1080,63 @@ static void recordTagEntryInQueue (const tagEntryInfo *const tag, tagEntryInfo* slot->extensionFields.typeRef[0] = eStrdup (slot->extensionFields.typeRef[0]); if (slot->extensionFields.typeRef[1]) slot->extensionFields.typeRef[1] = eStrdup (slot->extensionFields.typeRef[1]); -/* GEANY DIFF */ - if (slot->extensionFields.varType) - slot->extensionFields.varType = eStrdup (slot->extensionFields.varType); -/* GEANY DIFF END */ #ifdef HAVE_LIBXML if (slot->extensionFields.xpath) slot->extensionFields.xpath = eStrdup (slot->extensionFields.xpath); #endif + if (slot->extraDynamic) + { + int n = countXtags () - XTAG_COUNT; + slot->extraDynamic = xCalloc ((n / 8) + 1, uint8_t); + memcpy (slot->extraDynamic, tag->extraDynamic, (n / 8) + 1); + } + if (slot->sourceFileName) slot->sourceFileName = eStrdup (slot->sourceFileName); + slot->usedParserFields = 0; + slot->parserFieldsDynamic = NULL; copyParserFields (tag, slot); + if (slot->parserFieldsDynamic) + PARSER_TRASH_BOX_TAKE_BACK(slot->parserFieldsDynamic); + + return x; } static void clearParserFields (tagEntryInfo *const tag) { - unsigned int i; + unsigned int i, n; const char* value; - for (i = 0; i < tag->usedParserFields; i++) + if ( tag->usedParserFields < PRE_ALLOCATED_PARSER_FIELDS ) + n = tag->usedParserFields; + else + n = PRE_ALLOCATED_PARSER_FIELDS; + + for (i = 0; i < n; i++) { value = tag->parserFields[i].value; - if (value) + if (value && tag->parserFields[i].valueOwner) eFree ((char *)value); tag->parserFields[i].value = NULL; tag->parserFields[i].ftype = FIELD_UNKNOWN; } + if (tag->parserFieldsDynamic) + { + ptrArrayDelete (tag->parserFieldsDynamic); + tag->parserFieldsDynamic = NULL; + } } -static void clearTagEntryInQueue (tagEntryInfo* slot) +static void deleteTagEnry (void *data) { + tagEntryInfo *slot = data; + + if (slot->kindIndex == KIND_FILE_INDEX) + goto out; + if (slot->pattern) eFree ((char *)slot->pattern); eFree ((char *)slot->inputFileName); @@ -1015,139 +1158,409 @@ static void clearTagEntryInQueue (tagEntryInfo* slot) eFree ((char *)slot->extensionFields.typeRef[0]); if (slot->extensionFields.typeRef[1]) eFree ((char *)slot->extensionFields.typeRef[1]); -/* GEANY DIFF */ - if (slot->extensionFields.varType) - eFree ((char *)slot->extensionFields.varType); -/* GEANY DIFF END */ #ifdef HAVE_LIBXML if (slot->extensionFields.xpath) eFree ((char *)slot->extensionFields.xpath); #endif + if (slot->extraDynamic) + eFree (slot->extraDynamic); + if (slot->sourceFileName) eFree ((char *)slot->sourceFileName); clearParserFields (slot); + + out: + eFree (slot); } -static unsigned int queueTagEntry(const tagEntryInfo *const tag) +static void corkSymtabPut (tagEntryInfoX *scope, const char* name, tagEntryInfoX *item) { - unsigned int i; - void *tmp; - tagEntryInfo * slot; + struct rb_root *root = &scope->symtab; + + struct rb_node **new = &(root->rb_node), *parent = NULL; - if (! (TagFile.corkQueue.count < TagFile.corkQueue.length)) + while (*new) { - if (!TagFile.corkQueue.length) - TagFile.corkQueue.length = 1; + tagEntryInfoX *this = container_of(*new, tagEntryInfoX, symnode); + int result = strcmp(item->slot.name, this->slot.name); + + parent = *new; - tmp = eRealloc (TagFile.corkQueue.queue, - sizeof (*TagFile.corkQueue.queue) * TagFile.corkQueue.length * 2); + if (result < 0) + new = &((*new)->rb_left); + else if (result > 0) + new = &((*new)->rb_right); + else + { + unsigned long lthis = this->slot.lineNumber; + unsigned long litem = item->slot.lineNumber; + + /* Comparing lineNumber */ + if (litem < lthis) + new = &((*new)->rb_left); + else if (litem > lthis) + new = &((*new)->rb_right); + else + { + /* Comparing memory address */ + if (item < this) + new = &((*new)->rb_left); + else if (item > this) + new = &((*new)->rb_right); + else + { + AssertNotReached(); /* registering the same object twice. */ + return; + } + } + } + } - TagFile.corkQueue.length *= 2; - TagFile.corkQueue.queue = tmp; + verbose ("symtbl[:=] %s<-%s/%p (line: %lu)\n", + *new? container_of(*new, tagEntryInfoX, symnode)->slot.name: "*root*", + item->slot.name, &item->slot, item->slot.lineNumber); + /* Add new node and rebalance tree. */ + rb_link_node(&item->symnode, parent, new); + rb_insert_color(&item->symnode, root); +} + +extern bool foreachEntriesInScope (int corkIndex, + const char *name, + entryForeachFunc func, + void *data) +{ + tagEntryInfoX *x = ptrArrayItem (TagFile.corkQueue, corkIndex); + + struct rb_root *root = &x->symtab; + tagEntryInfoX *rep = NULL; + + /* More than one tag can have a same name. + * Visit them from the last. + * + * 1. find one of them as the representative, + * 2. find the last one of them from the representative with rb_next, + * 3. call FUNC iteratively from the last to the first. + */ + if (name) + { + struct rb_node *node = root->rb_node; + while (node) + { + tagEntryInfoX *entry = container_of(node, tagEntryInfoX, symnode); + int result; + + result = strcmp(name, entry->slot.name); + + if (result < 0) + node = node->rb_left; + else if (result > 0) + node = node->rb_right; + else + { + rep = entry; + break; + } + } + if (rep == NULL) + return true; + + verbose("symtbl[<>] %s->%p\n", name, &rep->slot); + } + + struct rb_node *last; + + if (name) + { + struct rb_node *tmp = &rep->symnode; + last = tmp; + + while ((tmp = rb_next (tmp))) + { + tagEntryInfoX *entry = container_of(tmp, tagEntryInfoX, symnode); + if (strcmp(name, entry->slot.name) == 0) + { + verbose ("symtbl[ >] %s->%p\n", name, &container_of(tmp, tagEntryInfoX, symnode)->slot); + last = tmp; + } + else + break; + } + } + else + { + last = rb_last(root); + verbose ("last for %d<%p>: %p\n", corkIndex, root, last); } - i = TagFile.corkQueue.count; - TagFile.corkQueue.count++; + if (!last) + { + verbose ("symtbl[>V] %s->%p\n", name? name: "(null)", NULL); + return true; /* Nothing here in this node. */ + } + verbose ("symtbl[>|] %s->%p\n", name, &container_of(last, tagEntryInfoX, symnode)->slot); - slot = TagFile.corkQueue.queue + i; - recordTagEntryInQueue (tag, slot); + struct rb_node *cursor = last; + bool revisited_rep = false; + do + { + tagEntryInfoX *entry = container_of(cursor, tagEntryInfoX, symnode); + if (!revisited_rep || !name || strcmp(name, entry->slot.name)) + { + verbose ("symtbl[< ] %s->%p\n", name, &entry->slot); + if (!func (entry->corkIndex, &entry->slot, data)) + return false; + if (cursor == &rep->symnode) + revisited_rep = true; + } + else if (name) + break; + } + while ((cursor = rb_prev(cursor))); - return i; + return true; } +static bool findName (int corkIndex, tagEntryInfo *entry, void *data) +{ + int *index = data; -static void *writerData; -static tagWriter *writer; + *index = corkIndex; + return false; +} -extern void setTagWriter (tagWriter *t) +int anyEntryInScope (int corkIndex, const char *name) { - writer = t; + int index = CORK_NIL; + + if (foreachEntriesInScope (corkIndex, name, findName, &index) == false) + return index; + + return CORK_NIL; } -extern bool outpuFormatUsedStdoutByDefault (void) +struct anyKindsEntryInScopeData { + int index; + const int *kinds; + int count; +}; + +static bool findNameOfKinds (int corkIndex, tagEntryInfo *entry, void *data) { - return writer->useStdoutByDefault; + struct anyKindsEntryInScopeData * kdata = data; + + for (int i = 0; i < kdata->count; i++) + { + int k = kdata->kinds [i]; + if (entry->kindIndex == k) + { + kdata->index = corkIndex; + return false; + } + } + return true; } -extern void setupWriter (void) +int anyKindEntryInScope (int corkIndex, + const char *name, int kind) { - if (writer->preWriteEntry) - writerData = writer->preWriteEntry (TagFile.mio); - else - writerData = NULL; + return anyKindsEntryInScope (corkIndex, name, &kind, 1); } -extern void teardownWriter (const char *filename) +int anyKindsEntryInScope (int corkIndex, + const char *name, + const int *kinds, int count) { - if (writer->postWriteEntry) - writer->postWriteEntry (TagFile.mio, filename, writerData); + struct anyKindsEntryInScopeData data = { + .index = CORK_NIL, + .kinds = kinds, + .count = count, + }; + + if (foreachEntriesInScope (corkIndex, name, findNameOfKinds, &data) == false) + return data.index; + + return CORK_NIL; } -static void buildFqTagCache (const tagEntryInfo *const tag) +int anyKindsEntryInScopeRecursive (int corkIndex, + const char *name, + const int *kinds, int count) { - renderFieldEscaped (FIELD_SCOPE_KIND_LONG, tag, NO_PARSER_FIELD); - renderFieldEscaped (FIELD_SCOPE, tag, NO_PARSER_FIELD); + struct anyKindsEntryInScopeData data = { + .index = CORK_NIL, + .kinds = kinds, + .count = count, + }; + + tagEntryInfo *e; + do + { + if (foreachEntriesInScope (corkIndex, name, findNameOfKinds, &data) == false) + return data.index; + + if (corkIndex == CORK_NIL) + break; + + e = getEntryInCorkQueue (corkIndex); + if (!e) + break; + corkIndex = e->extensionFields.scopeIndex; + } + while (1); + + return CORK_NIL; } -#ifdef CTAGS_LIB -static void initCtagsTag(ctagsTag *tag, const tagEntryInfo *info) +extern void registerEntry (int corkIndex) { - tag->name = info->name; - tag->signature = info->extensionFields.signature; - tag->scopeName = info->extensionFields.scopeName; - tag->inheritance = info->extensionFields.inheritance; - tag->varType = info->extensionFields.varType; - tag->access = info->extensionFields.access; - tag->implementation = info->extensionFields.implementation; - tag->kindLetter = getLanguageKind(info->langType, info->kindIndex)->letter; - tag->isFileScope = info->isFileScope; - tag->lineNumber = info->lineNumber; - tag->lang = info->langType; + Assert (TagFile.corkFlags & CORK_SYMTAB); + Assert (corkIndex != CORK_NIL); + + tagEntryInfoX *e = ptrArrayItem (TagFile.corkQueue, corkIndex); + { + tagEntryInfoX *scope = ptrArrayItem (TagFile.corkQueue, e->slot.extensionFields.scopeIndex); + corkSymtabPut (scope, e->slot.name, e); + } +} + +static int queueTagEntry(const tagEntryInfo *const tag) +{ + static bool warned; + + int corkIndex; + tagEntryInfoX * entry = copyTagEntry (tag, + TagFile.corkFlags); + + if (ptrArrayCount (TagFile.corkQueue) == (size_t)INT_MAX) + { + if (!warned) + { + warned = true; + error (WARNING, + "The tag entry queue overflows; drop the tag entry at %lu in %s", + tag->lineNumber, + tag->inputFileName); + } + return CORK_NIL; + } + warned = false; + + corkIndex = (int)ptrArrayAdd (TagFile.corkQueue, entry); + entry->corkIndex = corkIndex; + + return corkIndex; +} + +extern void setupWriter (void *writerClientData) +{ + writerSetup (TagFile.mio, writerClientData); +} + +extern bool teardownWriter (const char *filename) +{ + return writerTeardown (TagFile.mio, filename); +} + +static bool isTagWritable(const tagEntryInfo *const tag) +{ + if (tag->placeholder) + return false; + + if (! isLanguageKindEnabled(tag->langType, tag->kindIndex)) + return false; + + if (tag->extensionFields.roleBits) + { + size_t available_roles; + + if (!isXtagEnabled (XTAG_REFERENCE_TAGS)) + return false; + + available_roles = countLanguageRoles(tag->langType, + tag->kindIndex); + if (tag->extensionFields.roleBits >= + (makeRoleBit(available_roles))) + return false; + + /* TODO: optimization + A Bitmasks representing all enabled roles can be calculated at the + end of initializing the parser. Calculating each time when checking + a tag entry is not needed. */ + for (unsigned int roleIndex = 0; roleIndex < available_roles; roleIndex++) + { + if (isRoleAssigned(tag, roleIndex)) + { + if (isLanguageRoleEnabled (tag->langType, tag->kindIndex, + roleIndex)) + return true; + } + + } + return false; + } + else if (isLanguageKindRefOnly(tag->langType, tag->kindIndex)) + { + error (WARNING, "definition tag for refonly kind(%s) is made: %s", + getLanguageKind(tag->langType, tag->kindIndex)->name, + tag->name); + /* This one is not so critical. */ + } + + if (!isXtagEnabled(XTAG_ANONYMOUS) + && isTagExtraBitMarked(tag, XTAG_ANONYMOUS)) + return false; + + return true; +} + +static void buildFqTagCache (tagEntryInfo *const tag) +{ + getTagScopeInformation (tag, NULL, NULL); } -#endif static void writeTagEntry (const tagEntryInfo *const tag) { int length = 0; - if (tag->placeholder) - return; -#ifndef CTAGS_LIB - if (! tag->kind->enabled) - return; -#endif - if (tag->extensionFields.roleIndex != ROLE_INDEX_DEFINITION - && ! isXtagEnabled (XTAG_REFERENCE_TAGS)) - return; + Assert (tag->kindIndex != KIND_GHOST_INDEX); DebugStatement ( debugEntry (tag); ) - Assert (writer); + +#ifdef WIN32 + if (getFilenameSeparator(Option.useSlashAsFilenameSeparator) == FILENAME_SEP_USE_SLASH) + { + Assert (((const tagEntryInfo *)tag)->inputFileName); + char *c = (char *)(((tagEntryInfo *const)tag)->inputFileName); + while (*c) + { + if (*c == PATH_SEPARATOR) + *c = OUTPUT_PATH_SEPARATOR; + c++; + } + } +#endif if (includeExtensionFlags () && isXtagEnabled (XTAG_QUALIFIED_TAGS) - && doesInputLanguageRequestAutomaticFQTag ()) - buildFqTagCache (tag); + && doesInputLanguageRequestAutomaticFQTag () + && !isTagExtraBitMarked (tag, XTAG_QUALIFIED_TAGS) + && !tag->skipAutoFQEmission) + { + /* const is discarded to update the cache field of TAG. */ + buildFqTagCache ( (tagEntryInfo *const)tag); + } -#ifdef CTAGS_LIB - getTagScopeInformation((tagEntryInfo *)tag, NULL, NULL); + length = writerWriteTag (TagFile.mio, tag); - if (TagEntryFunction != NULL) + if (length > 0) { - ctagsTag t; - - initCtagsTag(&t, tag); - length = TagEntryFunction(&t, TagEntryUserData); + ++TagFile.numTags.added; + rememberMaxLengths (strlen (tag->name), (size_t) length); } -#else - length = writer->writeEntry (TagFile.mio, tag, writerData); -#endif - - ++TagFile.numTags.added; - rememberMaxLengths (strlen (tag->name), (size_t) length); - DebugStatement ( mio_flush (TagFile.mio); ) + DebugStatement ( if (TagFile.mio) mio_flush (TagFile.mio); ) abort_if_ferror (TagFile.mio); } @@ -1159,11 +1572,11 @@ extern bool writePseudoTag (const ptagDesc *desc, { int length; - if (writer->writePtagEntry == NULL) + length = writerWritePtag (TagFile.mio, desc, fileName, + pattern, parserName); + if (length < 0) return false; - length = writer->writePtagEntry (TagFile.mio, desc, fileName, - pattern, parserName, writerData); abort_if_ferror (TagFile.mio); ++TagFile.numTags.added; @@ -1172,15 +1585,15 @@ extern bool writePseudoTag (const ptagDesc *desc, return true; } -extern void corkTagFile(void) +extern void corkTagFile(unsigned int corkFlags) { TagFile.cork++; if (TagFile.cork == 1) { - TagFile.corkQueue.length = 1; - TagFile.corkQueue.count = 1; - TagFile.corkQueue.queue = eMalloc (sizeof (*TagFile.corkQueue.queue)); - memset (TagFile.corkQueue.queue, 0, sizeof (*TagFile.corkQueue.queue)); + TagFile.corkFlags = corkFlags; + TagFile.corkQueue = ptrArrayNew (deleteTagEnry); + tagEntryInfo *nil = newNilTagEntry (corkFlags); + ptrArrayAdd (TagFile.corkQueue, nil); } } @@ -1193,32 +1606,36 @@ extern void uncorkTagFile(void) if (TagFile.cork > 0) return ; - for (i = 1; i < TagFile.corkQueue.count; i++) + for (i = 1; i < ptrArrayCount (TagFile.corkQueue); i++) { - tagEntryInfo *tag = TagFile.corkQueue.queue + i; + tagEntryInfo *tag = ptrArrayItem (TagFile.corkQueue, i); + + if (!isTagWritable(tag)) + continue; + writeTagEntry (tag); + if (doesInputLanguageRequestAutomaticFQTag () && isXtagEnabled (XTAG_QUALIFIED_TAGS) - && (tag->extensionFields.scopeKindIndex != KIND_GHOST_INDEX) - && tag->extensionFields.scopeName - && tag->extensionFields.scopeIndex) + && !isTagExtraBitMarked (tag, XTAG_QUALIFIED_TAGS) + && !tag->skipAutoFQEmission + && ((tag->extensionFields.scopeKindIndex != KIND_GHOST_INDEX + && tag->extensionFields.scopeName != NULL + && tag->extensionFields.scopeIndex != CORK_NIL) + || (tag->extensionFields.scopeKindIndex == KIND_GHOST_INDEX + && tag->extensionFields.scopeName == NULL + && tag->extensionFields.scopeIndex == CORK_NIL))) makeQualifiedTagEntry (tag); } - for (i = 1; i < TagFile.corkQueue.count; i++) - clearTagEntryInQueue (TagFile.corkQueue.queue + i); - memset (TagFile.corkQueue.queue, 0, - sizeof (*TagFile.corkQueue.queue) * TagFile.corkQueue.count); - TagFile.corkQueue.count = 0; - eFree (TagFile.corkQueue.queue); - TagFile.corkQueue.queue = NULL; - TagFile.corkQueue.length = 0; + ptrArrayDelete (TagFile.corkQueue); + TagFile.corkQueue = NULL; } -extern tagEntryInfo *getEntryInCorkQueue (unsigned int n) +extern tagEntryInfo *getEntryInCorkQueue (int n) { - if ((CORK_NIL < n) && (n < TagFile.corkQueue.count)) - return TagFile.corkQueue.queue + n; + if ((CORK_NIL < n) && (((size_t)n) < ptrArrayCount (TagFile.corkQueue))) + return ptrArrayItem (TagFile.corkQueue, n); else return NULL; } @@ -1232,25 +1649,38 @@ extern tagEntryInfo *getEntryOfNestingLevel (const NestingLevel *nl) extern size_t countEntryInCorkQueue (void) { - return TagFile.corkQueue.count; + return ptrArrayCount (TagFile.corkQueue); +} + +extern int makePlaceholder (const char *const name) +{ + tagEntryInfo e; + + initTagEntry (&e, name, KIND_GHOST_INDEX); + e.placeholder = 1; + + /* + * makePlaceholder may be called even before reading any bytes + * from the input stream. In such case, initTagEntry fills + * the lineNumber field of the placeholder tag with 0. + * This breaks an assertion in makeTagEntry. Following adjustment + * is for avoiding it. + */ + if (e.lineNumber == 0) + e.lineNumber = 1; + + return makeTagEntry (&e); } extern int makeTagEntry (const tagEntryInfo *const tag) { int r = CORK_NIL; Assert (tag->name != NULL); + Assert(tag->lineNumber > 0); -#ifndef CTAGS_LIB - if (getInputLanguageFileKind() != tag->kind) - { - if (! isInputLanguageKindEnabled (tag->kind->letter) && - (tag->extensionFields.roleIndex == ROLE_INDEX_DEFINITION)) - return CORK_NIL; - if ((tag->extensionFields.roleIndex != ROLE_INDEX_DEFINITION) - && (! tag->kind->roles[tag->extensionFields.roleIndex].enabled)) - return CORK_NIL; - } -#endif + if (!TagFile.cork) + if (!isTagWritable (tag)) + goto out; if (tag->name [0] == '\0' && (!tag->placeholder)) { @@ -1264,6 +1694,10 @@ extern int makeTagEntry (const tagEntryInfo *const tag) r = queueTagEntry (tag); else writeTagEntry (tag); + + if (r != CORK_NIL) + notifyMakeTagEntry (tag, r); + out: return r; } @@ -1272,7 +1706,7 @@ extern int makeQualifiedTagEntry (const tagEntryInfo *const e) { int r = CORK_NIL; tagEntryInfo x; - char xk; + int xk; const char *sep; static vString *fqn; @@ -1281,7 +1715,7 @@ extern int makeQualifiedTagEntry (const tagEntryInfo *const e) x = *e; markTagExtraBit (&x, XTAG_QUALIFIED_TAGS); - fqn = vStringNewOrClear (fqn); + fqn = vStringNewOrClearWithAutoRelease (fqn); if (e->extensionFields.scopeName) { @@ -1298,7 +1732,7 @@ extern int makeQualifiedTagEntry (const tagEntryInfo *const e) if (sep == NULL) { /* No root separator. The name of the - oritinal tag and that of full qualified tag + optional tag and that of full qualified tag are the same; recording the full qualified tag is meaningless. */ return r; @@ -1310,53 +1744,35 @@ extern int makeQualifiedTagEntry (const tagEntryInfo *const e) x.name = vStringValue (fqn); /* makeExtraTagEntry of c.c doesn't clear scope - releated fields. */ + related fields. */ #if 0 x.extensionFields.scopeKind = NULL; x.extensionFields.scopeName = NULL; + x.extensionFields.scopeIndex = CORK_NIL; #endif + + bool in_subparser + = isTagExtraBitMarked (&x, + XTAG_SUBPARSER); + + if (in_subparser) + pushLanguage(x.langType); + r = makeTagEntry (&x); + + if (in_subparser) + popLanguage(); } return r; } -extern void initTagEntry (tagEntryInfo *const e, const char *const name, - int kindIndex) -{ - initTagEntryFull(e, name, - getInputLineNumber (), - getInputLanguage (), - getInputFilePosition (), - getInputFileTagPath (), - kindIndex, - ROLE_INDEX_DEFINITION, - getSourceFileTagPath(), - getSourceLanguage(), - getSourceLineNumber() - getInputLineNumber ()); -} - -extern void initRefTagEntry (tagEntryInfo *const e, const char *const name, - int kindIndex, int roleIndex) -{ - initTagEntryFull(e, name, - getInputLineNumber (), - getInputLanguage (), - getInputFilePosition (), - getInputFileTagPath (), - kindIndex, - roleIndex, - getSourceFileTagPath(), - getSourceLanguage(), - getSourceLineNumber() - getInputLineNumber ()); -} - -extern void initTagEntryFull (tagEntryInfo *const e, const char *const name, +static void initTagEntryFull (tagEntryInfo *const e, const char *const name, unsigned long lineNumber, langType langType_, MIOPos filePosition, const char *inputFileName, int kindIndex, - int roleIndex, + roleBitsType roleBits, const char *sourceFileName, langType sourceLangType, long sourceLineNumberDifference) @@ -1380,12 +1796,17 @@ extern void initTagEntryFull (tagEntryInfo *const e, const char *const name, Assert (kindIndex < 0 || kindIndex < (int)countLanguageKinds(langType_)); e->kindIndex = kindIndex; - Assert (roleIndex >= ROLE_INDEX_DEFINITION); - Assert (kind == NULL || roleIndex < kind->nRoles); - e->extensionFields.roleIndex = roleIndex; - if (roleIndex > ROLE_INDEX_DEFINITION) + Assert (roleBits == 0 + || (roleBits < (makeRoleBit(countLanguageRoles(langType_, kindIndex))))); + e->extensionFields.roleBits = roleBits; + if (roleBits) markTagExtraBit (e, XTAG_REFERENCE_TAGS); + if (doesParserRunAsGuest ()) + markTagExtraBit (e, XTAG_GUEST); + if (doesSubparserRun ()) + markTagExtraBit (e, XTAG_SUBPARSER); + e->sourceLangType = sourceLangType; e->sourceFileName = sourceFileName; e->sourceLineNumberDifference = sourceLineNumberDifference; @@ -1394,30 +1815,163 @@ extern void initTagEntryFull (tagEntryInfo *const e, const char *const name, for ( i = 0; i < PRE_ALLOCATED_PARSER_FIELDS; i++ ) e->parserFields[i].ftype = FIELD_UNKNOWN; + + if (isParserMarkedNoEmission ()) + e->placeholder = 1; } -extern void markTagExtraBit (tagEntryInfo *const tag, xtagType extra) +extern void initTagEntry (tagEntryInfo *const e, const char *const name, + int kindIndex) +{ + initTagEntryFull(e, name, + getInputLineNumber (), + getInputLanguage (), + getInputFilePosition (), + getInputFileTagPath (), + kindIndex, + 0, + getSourceFileTagPath(), + getSourceLanguage(), + getSourceLineNumber() - getInputLineNumber ()); +} + +extern void initRefTagEntry (tagEntryInfo *const e, const char *const name, + int kindIndex, int roleIndex) +{ + initForeignRefTagEntry (e, name, getInputLanguage (), kindIndex, roleIndex); +} + +extern void initForeignRefTagEntry (tagEntryInfo *const e, const char *const name, + langType langType, + int kindIndex, int roleIndex) +{ + initTagEntryFull(e, name, + getInputLineNumber (), + langType, + getInputFilePosition (), + getInputFileTagPath (), + kindIndex, + makeRoleBit(roleIndex), + getSourceFileTagPath(), + getSourceLanguage(), + getSourceLineNumber() - getInputLineNumber ()); +} + +static void markTagExtraBitFull (tagEntryInfo *const tag, xtagType extra, bool mark) { unsigned int index; unsigned int offset; + uint8_t *slot; - Assert (extra < XTAG_COUNT); Assert (extra != XTAG_UNKNOWN); - index = (extra / 8); - offset = (extra % 8); - tag->extra [ index ] |= (1 << offset); + if (extra < XTAG_COUNT) + { + index = (extra / 8); + offset = (extra % 8); + slot = tag->extra; + } + else if (tag->extraDynamic) + { + Assert (extra < countXtags ()); + + index = ((extra - XTAG_COUNT) / 8); + offset = ((extra - XTAG_COUNT) % 8); + slot = tag->extraDynamic; + } + else + { + Assert (extra < countXtags ()); + + int n = countXtags () - XTAG_COUNT; + tag->extraDynamic = xCalloc ((n / 8) + 1, uint8_t); + PARSER_TRASH_BOX(tag->extraDynamic, eFree); + markTagExtraBit (tag, extra); + return; + } + + if (mark) + slot [ index ] |= (1 << offset); + else + slot [ index ] &= ~(1 << offset); +} + +extern void markTagExtraBit (tagEntryInfo *const tag, xtagType extra) +{ + markTagExtraBitFull (tag, extra, true); } extern bool isTagExtraBitMarked (const tagEntryInfo *const tag, xtagType extra) { - unsigned int index = (extra / 8); - unsigned int offset = (extra % 8); + unsigned int index; + unsigned int offset; + const uint8_t *slot; - Assert (extra < XTAG_COUNT); Assert (extra != XTAG_UNKNOWN); - return !! ((tag->extra [ index ]) & (1 << offset)); + if (extra < XTAG_COUNT) + { + index = (extra / 8); + offset = (extra % 8); + slot = tag->extra; + + } + else if (!tag->extraDynamic) + return false; + else + { + Assert (extra < countXtags ()); + index = ((extra - XTAG_COUNT) / 8); + offset = ((extra - XTAG_COUNT) % 8); + slot = tag->extraDynamic; + + } + return !! ((slot [ index ]) & (1 << offset)); +} + +extern bool isTagExtra (const tagEntryInfo *const tag) +{ + for (unsigned int i = 0; i < XTAG_COUNT; i++) + if (isTagExtraBitMarked (tag, i)) + return true; + return false; +} + +static void assignRoleFull(tagEntryInfo *const e, int roleIndex, bool assign) +{ + if (roleIndex == ROLE_DEFINITION_INDEX) + { + if (assign) + { + e->extensionFields.roleBits = 0; + markTagExtraBitFull (e, XTAG_REFERENCE_TAGS, false); + } + } + else if (roleIndex > ROLE_DEFINITION_INDEX) + { + Assert (roleIndex < (int)countLanguageRoles(e->langType, e->kindIndex)); + + if (assign) + e->extensionFields.roleBits |= (makeRoleBit(roleIndex)); + else + e->extensionFields.roleBits &= ~(makeRoleBit(roleIndex)); + markTagExtraBitFull (e, XTAG_REFERENCE_TAGS, e->extensionFields.roleBits); + } + else + AssertNotReached(); +} + +extern void assignRole(tagEntryInfo *const e, int roleIndex) +{ + assignRoleFull(e, roleIndex, true); +} + +extern bool isRoleAssigned(const tagEntryInfo *const e, int roleIndex) +{ + if (roleIndex == ROLE_DEFINITION_INDEX) + return (!e->extensionFields.roleBits); + else + return (e->extensionFields.roleBits & makeRoleBit(roleIndex)); } extern unsigned long numTagsAdded(void) @@ -1447,23 +2001,39 @@ extern void invalidatePatternCache(void) extern void tagFilePosition (MIOPos *p) { - mio_getpos (TagFile.mio, p); + /* mini-geany doesn't set TagFile.mio. */ + if (TagFile.mio == NULL) + return; + + if (mio_getpos (TagFile.mio, p) == -1) + error (FATAL|PERROR, + "failed to get file position of the tag file\n"); } extern void setTagFilePosition (MIOPos *p) { - mio_setpos (TagFile.mio, p); + /* mini-geany doesn't set TagFile.mio. */ + if (TagFile.mio == NULL) + return; + + if (mio_setpos (TagFile.mio, p) == -1) + error (FATAL|PERROR, + "failed to set file position of the tag file\n"); } extern const char* getTagFileDirectory (void) { return TagFile.directory; } -#ifdef CTAGS_LIB -extern void setTagEntryFunction(tagEntryFunction entry_function, void *user_data) +static bool markAsPlaceholder (int index, tagEntryInfo *e, void *data CTAGS_ATTR_UNUSED) { - TagEntryFunction = entry_function; - TagEntryUserData = user_data; + e->placeholder = 1; + markAllEntriesInScopeAsPlaceholder (index); + return true; +} + +extern void markAllEntriesInScopeAsPlaceholder (int index) +{ + foreachEntriesInScope (index, NULL, markAsPlaceholder, NULL); } -#endif Modified: ctags/main/entry.h 207 lines changed, 146 insertions(+), 61 deletions(-) =================================================================== @@ -16,30 +16,28 @@ #include "types.h" #include <stdint.h> -#include <stdio.h> #include "field.h" -#include "kind.h" -#include "vstring.h" #include "xtag.h" #include "mio.h" +#include "ptrarray.h" #include "nestlevel.h" -#include "ctags-api.h" /* * MACROS */ -#define WHOLE_FILE -1L -#define includeExtensionFlags() (Option.tagFileFormat > 1) /* * DATA DECLARATIONS */ typedef struct sTagField { fieldType ftype; const char* value; + bool valueOwner; /* used only in parserFieldsDynamic */ } tagField; +typedef uint64_t roleBitsType; + /* Information about the current tag candidate. */ struct sTagEntryInfo { @@ -50,6 +48,10 @@ struct sTagEntryInfo { unsigned int placeholder :1; /* This is just a part of scope context. Put this entry to cork queue but don't print it to tags file. */ + unsigned int skipAutoFQEmission:1; /* If a parser makes a fq tag for the + current tag by itself, set this. */ + unsigned int isPseudoTag:1; /* Used only in xref output. + If a tag is a pseudo, set this. */ unsigned long lineNumber; /* line number of tag */ const char* pattern; /* pattern for locating input line @@ -60,7 +62,8 @@ struct sTagEntryInfo { const char *inputFileName; /* name of input file */ const char *name; /* name of the tag */ int kindIndex; /* kind descriptor */ - unsigned char extra[ ((XTAG_COUNT) / 8) + 1 ]; + uint8_t extra[ ((XTAG_COUNT) / 8) + 1 ]; + uint8_t *extraDynamic; /* Dynamically allocated but freed by per parser TrashBox */ struct { const char* access; @@ -84,23 +87,27 @@ struct sTagEntryInfo { /* type (union/struct/etc.) and name for a variable or typedef. */ const char* typeRef [2]; /* e.g., "struct" and struct name */ -/* GEANY DIFF */ - const char *varType; -/* GEANY DIFF END */ - -#define ROLE_INDEX_DEFINITION -1 - int roleIndex; /* for role of reference tag */ +#define ROLE_DEFINITION_INDEX -1 +#define ROLE_DEFINITION_NAME "def" +#define ROLE_MAX_COUNT (sizeof(roleBitsType) * 8) + roleBitsType roleBits; /* for role of reference tag */ #ifdef HAVE_LIBXML const char* xpath; #endif unsigned long endLine; } extensionFields; /* list of extension fields*/ + /* `usedParserFields' tracks how many parser own fields are + used. If it is a few (less than PRE_ALLOCATED_PARSER_FIELDS), + statically allocated parserFields is used. If more fields than + PRE_ALLOCATED_PARSER_FIELDS is defined and attached, parserFieldsDynamic + is used. */ + unsigned int usedParserFields; #define PRE_ALLOCATED_PARSER_FIELDS 5 #define NO_PARSER_FIELD -1 - unsigned int usedParserFields; tagField parserFields [PRE_ALLOCATED_PARSER_FIELDS]; + ptrArray * parserFieldsDynamic; /* Following source* fields are used only when #line is found in input and --line-directive is given in ctags command line. */ @@ -109,6 +116,9 @@ struct sTagEntryInfo { unsigned long sourceLineNumberDifference; }; +typedef bool (* entryForeachFunc) (int corkIndex, + tagEntryInfo * entry, + void * data); /* * GLOBAL VARIABLES @@ -118,71 +128,146 @@ struct sTagEntryInfo { /* * FUNCTION PROTOTYPES */ -extern void freeTagFileResources (void); -extern const char *tagFileName (void); -extern void openTagFile (void); -extern void closeTagFile (const bool resize); -extern void setupWriter (void); -extern void teardownWriter (const char *inputFilename); extern int makeTagEntry (const tagEntryInfo *const tag); extern void initTagEntry (tagEntryInfo *const e, const char *const name, int kindIndex); extern void initRefTagEntry (tagEntryInfo *const e, const char *const name, int kindIndex, int roleIndex); -extern void initTagEntryFull (tagEntryInfo *const e, const char *const name, - unsigned long lineNumber, - langType langType_, - MIOPos filePosition, - const char *inputFileName, - int kindIndex, - int roleIndex, - const char *sourceFileName, - langType sourceLangType, - long sourceLineNumberDifference); +extern void initForeignRefTagEntry (tagEntryInfo *const e, const char *const name, + langType type, + int kindIndex, int roleIndex); +extern void assignRole(tagEntryInfo *const e, int roleIndex); +extern bool isRoleAssigned(const tagEntryInfo *const e, int roleIndex); + extern int makeQualifiedTagEntry (const tagEntryInfo *const e); -extern unsigned long numTagsAdded(void); -extern void setNumTagsAdded (unsigned long nadded); -extern unsigned long numTagsTotal(void); -extern unsigned long maxTagsLine(void); -extern void invalidatePatternCache(void); -extern void tagFilePosition (MIOPos *p); -extern void setTagFilePosition (MIOPos *p); -extern const char* getTagFileDirectory (void); -extern void getTagScopeInformation (tagEntryInfo *const tag, - const char **kind, const char **name); -/* Getting line associated with tag */ -extern char *readLineFromBypassAnyway (vString *const vLine, const tagEntryInfo *const tag, - long *const pSeekValue); +#define CORK_NIL 0 +tagEntryInfo *getEntryInCorkQueue (int n); +tagEntryInfo *getEntryOfNestingLevel (const NestingLevel *nl); +size_t countEntryInCorkQueue (void); + +/* If a parser sets (CORK_QUEUE and )CORK_SYMTAB to useCork, + * the parsesr can use symbol lookup tables for the current input. + * Each scope has a symbol lookup table. + * To register an tag to the table, use registerEntry(). + * registerEntry registers CORKINDEX to a symbol table of a parent tag + * specified in the scopeIndex field of the tag specified with CORKINDEX. + */ +void registerEntry (int corkIndex); -/* Generating pattern associated tag, caller must do eFree for the returned value. */ -extern char* makePatternString (const tagEntryInfo *const tag); +/* foreachEntriesInScope is for traversing the symbol table for a table + * specified with CORKINDEX. If CORK_NIL is given, this function traverses + * top-level entries. If name is NULL, this function traverses all entries + * under the scope. + * + * If FUNC returns false, this function returns false. + * If FUNC never returns false, this func returns true. + * If FUNC is not called because no node for NAME in the symbol table. + */ +bool foreachEntriesInScope (int corkIndex, + const char *name, /* or NULL */ + entryForeachFunc func, + void *data); +/* Return the cork index for NAME in the scope specified with CORKINDEX. + * Even if more than one entries for NAME are in the scope, this function + * just returns one of them. Returning CORK_NIL means there is no entry + * for NAME. + */ +int anyEntryInScope (int corkIndex, + const char *name); -/* language is optional: can be NULL. */ -extern bool writePseudoTag (const ptagDesc *pdesc, - const char *const fileName, - const char *const pattern, - const char *const parserName); +int anyKindEntryInScope (int corkIndex, + const char *name, int kind); -#define CORK_NIL 0 -void corkTagFile(void); -void uncorkTagFile(void); -tagEntryInfo *getEntryInCorkQueue (unsigned int n); -tagEntryInfo *getEntryOfNestingLevel (const NestingLevel *nl); -size_t countEntryInCorkQueue (void); +int anyKindsEntryInScope (int corkIndex, + const char *name, + const int * kinds, int count); -extern void makeFileTag (const char *const fileName); +int anyKindsEntryInScopeRecursive (int corkIndex, + const char *name, + const int * kinds, int count); extern void markTagExtraBit (tagEntryInfo *const tag, xtagType extra); extern bool isTagExtraBitMarked (const tagEntryInfo *const tag, xtagType extra); -extern void attachParserField (tagEntryInfo *const tag, fieldType ftype, const char* value); +/* If any extra bit is on, return true. */ +extern bool isTagExtra (const tagEntryInfo *const tag); + +/* Functions for attaching parser specific fields + * + * Which function you should use? + * ------------------------------ + * Case A: + * + * If your parser uses the Cork API, and your parser called + * makeTagEntry () already, you can use both + * attachParserFieldToCorkEntry () and attachParserField (). Your + * parser has the cork index returned from makeTagEntry (). With the + * cork index, your parser can call attachParserFieldToCorkEntry (). + * If your parser already call getEntryInCorkQueue () to get the tag + * entry for the cork index, your parser can call attachParserField () + * with passing true for `inCorkQueue' parameter. attachParserField () + * is a bit faster than attachParserFieldToCorkEntry (). + * + * attachParserField () and attachParserFieldToCorkEntry () duplicates + * the memory object specified with `value' and stores the duplicated + * object to the entry on the cork queue. So the parser must/can free + * the original one passed to the functions after calling. The cork + * queue manages the life of the duplicated object. It is not the + * parser's concern. + * + * + * Case B: + * + * If your parser called one of initTagEntry () family but didn't call + * makeTagEntry () for a tagEntry yet, use attachParserField () with + * false for `inCorkQueue' whether your parser uses the Cork API or + * not. + * + * The parser (== caller) keeps the memory object specified with `value' + * till calling makeTagEntry (). The parser must free the memory object + * after calling makeTagEntry () if it is allocated dynamically in the + * parser side. + * + * Interpretation of VALUE + * ----------------------- + * For FIELDTYPE_STRING: + * Both json writer and xref writer prints it as-is. + * + * For FIELDTYPE_STRING|FIELDTYPE_BOOL: + * If VALUE points "" (empty C string), the json writer prints it as + * false, and the xref writer prints it as -. + * If VALUE points a non-empty C string, Both json writer and xref + * writer print it as-is. + * + * For FIELDTYPE_BOOL + * The json writer always prints true. + * The xref writer always prints the name of field. + * Set "" explicitly though the value pointed by VALUE is not referred, + * + * + * The other data type and the combination of types are not implemented yet. + * + */ +extern void attachParserField (tagEntryInfo *const tag, bool inCorkQueue, fieldType ftype, const char* value); extern void attachParserFieldToCorkEntry (int index, fieldType ftype, const char* value); +extern const char* getParserFieldValueForType (tagEntryInfo *const tag@@ Diff output truncated at 100000 characters. @@ -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

[geany/geany] b1c909: Add bundled fnmatch from upstream ctags
by Colomban Wendling 07 Feb '21

07 Feb '21

1 0

[geany/geany] c5303c: Check for asprintf and tempnam and cleanup function checks
by Colomban Wendling 07 Feb '21

07 Feb '21

Branch: refs/heads/master Author: Colomban Wendling <ban(a)herbesfolles.org> Committer: Colomban Wendling <ban(a)herbesfolles.org> Date: Sun, 22 Nov 2020 21:35:24 UTC Commit: c5303c9499bb6f0d632e0ca5b42e15f84d64ac52 https://github.com/geany/geany/commit/c5303c9499bb6f0d632e0ca5b42e15f84d64a… Log Message: ----------- Check for asprintf and tempnam and cleanup function checks - Add the missing check for asprintf which brakes build on Windows/MSYS. - Add fallback check for tempnam if mkstemp is not found. - Group functions checks for u-ctags together. Modified Paths: -------------- configure.ac Modified: configure.ac 4 lines changed, 3 insertions(+), 1 deletions(-) =================================================================== @@ -61,9 +61,11 @@ AC_TYPE_SIZE_T AC_STRUCT_TM # Checks for library functions. -AC_CHECK_FUNCS([fgetpos fnmatch mkstemp strerror strstr realpath]) +AC_CHECK_FUNCS([realpath]) # Function checks for u-ctags +AC_CHECK_FUNCS([strerror strstr asprintf]) +AC_CHECK_FUNCS([mkstemp tempnam], [break]) AC_CHECK_FUNCS([strcasecmp stricmp], [break]) AC_CHECK_FUNCS([strncasecmp strnicmp], [break]) AC_CHECK_FUNCS([truncate ftruncate chsize], [break]) -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

[geany/geany] 3f0bb8: Add a script performing update of Geany ctags from universal ctags
by Jiří Techet 07 Feb '21

07 Feb '21

Branch: refs/heads/master Author: Jiří Techet <techet(a)gmail.com> Committer: Jiří Techet <techet(a)gmail.com> Date: Sat, 21 Nov 2020 21:43:20 UTC Commit: 3f0bb8ed4c2073387eba6686774497f974da7424 https://github.com/geany/geany/commit/3f0bb8ed4c2073387eba6686774497f974da7… Log Message: ----------- Add a script performing update of Geany ctags from universal ctags The script: 1. Copies all parsers from universal ctags not starting with geany_ 2. Copies all files from universal ctags main 3. Prints files which were added/removed to/from main so the corresponding changes can be done manually in Makefile.am 4. Patches main with the provided patch Modified Paths: -------------- scripts/update-ctags.py Modified: scripts/update-ctags.py 42 lines changed, 42 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,42 @@ +#!/usr/bin/env python3 + +import glob +import os +import shutil +import sys + +if len(sys.argv) != 3: + print('Usage: update-ctags.py <universal ctags directory> <geany ctags directory>') + +srcdir = os.path.abspath(sys.argv[1]) +dstdir = os.path.abspath(sys.argv[2]) + +os.chdir(dstdir + '/parsers') +parser_dst_files = glob.glob('*.c') + glob.glob('*.h') +parser_dst_files = list(filter(lambda x: not x.startswith('geany_'), parser_dst_files)) +os.chdir(srcdir + '/parsers') +print('Copying parsers... ({} files)'.format(len(parser_dst_files))) +for f in parser_dst_files: + shutil.copy(f, dstdir + '/parsers') + +os.chdir(srcdir) +main_src_files = glob.glob('main/*.c') + glob.glob('main/*.h') +os.chdir(dstdir) +main_dst_files = glob.glob('main/*.c') + glob.glob('main/*.h') + +for f in main_dst_files: + os.remove(f) +os.chdir(srcdir) +print('Copying main... ({} files)'.format(len(main_src_files))) +for f in main_src_files: + shutil.copy(f, dstdir + '/main') + +main_diff = set(main_dst_files) - set(main_src_files) +if main_diff: + print('Files removed from main: ' + str(main_diff)) +main_diff = set(main_src_files) - set(main_dst_files) +if main_diff: + print('Files added to main: ' + str(main_diff)) + +os.chdir(dstdir) +os.system('patch -p1 <ctags_changes.patch') -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

[geany/geany] ab6816: Don't use tag identifier F
by Jiří Techet 07 Feb '21

07 Feb '21

Branch: refs/heads/master Author: Jiří Techet <techet(a)gmail.com> Committer: Jiří Techet <techet(a)gmail.com> Date: Thu, 19 Nov 2020 10:30:28 UTC Commit: ab681681312acea05b2a484e99dd25647d21dddd https://github.com/geany/geany/commit/ab681681312acea05b2a484e99dd25647d21d… Log Message: ----------- Don't use tag identifier F It's used for file tags upstram and might interfere with it. Modified Paths: -------------- ctags/parsers/geany_objc.c ctags/parsers/geany_ruby.c ctags/parsers/geany_rust.c ctags/parsers/geany_sql.c src/tagmanager/tm_parser.c Modified: ctags/parsers/geany_objc.c 2 lines changed, 1 insertions(+), 1 deletions(-) =================================================================== @@ -46,7 +46,7 @@ static kindDefinition ObjcKinds[] = { {true, 'm', "method", "Object's method"}, {true, 'c', "class", "Class' method"}, {true, 'v', "var", "Global variable"}, - {true, 'F', "field", "Object field"}, + {true, 'E', "field", "Object field"}, {true, 'f', "function", "A function"}, {true, 'p', "property", "A property"}, {true, 't', "typedef", "A type alias"}, Modified: ctags/parsers/geany_ruby.c 2 lines changed, 1 insertions(+), 1 deletions(-) =================================================================== @@ -39,7 +39,7 @@ static kindDefinition RubyKinds [] = { { true, 'c', "class", "classes" }, { true, 'f', "method", "methods" }, { true, 'm', "module", "modules" }, - { true, 'F', "singletonMethod", "singleton methods" }, + { true, 'S', "singletonMethod", "singleton methods" }, #if 0 /* Following two kinds are reserved. */ { true, 'd', "describe", "describes and contexts for Rspec" }, Modified: ctags/parsers/geany_rust.c 2 lines changed, 1 insertions(+), 1 deletions(-) =================================================================== @@ -58,7 +58,7 @@ static kindDefinition rustKinds[] = { {true, 'M', "macro", "Macro Definition"}, {true, 'm', "field", "A struct field"}, {true, 'e', "enumerator", "An enum variant"}, - {true, 'F', "method", "A method"}, + {true, 'P', "method", "A method"}, }; typedef enum { Modified: ctags/parsers/geany_sql.c 2 lines changed, 1 insertions(+), 1 deletions(-) =================================================================== @@ -204,7 +204,7 @@ static kindDefinition SqlKinds [] = { { true, 'c', "cursor", "cursors" }, { false, 'd', "prototype", "prototypes" }, { true, 'f', "function", "functions" }, - { true, 'F', "field", "record fields" }, + { true, 'E', "field", "record fields" }, { false, 'l', "local", "local variables" }, { true, 'L', "label", "block label" }, { true, 'P', "package", "packages" }, Modified: src/tagmanager/tm_parser.c 8 lines changed, 4 insertions(+), 4 deletions(-) =================================================================== @@ -159,7 +159,7 @@ static TMParserMapEntry map_SQL[] = { {'c', tm_tag_undef_t}, {'d', tm_tag_prototype_t}, {'f', tm_tag_function_t}, - {'F', tm_tag_field_t}, + {'E', tm_tag_field_t}, {'l', tm_tag_undef_t}, {'L', tm_tag_undef_t}, {'P', tm_tag_package_t}, @@ -209,7 +209,7 @@ static TMParserMapEntry map_RUBY[] = { {'c', tm_tag_class_t}, {'f', tm_tag_method_t}, {'m', tm_tag_namespace_t}, - {'F', tm_tag_member_t}, + {'S', tm_tag_member_t}, }; static TMParserMapEntry map_TCL[] = { @@ -458,7 +458,7 @@ static TMParserMapEntry map_OBJC[] = { {'m', tm_tag_method_t}, {'c', tm_tag_class_t}, {'v', tm_tag_variable_t}, - {'F', tm_tag_field_t}, + {'E', tm_tag_field_t}, {'f', tm_tag_function_t}, {'p', tm_tag_undef_t}, {'t', tm_tag_typedef_t}, @@ -496,7 +496,7 @@ static TMParserMapEntry map_RUST[] = { {'M', tm_tag_macro_t}, {'m', tm_tag_field_t}, {'e', tm_tag_enumerator_t}, - {'F', tm_tag_method_t}, + {'P', tm_tag_method_t}, }; static TMParserMapEntry map_GO[] = { -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

[geany/geany] a2ecda: Update TM to use latest universal ctags
by Jiří Techet 07 Feb '21

07 Feb '21

Branch: refs/heads/master Author: Jiří Techet <techet(a)gmail.com> Committer: Jiří Techet <techet(a)gmail.com> Date: Thu, 19 Nov 2020 10:30:21 UTC Commit: a2ecdab6d79b19806a5be3feae2058292620f0cc https://github.com/geany/geany/commit/a2ecdab6d79b19806a5be3feae2058292620f… Log Message: ----------- Update TM to use latest universal ctags Introduce new tm_ctags.c/h files which roughly correspond to the previously used ctags-api.c/h files and which serve as an interface between ctags and Geany. Move init_tag() and update_python_arglist() from tm_source_file.c to tm_ctags.c. Use the new functions from tm_ctags in the rest of the tag manager. Define external parser list inside tm_parsers.h which is injected to ctags using the EXTERNAL_PARSER_LIST macro. Modified Paths: -------------- src/tagmanager/Makefile.am src/tagmanager/tm_ctags.c src/tagmanager/tm_ctags.h src/tagmanager/tm_parser.c src/tagmanager/tm_parser.h src/tagmanager/tm_parsers.h src/tagmanager/tm_source_file.c src/tagmanager/tm_source_file.h src/tagmanager/tm_tag.c src/tagmanager/tm_workspace.c Modified: src/tagmanager/Makefile.am 5 lines changed, 4 insertions(+), 1 deletions(-) =================================================================== @@ -17,9 +17,12 @@ tagmanager_include_HEADERS = \ tm_parser.h -libtagmanager_la_SOURCES =\ +libtagmanager_la_SOURCES = \ + tm_ctags.h \ + tm_ctags.c \ tm_parser.h \ tm_parser.c \ + tm_parsers.h \ tm_source_file.h \ tm_source_file.c \ tm_tag.h \ Modified: src/tagmanager/tm_ctags.c 275 lines changed, 275 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,275 @@ +/* +* Copyright (c) 2016, Jiri Techet +* +* This source code is released for free distribution under the terms of the +* GNU General Public License version 2 or (at your option) any later version. +* +* Encapsulates ctags so it is isolated from the rest of Geany. +*/ + +#include "tm_ctags.h" +#include "tm_tag.h" + +#include "general.h" /* must always come before the rest of ctags headers */ +#include "entry_p.h" +#include "error_p.h" +#include "field_p.h" +#include "options_p.h" +#include "parse_p.h" +#include "trashbox_p.h" +#include "writer_p.h" +#include "xtag_p.h" + +#include <string.h> + + +static gint write_entry(tagWriter *writer, MIO * mio, const tagEntryInfo *const tag, void *user_data); +static void rescan_failed(tagWriter *writer, gulong valid_tag_num, void *user_data); + +tagWriter geanyWriter = { + .writeEntry = write_entry, + .writePtagEntry = NULL, /* no pseudo-tags */ + .preWriteEntry = NULL, + .postWriteEntry = NULL, + .rescanFailedEntry = rescan_failed, + .treatFieldAsFixed = NULL, + .defaultFileName = "geany_tags_file_which_should_never_appear_anywhere", + .private = NULL, + .type = WRITER_CUSTOM +}; + + +static bool nonfatal_error_printer(const errorSelection selection, + const gchar *const format, + va_list ap, void *data CTAGS_ATTR_UNUSED) +{ + g_logv(G_LOG_DOMAIN, G_LOG_LEVEL_WARNING, format, ap); + + return false; +} + + +static void enable_all_lang_kinds() +{ + TMParserType lang; + + for (lang = 0; lang < countParsers(); lang++) + { + guint kind_num = countLanguageKinds(lang); + guint kind; + + for (kind = 0; kind < kind_num; kind++) + { + kindDefinition *def = getLanguageKind(lang, kind); + enableKind(def, true); + } + } +} + + +/* + Initializes a TMTag structure with information from a ctagsTag struct + used by the ctags parsers. Note that the TMTag structure must be malloc()ed + before calling this function. + @param tag The TMTag structure to initialize + @param file Pointer to a TMSourceFile struct (it is assigned to the file member) + @param tag_entry Tag information gathered by the ctags parser + @return TRUE on success, FALSE on failure +*/ +static gboolean init_tag(TMTag *tag, TMSourceFile *file, const tagEntryInfo *tag_entry) +{ + TMTagType type; + guchar kind_letter; + TMParserType lang; + + if (!tag_entry) + return FALSE; + + lang = tag_entry->langType; + kind_letter = getLanguageKind(tag_entry->langType, tag_entry->kindIndex)->letter; + type = tm_parser_get_tag_type(kind_letter, lang); + if (file->lang != lang) /* this is a tag from a subparser */ + { + /* check for possible re-definition of subparser type */ + type = tm_parser_get_subparser_type(file->lang, lang, type); + } + + if (!tag_entry->name || type == tm_tag_undef_t) + return FALSE; + + tag->name = g_strdup(tag_entry->name); + tag->type = type; + tag->local = tag_entry->isFileScope; + tag->pointerOrder = 0; /* backward compatibility (use var_type instead) */ + tag->line = tag_entry->lineNumber; + if (NULL != tag_entry->extensionFields.signature) + tag->arglist = g_strdup(tag_entry->extensionFields.signature); + if ((NULL != tag_entry->extensionFields.scopeName) && + (0 != tag_entry->extensionFields.scopeName[0])) + tag->scope = g_strdup(tag_entry->extensionFields.scopeName); + if (tag_entry->extensionFields.inheritance != NULL) + tag->inheritance = g_strdup(tag_entry->extensionFields.inheritance); + if (tag_entry->extensionFields.typeRef[1] != NULL) + tag->var_type = g_strdup(tag_entry->extensionFields.typeRef[1]); + if (tag_entry->extensionFields.access != NULL) + tag->access = tm_source_file_get_tag_access(tag_entry->extensionFields.access); + if (tag_entry->extensionFields.implementation != NULL) + tag->impl = tm_source_file_get_tag_impl(tag_entry->extensionFields.implementation); + if ((tm_tag_macro_t == tag->type) && (NULL != tag->arglist)) + tag->type = tm_tag_macro_with_arg_t; + tag->file = file; + /* redefine lang also for subparsers because the rest of Geany assumes that + * tags from a single file are from a single language */ + tag->lang = file->lang; + return TRUE; +} + + +/* add argument list of __init__() Python methods to the class tag */ +static void update_python_arglist(const TMTag *tag, TMSourceFile *source_file) +{ + guint i; + const gchar *parent_tag_name; + + if (tag->type != tm_tag_method_t || tag->scope == NULL || + g_strcmp0(tag->name, "__init__") != 0) + return; + + parent_tag_name = strrchr(tag->scope, '.'); + if (parent_tag_name) + parent_tag_name++; + else + parent_tag_name = tag->scope; + + /* going in reverse order because the tag was added recently */ + for (i = source_file->tags_array->len; i > 0; i--) + { + TMTag *prev_tag = (TMTag *) source_file->tags_array->pdata[i - 1]; + if (g_strcmp0(prev_tag->name, parent_tag_name) == 0) + { + g_free(prev_tag->arglist); + prev_tag->arglist = g_strdup(tag->arglist); + break; + } + } +} + + +static gint write_entry(tagWriter *writer, MIO * mio, const tagEntryInfo *const tag, void *user_data) +{ + TMSourceFile *source_file = user_data; + TMTag *tm_tag = tm_tag_new(); + + getTagScopeInformation((tagEntryInfo *)tag, NULL, NULL); + + if (!init_tag(tm_tag, source_file, tag)) + { + tm_tag_unref(tm_tag); + return 0; + } + + if (tm_tag->lang == TM_PARSER_PYTHON) + update_python_arglist(tm_tag, source_file); + + g_ptr_array_add(source_file->tags_array, tm_tag); + + /* output length - we don't write anything to the MIO */ + return 0; +} + + +static void rescan_failed(tagWriter *writer, gulong valid_tag_num, void *user_data) +{ + TMSourceFile *source_file = user_data; + GPtrArray *tags_array = source_file->tags_array; + + if (tags_array->len > valid_tag_num) + { + guint i; + for (i = valid_tag_num; i < tags_array->len; i++) + tm_tag_unref(tags_array->pdata[i]); + g_ptr_array_set_size(tags_array, valid_tag_num); + } +} + + +/* keep in sync with ctags main() - use only things interesting for us */ +void tm_ctags_init(void) +{ + initDefaultTrashBox(); + + setErrorPrinter(nonfatal_error_printer, NULL); + setTagWriter(WRITER_CUSTOM, &geanyWriter); + + checkRegex(); + initFieldObjects(); + initXtagObjects(); + + initializeParsing(); + initOptions(); + + /* make sure all parsers are initialized */ + initializeParser(LANG_AUTO); + + /* change default values which are false */ + enableXtag(XTAG_TAGS_GENERATED_BY_GUEST_PARSERS, true); + enableXtag(XTAG_REFERENCE_TAGS, true); + + /* some kinds we are interested in are disabled by default */ + enable_all_lang_kinds(); +} + + +void tm_ctags_parse(guchar *buffer, gsize buffer_size, + const gchar *file_name, TMParserType language, TMSourceFile *source_file) +{ + g_return_if_fail(buffer != NULL || file_name != NULL); + + parseRawBuffer(file_name, buffer, buffer_size, language, source_file); +} + + +const gchar *tm_ctags_get_lang_name(TMParserType lang) +{ + return getLanguageName(lang); +} + + +TMParserType tm_ctags_get_named_lang(const gchar *name) +{ + return getNamedLanguage(name, 0); +} + + +const gchar *tm_ctags_get_lang_kinds(TMParserType lang) +{ + guint kind_num = countLanguageKinds(lang); + static gchar kinds[257]; + guint i; + + for (i = 0; i < kind_num; i++) + kinds[i] = getLanguageKind(lang, i)->letter; + kinds[i] = '\0'; + + return kinds; +} + + +const gchar *tm_ctags_get_kind_name(gchar kind, TMParserType lang) +{ + kindDefinition *def = getLanguageKindForLetter(lang, kind); + return def ? def->name : "unknown"; +} + + +gchar tm_ctags_get_kind_from_name(const gchar *name, TMParserType lang) +{ + kindDefinition *def = getLanguageKindForName(lang, name); + return def ? def->letter : '-'; +} + + +guint tm_ctags_get_lang_count(void) +{ + return countParsers(); +} Modified: src/tagmanager/tm_ctags.h 34 lines changed, 34 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,34 @@ +/* +* Copyright (c) 2016, Jiri Techet +* +* This source code is released for free distribution under the terms of the +* GNU General Public License version 2 or (at your option) any later version. +* +* Encapsulates ctags so it is isolated from the rest of Geany. +*/ +#ifndef TM_CTAGS_H +#define TM_CTAGS_H + +#include <glib.h> + +#include "tm_source_file.h" + +G_BEGIN_DECLS + +#ifdef GEANY_PRIVATE + +void tm_ctags_init(void); +void tm_ctags_parse(guchar *buffer, gsize buffer_size, + const gchar *file_name, TMParserType language, TMSourceFile *source_file); +const gchar *tm_ctags_get_lang_name(TMParserType lang); +TMParserType tm_ctags_get_named_lang(const gchar *name); +const gchar *tm_ctags_get_lang_kinds(TMParserType lang); +const gchar *tm_ctags_get_kind_name(gchar kind, TMParserType lang); +gchar tm_ctags_get_kind_from_name(const gchar *name, TMParserType lang); +guint tm_ctags_get_lang_count(void); + +#endif /* GEANY_PRIVATE */ + +G_END_DECLS + +#endif /* TM_CTAGS_H */ Modified: src/tagmanager/tm_parser.c 21 lines changed, 8 insertions(+), 13 deletions(-) =================================================================== @@ -19,7 +19,7 @@ */ #include "tm_parser.h" -#include "ctags-api.h" +#include "tm_ctags.h" #include <string.h> @@ -691,28 +691,23 @@ void tm_parser_verify_type_mappings(void) { TMParserType lang; - if (TM_PARSER_COUNT > ctagsGetLangCount()) + if (TM_PARSER_COUNT > tm_ctags_get_lang_count()) g_error("More parsers defined in Geany than in ctags"); for (lang = 0; lang < TM_PARSER_COUNT; lang++) { - const gchar *kinds = ctagsGetLangKinds(lang); + const gchar *kinds = tm_ctags_get_lang_kinds(lang); TMParserMap *map = &parser_map[lang]; gchar presence_map[256]; guint i; if (! map->entries || map->size < 1) g_error("No tag types in TM for %s, is the language listed in parser_map?", - ctagsGetLangName(lang)); - - /* TODO: check also regex parser mappings. At the moment there's no way - * to access regex parser definitions in ctags */ - if (ctagsIsUsingRegexParser(lang)) - continue; + tm_ctags_get_lang_name(lang)); if (map->size != strlen(kinds)) g_error("Different number of tag types in TM (%d) and ctags (%d) for %s", - map->size, (int)strlen(kinds), ctagsGetLangName(lang)); + map->size, (int)strlen(kinds), tm_ctags_get_lang_name(lang)); memset(presence_map, 0, sizeof(presence_map)); for (i = 0; i < map->size; i++) @@ -734,10 +729,10 @@ void tm_parser_verify_type_mappings(void) } if (!ctags_found) g_error("Tag type '%c' found in TM but not in ctags for %s", - map->entries[i].kind, ctagsGetLangName(lang)); + map->entries[i].kind, tm_ctags_get_lang_name(lang)); if (!tm_found) g_error("Tag type '%c' found in ctags but not in TM for %s", - kinds[i], ctagsGetLangName(lang)); + kinds[i], tm_ctags_get_lang_name(lang)); presence_map[(unsigned char) map->entries[i].kind]++; } @@ -746,7 +741,7 @@ void tm_parser_verify_type_mappings(void) { if (presence_map[i] > 1) g_error("Duplicate tag type '%c' found for %s", - (gchar)i, ctagsGetLangName(lang)); + (gchar)i, tm_ctags_get_lang_name(lang)); } } } Modified: src/tagmanager/tm_parser.h 2 lines changed, 1 insertions(+), 1 deletions(-) =================================================================== @@ -54,7 +54,7 @@ typedef gint TMParserType; #ifdef GEANY_PRIVATE -/* keep in sync with ctags/parsers.h */ +/* keep in sync with tm_parsers.h and parser_map in tm_parser.c */ enum { TM_PARSER_NONE = -2, /* keep in sync with ctags LANG_IGNORE */ Modified: src/tagmanager/tm_parsers.h 70 lines changed, 70 insertions(+), 0 deletions(-) =================================================================== @@ -0,0 +1,70 @@ +/* +* Copyright (c) 2019, Jiri Techet +* +* This source code is released for free distribution under the terms of the +* GNU General Public License version 2 or (at your option) any later version. +* +* Declares parsers used by ctags +*/ +#ifndef TM_PARSERS_H +#define TM_PARSERS_H + +/* This file is included by ctags by defining EXTERNAL_PARSER_LIST_FILE inside + * ctags/Makefile.am */ + +/* Keep in sync with tm_parser.h */ +#define EXTERNAL_PARSER_LIST \ + CParser, \ + CppParser, \ + JavaParser, \ + MakefileParser, \ + PascalParser, \ + PerlParser, \ + PhpParser, \ + PythonParser, \ + TexParser, \ + AsmParser, \ + ConfParser, \ + SqlParser, \ + DocBookParser, \ + ErlangParser, \ + CssParser, \ + RubyParser, \ + TclParser, \ + ShParser, \ + DParser, \ + FortranParser, \ + FeriteParser, \ + DiffParser, \ + VhdlParser, \ + LuaParser, \ + JavaScriptParser, \ + HaskellParser, \ + CsharpParser, \ + BasicParser,\ + HaxeParser,\ + RstParser, \ + HtmlParser, \ + F77Parser, \ + GLSLParser, \ + MatLabParser, \ + ValaParser, \ + FlexParser, \ + NsisParser, \ + MarkdownParser, \ + Txt2tagsParser, \ + AbcParser, \ + VerilogParser, \ + RParser, \ + CobolParser, \ + ObjcParser, \ + AsciidocParser, \ + AbaqusParser, \ + RustParser, \ + GoParser, \ + JsonParser, \ + ZephirParser, \ + PowerShellParser, \ + BibtexParser + +#endif Modified: src/tagmanager/tm_source_file.c 133 lines changed, 11 insertions(+), 122 deletions(-) =================================================================== @@ -32,7 +32,7 @@ #include "tm_source_file.h" #include "tm_tag.h" #include "tm_parser.h" -#include "ctags-api.h" +#include "tm_ctags.h" typedef struct { @@ -136,7 +136,7 @@ gchar *tm_get_real_path(const gchar *file_name) return NULL; } -static char get_tag_impl(const char *impl) +gchar tm_source_file_get_tag_impl(const gchar *impl) { if ((0 == strcmp("virtual", impl)) || (0 == strcmp("pure virtual", impl))) @@ -148,7 +148,7 @@ static char get_tag_impl(const char *impl) return TAG_IMPL_UNKNOWN; } -static char get_tag_access(const char *access) +gchar tm_source_file_get_tag_access(const gchar *access) { if (0 == strcmp("public", access)) return TAG_ACCESS_PUBLIC; @@ -167,59 +167,6 @@ static char get_tag_access(const char *access) return TAG_ACCESS_UNKNOWN; } -/* - Initializes a TMTag structure with information from a ctagsTag struct - used by the ctags parsers. Note that the TMTag structure must be malloc()ed - before calling this function. - @param tag The TMTag structure to initialize - @param file Pointer to a TMSourceFile struct (it is assigned to the file member) - @param tag_entry Tag information gathered by the ctags parser - @return TRUE on success, FALSE on failure -*/ -static gboolean init_tag(TMTag *tag, TMSourceFile *file, const ctagsTag *tag_entry) -{ - TMTagType type; - - if (!tag_entry) - return FALSE; - - type = tm_parser_get_tag_type(tag_entry->kindLetter, tag_entry->lang); - if (file->lang != tag_entry->lang) /* this is a tag from a subparser */ - { - /* check for possible re-definition of subparser type */ - type = tm_parser_get_subparser_type(file->lang, tag_entry->lang, type); - } - - if (!tag_entry->name || type == tm_tag_undef_t) - return FALSE; - - tag->name = g_strdup(tag_entry->name); - tag->type = type; - tag->local = tag_entry->isFileScope; - tag->pointerOrder = 0; /* backward compatibility (use var_type instead) */ - tag->line = tag_entry->lineNumber; - if (NULL != tag_entry->signature) - tag->arglist = g_strdup(tag_entry->signature); - if ((NULL != tag_entry->scopeName) && - (0 != tag_entry->scopeName[0])) - tag->scope = g_strdup(tag_entry->scopeName); - if (tag_entry->inheritance != NULL) - tag->inheritance = g_strdup(tag_entry->inheritance); - if (tag_entry->varType != NULL) - tag->var_type = g_strdup(tag_entry->varType); - if (tag_entry->access != NULL) - tag->access = get_tag_access(tag_entry->access); - if (tag_entry->implementation != NULL) - tag->impl = get_tag_impl(tag_entry->implementation); - if ((tm_tag_macro_t == tag->type) && (NULL != tag->arglist)) - tag->type = tm_tag_macro_with_arg_t; - tag->file = file; - /* redefine lang also for subparsers because the rest of Geany assumes that - * tags from a single file are from a single language */ - tag->lang = file->lang; - return TRUE; -} - /* Initializes an already malloc()ed TMTag structure by reading a tag entry line from a file. The structure should be allocated beforehand. @@ -429,7 +376,7 @@ static gboolean init_tag_from_file_ctags(TMTag *tag, TMSourceFile *file, FILE *f const gchar *kind = value ? value : key; if (kind[0] && kind[1]) - tag->type = tm_parser_get_tag_type(ctagsGetKindFromName(kind, lang), lang); + tag->type = tm_parser_get_tag_type(tm_ctags_get_kind_from_name(kind, lang), lang); else tag->type = tm_parser_get_tag_type(*kind, lang); } @@ -439,11 +386,11 @@ static gboolean init_tag_from_file_ctags(TMTag *tag, TMSourceFile *file, FILE *f tag->inheritance = g_strdup(value); } else if (0 == strcmp(key, "implementation")) /* implementation limit */ - tag->impl = get_tag_impl(value); + tag->impl = tm_source_file_get_tag_impl(value); else if (0 == strcmp(key, "line")) /* line */ tag->line = atol(value); else if (0 == strcmp(key, "access")) /* access */ - tag->access = get_tag_access(value); + tag->access = tm_source_file_get_tag_access(value); else if (0 == strcmp(key, "class") || 0 == strcmp(key, "enum") || 0 == strcmp(key, "function") || @@ -614,64 +561,6 @@ gboolean tm_source_file_write_tags_file(const gchar *tags_file, GPtrArray *tags_ return ret; } -/* add argument list of __init__() Python methods to the class tag */ -static void update_python_arglist(const TMTag *tag, TMSourceFile *current_source_file) -{ - guint i; - const char *parent_tag_name; - - if (tag->type != tm_tag_method_t || tag->scope == NULL || - g_strcmp0(tag->name, "__init__") != 0) - return; - - parent_tag_name = strrchr(tag->scope, '.'); - if (parent_tag_name) - parent_tag_name++; - else - parent_tag_name = tag->scope; - - /* going in reverse order because the tag was added recently */ - for (i = current_source_file->tags_array->len; i > 0; i--) - { - TMTag *prev_tag = (TMTag *) current_source_file->tags_array->pdata[i - 1]; - if (g_strcmp0(prev_tag->name, parent_tag_name) == 0) - { - g_free(prev_tag->arglist); - prev_tag->arglist = g_strdup(tag->arglist); - break; - } - } -} - -/* new parsing pass ctags callback function */ -static bool ctags_pass_start(void *user_data) -{ - TMSourceFile *current_source_file = user_data; - - tm_tags_array_free(current_source_file->tags_array, FALSE); - return TRUE; -} - -/* new tag ctags callback function */ -static bool ctags_new_tag(const ctagsTag *const tag, - void *user_data) -{ - TMSourceFile *current_source_file = user_data; - TMTag *tm_tag = tm_tag_new(); - - if (!init_tag(tm_tag, current_source_file, tag)) - { - tm_tag_unref(tm_tag); - return TRUE; - } - - if (tm_tag->lang == TM_PARSER_PYTHON) - update_python_arglist(tm_tag, current_source_file); - - g_ptr_array_add(current_source_file->tags_array, tm_tag); - - return TRUE; -} /* Initializes a TMSourceFile structure from a file name. */ static gboolean tm_source_file_init(TMSourceFile *source_file, const char *file_name, @@ -710,7 +599,7 @@ static gboolean tm_source_file_init(TMSourceFile *source_file, const char *file_ if (name == NULL) source_file->lang = TM_PARSER_NONE; else - source_file->lang = ctagsGetNamedLang(name); + source_file->lang = tm_ctags_get_named_lang(name); return TRUE; } @@ -827,8 +716,8 @@ gboolean tm_source_file_parse(TMSourceFile *source_file, guchar* text_buf, gsize tm_tags_array_free(source_file->tags_array, FALSE); - ctagsParse(use_buffer ? text_buf : NULL, buf_size, file_name, - source_file->lang, ctags_new_tag, ctags_pass_start, source_file); + tm_ctags_parse(use_buffer ? text_buf : NULL, buf_size, file_name, + source_file->lang, source_file); return !retry; } @@ -839,7 +728,7 @@ gboolean tm_source_file_parse(TMSourceFile *source_file, guchar* text_buf, gsize */ const gchar *tm_source_file_get_lang_name(TMParserType lang) { - return ctagsGetLangName(lang); + return tm_ctags_get_lang_name(lang); } /* Gets the language index for \a name. @@ -848,5 +737,5 @@ const gchar *tm_source_file_get_lang_name(TMParserType lang) */ TMParserType tm_source_file_get_named_lang(const gchar *name) { - return ctagsGetNamedLang(name); + return tm_ctags_get_named_lang(name); } Modified: src/tagmanager/tm_source_file.h 4 lines changed, 4 insertions(+), 0 deletions(-) =================================================================== @@ -62,6 +62,10 @@ GPtrArray *tm_source_file_read_tags_file(const gchar *tags_file, TMParserType mo gboolean tm_source_file_write_tags_file(const gchar *tags_file, GPtrArray *tags_array); +gchar tm_source_file_get_tag_impl(const gchar *impl); + +gchar tm_source_file_get_tag_access(const gchar *access); + #endif /* GEANY_PRIVATE */ G_END_DECLS Modified: src/tagmanager/tm_tag.c 2 lines changed, 1 insertions(+), 1 deletions(-) =================================================================== @@ -13,7 +13,7 @@ #include <glib-object.h> #include "tm_tag.h" -#include "ctags-api.h" +#include "tm_ctags.h" #define TAG_NEW(T) ((T) = g_slice_new0(TMTag)) Modified: src/tagmanager/tm_workspace.c 10 lines changed, 4 insertions(+), 6 deletions(-) =================================================================== @@ -17,8 +17,6 @@ and a set of individual source files. */ -#include "general.h" - #include <stdio.h> #include <stdlib.h> #include <unistd.h> @@ -32,7 +30,7 @@ #include <glib/gstdio.h> #include "tm_workspace.h" -#include "ctags-api.h" +#include "tm_ctags.h" #include "tm_tag.h" #include "tm_parser.h" @@ -79,7 +77,7 @@ static gboolean tm_create_workspace(void) theWorkspace->typename_array = g_ptr_array_new(); theWorkspace->global_typename_array = g_ptr_array_new(); - ctagsInit(); + tm_ctags_init(); tm_parser_verify_type_mappings(); return TRUE; @@ -664,7 +662,7 @@ static void fill_find_tags_array(GPtrArray *dst, const GPtrArray *src, @param scope The scope name of the tag to find, or NULL. @param type The tag types to return (TMTagType). Can be a bitmask. @param attrs The attributes to sort and dedup on (0 terminated integer array). - @param lang Specifies the language(see the table in parsers.h) of the tags to be found, + @param lang Specifies the language(see the table in tm_parsers.h) of the tags to be found, -1 for all @return Array of matching tags. */ @@ -712,7 +710,7 @@ static void fill_find_tags_array_prefix(GPtrArray *dst, const GPtrArray *src, /* Returns tags with the specified prefix sorted by name. If there are several tags with the same name, only one of them appears in the resulting array. @param prefix The prefix of the tag to find. - @param lang Specifies the language(see the table in parsers.h) of the tags to be found, + @param lang Specifies the language(see the table in tm_parsers.h) of the tags to be found, -1 for all. @param max_num The maximum number of tags to return. @return Array of matching tags sorted by their name. -------------- This E-Mail was brought to you by github_commit_mail.py (Source: https://github.com/geany/infrastructure).

1 0

2025

2024

2023

2022

2021

2020

2019

2018

2017

2016

2015

2014

2013

2012

2011

2010

2009

2008

2007

2006

Commits February 2021