Compare commits

...
Author SHA1 Message Date
Florian Weimer 0c34593491 locale: localdef input files are now encoded in UTF-8 2022-05-17 11:38:29 +02:00
Florian Weimer de0b9d6644 locale: Introduce get_string_U_char into linereader.c
This will permit reusing the Unicode character processing for different
character encodings, not just the current <U...> encoding.
2022-05-17 10:15:56 +02:00
Florian Weimer 7c0759f4e1 locale: Turn ADDC and ADDS into functions in linereader.c
And introduce struct lr_buffer.  The functions addc and adds can
be called from functions, enabling subsequent refactoring.
2022-05-17 09:58:07 +02:00
+205 -180
View File
@@ -416,36 +416,60 @@ get_toplvl_escape (struct linereader *lr)
return &lr->token;
}
/* Multibyte string buffer. */
struct lr_buffer
{
size_t act;
size_t max;
char *buf;
};
#define ADDC(ch) \
do \
{ \
if (bufact == bufmax) \
{ \
bufmax *= 2; \
buf = xrealloc (buf, bufmax); \
} \
buf[bufact++] = (ch); \
} \
while (0)
/* Initialize *LRB with a default-sized buffer. */
static void
lr_buffer_init (struct lr_buffer *lrb)
{
lrb->act = 0;
lrb->max = 56;
lrb->buf = xmalloc (lrb->max);
}
/* Transfers the buffer string from *LRB to LR->token.mbstr. */
static void
lr_buffer_to_token (struct lr_buffer *lrb, struct linereader *lr)
{
lr->token.val.str.startmb = xrealloc (lrb->buf, lrb->act + 1);
lr->token.val.str.startmb[lrb->act] = '\0';
lr->token.val.str.lenmb = lrb->act;
}
#define ADDS(s, l) \
do \
{ \
size_t _l = (l); \
if (bufact + _l > bufmax) \
{ \
if (bufact < _l) \
bufact = _l; \
bufmax *= 2; \
buf = xrealloc (buf, bufmax); \
} \
memcpy (&buf[bufact], s, _l); \
bufact += _l; \
} \
while (0)
/* Adds CH to *LRB. */
static void
addc (struct lr_buffer *lrb, char ch)
{
if (lrb->act == lrb->max)
{
lrb->max *= 2;
lrb->buf = xrealloc (lrb->buf, lrb->max);
}
lrb->buf[lrb->act++] = ch;
}
/* Adds L bytes at S to *LRB. */
static void
adds (struct lr_buffer *lrb, const unsigned char *s, size_t l)
{
if (lrb->max - lrb->act < l)
{
size_t required_size = lrb->act + l;
size_t new_max = 2 * lrb->max;
if (new_max < required_size)
new_max = required_size;
lrb->buf = xrealloc (lrb->buf, new_max);
lrb->max = new_max;
}
memcpy (lrb->buf + lrb->act, s, l);
lrb->act += l;
}
#define ADDWC(ch) \
do \
@@ -467,13 +491,11 @@ get_symname (struct linereader *lr)
1. reserved words
2. ISO 10646 position values
3. all other. */
char *buf;
size_t bufact = 0;
size_t bufmax = 56;
const struct keyword_t *kw;
int ch;
struct lr_buffer lrb;
buf = (char *) xmalloc (bufmax);
lr_buffer_init (&lrb);
do
{
@@ -481,13 +503,13 @@ get_symname (struct linereader *lr)
if (ch == lr->escape_char)
{
int c2 = lr_getc (lr);
ADDC (c2);
addc (&lrb, c2);
if (c2 == '\n')
ch = '\n';
}
else
ADDC (ch);
addc (&lrb, ch);
}
while (ch != '>' && ch != '\n');
@@ -495,39 +517,35 @@ get_symname (struct linereader *lr)
lr_error (lr, _("unterminated symbolic name"));
/* Test for ISO 10646 position value. */
if (buf[0] == 'U' && (bufact == 6 || bufact == 10))
if (lrb.buf[0] == 'U' && (lrb.act == 6 || lrb.act == 10))
{
char *cp = buf + 1;
while (cp < &buf[bufact - 1] && isxdigit (*cp))
char *cp = lrb.buf + 1;
while (cp < &lrb.buf[lrb.act - 1] && isxdigit (*cp))
++cp;
if (cp == &buf[bufact - 1])
if (cp == &lrb.buf[lrb.act - 1])
{
/* Yes, it is. */
lr->token.tok = tok_ucs4;
lr->token.val.ucs4 = strtoul (buf + 1, NULL, 16);
lr->token.val.ucs4 = strtoul (lrb.buf + 1, NULL, 16);
return &lr->token;
}
}
/* It is a symbolic name. Test for reserved words. */
kw = lr->hash_fct (buf, bufact - 1);
kw = lr->hash_fct (lrb.buf, lrb.act - 1);
if (kw != NULL && kw->symname_or_ident == 1)
{
lr->token.tok = kw->token;
free (buf);
free (lrb.buf);
}
else
{
lr->token.tok = tok_bsymbol;
buf = xrealloc (buf, bufact + 1);
buf[bufact] = '\0';
lr->token.val.str.startmb = buf;
lr->token.val.str.lenmb = bufact - 1;
lr_buffer_to_token (&lrb, lr);
--lr->token.val.str.lenmb; /* Hide the training '>'. */
}
return &lr->token;
@@ -537,16 +555,13 @@ get_symname (struct linereader *lr)
static struct token *
get_ident (struct linereader *lr)
{
char *buf;
size_t bufact;
size_t bufmax = 56;
const struct keyword_t *kw;
int ch;
struct lr_buffer lrb;
buf = xmalloc (bufmax);
bufact = 0;
lr_buffer_init (&lrb);
ADDC (lr->buf[lr->idx - 1]);
addc (&lrb, lr->buf[lr->idx - 1]);
while (!isspace ((ch = lr_getc (lr))) && ch != '"' && ch != ';'
&& ch != '<' && ch != ',' && ch != EOF)
@@ -560,32 +575,96 @@ get_ident (struct linereader *lr)
break;
}
}
ADDC (ch);
addc (&lrb, ch);
}
lr_ungetc (lr, ch);
kw = lr->hash_fct (buf, bufact);
kw = lr->hash_fct (lrb.buf, lrb.act);
if (kw != NULL && kw->symname_or_ident == 0)
{
lr->token.tok = kw->token;
free (buf);
free (lrb.buf);
}
else
{
lr->token.tok = tok_ident;
buf = xrealloc (buf, bufact + 1);
buf[bufact] = '\0';
lr->token.val.str.startmb = buf;
lr->token.val.str.lenmb = bufact;
lr_buffer_to_token (&lrb, lr);
}
return &lr->token;
}
/* Process a decoded Unicode character WCH in a string. */
static void
get_string_U_char (struct localedef_t *locale, const struct charmap_t *charmap,
const struct repertoire_t *repertoire,
uint32_t wch, struct lr_buffer *lrb, bool *illegal_string)
{
/* See whether the charmap contains the Uxxxxxxxx names. */
char utmp[10];
snprintf (utmp, sizeof (utmp), "U%08X", wch);
struct charseq *seq = charmap_find_value (charmap, utmp, 9);
if (seq == NULL)
{
/* No, this isn't the case. Now determine from
the repertoire the name of the character and
find it in the charmap. */
if (repertoire != NULL)
{
const char *symbol = repertoire_find_symbol (repertoire, wch);
if (symbol != NULL)
seq = charmap_find_value (charmap, symbol, strlen (symbol));
}
if (seq == NULL)
{
#ifndef NO_TRANSLITERATION
/* Transliterate if possible. */
if (locale != NULL)
{
if ((locale->avail & CTYPE_LOCALE) == 0)
{
/* Load the CTYPE data now. */
int old_needed = locale->needed;
locale->needed = 0;
locale = load_locale (LC_CTYPE, locale->name,
locale->repertoire_name,
charmap, locale);
locale->needed = old_needed;
}
uint32_t *translit;
if ((locale->avail & CTYPE_LOCALE) != 0
&& ((translit = find_translit (locale, charmap, wch))
!= NULL))
/* The CTYPE data contains a matching
transliteration. */
{
for (int i = 0; translit[i] != 0; ++i)
{
snprintf (utmp, sizeof (utmp), "U%08X", translit[i]);
seq = charmap_find_value (charmap, utmp, 9);
assert (seq != NULL);
adds (lrb, seq->bytes, seq->nbytes);
}
return;
}
}
#endif /* NO_TRANSLITERATION */
/* Not a known name. */
*illegal_string = true;
}
}
if (seq != NULL)
adds (lrb, seq->bytes, seq->nbytes);
}
static struct token *
get_string (struct linereader *lr, const struct charmap_t *charmap,
@@ -593,14 +672,10 @@ get_string (struct linereader *lr, const struct charmap_t *charmap,
int verbose)
{
int return_widestr = lr->return_widestr;
char *buf;
struct lr_buffer lrb;
wchar_t *buf2 = NULL;
size_t bufact;
size_t bufmax = 56;
/* We must return two different strings. */
buf = xmalloc (bufmax);
bufact = 0;
lr_buffer_init (&lrb);
/* We know it'll be a string. */
lr->token.tok = tok_string;
@@ -613,23 +688,27 @@ get_string (struct linereader *lr, const struct charmap_t *charmap,
buf2 = NULL;
while ((ch = lr_getc (lr)) != '"' && ch != '\n' && ch != EOF)
ADDC (ch);
{
if (ch >= 0x80)
lr_error (lr, _("illegal 8-bit character in untranslated string"));
addc (&lrb, ch);
}
/* Catch errors with trailing escape character. */
if (bufact > 0 && buf[bufact - 1] == lr->escape_char
&& (bufact == 1 || buf[bufact - 2] != lr->escape_char))
if (lrb.act > 0 && lrb.buf[lrb.act - 1] == lr->escape_char
&& (lrb.act == 1 || lrb.buf[lrb.act - 2] != lr->escape_char))
{
lr_error (lr, _("illegal escape sequence at end of string"));
--bufact;
--lrb.act;
}
else if (ch == '\n' || ch == EOF)
lr_error (lr, _("unterminated string"));
ADDC ('\0');
addc (&lrb, '\0');
}
else
{
int illegal_string = 0;
bool illegal_string = false;
size_t buf2act = 0;
size_t buf2max = 56 * sizeof (uint32_t);
int ch;
@@ -658,20 +737,42 @@ get_string (struct linereader *lr, const struct charmap_t *charmap,
if (ch == lr->escape_char)
{
ch = lr_getc (lr);
if (ch >= 0x80)
{
lr_error (lr, _("illegal 8-bit escape sequence"));
illegal_string = true;
break;
}
if (ch == '\n' || ch == EOF)
break;
}
else if (ch < 0x80)
{
wch = ch;
addc (&lrb, ch);
}
else /* UTF-8 sequence. */
{
if (!get_string_decode_utf8 (lr, ch, &wch))
{
illegal_string = true;
break;
}
get_string_U_char (locale, charmap, repertoire, wch,
&lrb, &illegal_string);
if (illegal_string)
break;
}
ADDC (ch);
if (return_widestr)
ADDWC ((uint32_t) ch);
ADDWC (wch);
continue;
}
/* Now we have to search for the end of the symbolic name, i.e.,
the closing '>'. */
startidx = bufact;
startidx = lrb.act;
while ((ch = lr_getc (lr)) != '>' && ch != '\n' && ch != EOF)
{
if (ch == lr->escape_char)
@@ -680,133 +781,58 @@ get_string (struct linereader *lr, const struct charmap_t *charmap,
if (ch == '\n' || ch == EOF)
break;
}
ADDC (ch);
addc (&lrb, ch);
}
if (ch == '\n' || ch == EOF)
/* Not a correct string. */
break;
if (bufact == startidx)
if (lrb.act == startidx)
{
/* <> is no correct name. Ignore it and also signal an
error. */
illegal_string = 1;
illegal_string = true;
continue;
}
/* It might be a Uxxxx symbol. */
if (buf[startidx] == 'U'
&& (bufact - startidx == 5 || bufact - startidx == 9))
if (lrb.buf[startidx] == 'U'
&& (lrb.act - startidx == 5 || lrb.act - startidx == 9))
{
char *cp = buf + startidx + 1;
while (cp < &buf[bufact] && isxdigit (*cp))
char *cp = lrb.buf + startidx + 1;
while (cp < &lrb.buf[lrb.act] && isxdigit (*cp))
++cp;
if (cp == &buf[bufact])
if (cp == &lrb.buf[lrb.act])
{
char utmp[10];
/* Yes, it is. */
ADDC ('\0');
wch = strtoul (buf + startidx + 1, NULL, 16);
addc (&lrb, '\0');
wch = strtoul (lrb.buf + startidx + 1, NULL, 16);
/* Now forget about the name we just added. */
bufact = startidx;
lrb.act = startidx;
if (return_widestr)
ADDWC (wch);
/* See whether the charmap contains the Uxxxxxxxx names. */
snprintf (utmp, sizeof (utmp), "U%08X", wch);
seq = charmap_find_value (charmap, utmp, 9);
if (seq == NULL)
{
/* No, this isn't the case. Now determine from
the repertoire the name of the character and
find it in the charmap. */
if (repertoire != NULL)
{
const char *symbol;
symbol = repertoire_find_symbol (repertoire, wch);
if (symbol != NULL)
seq = charmap_find_value (charmap, symbol,
strlen (symbol));
}
if (seq == NULL)
{
#ifndef NO_TRANSLITERATION
/* Transliterate if possible. */
if (locale != NULL)
{
uint32_t *translit;
if ((locale->avail & CTYPE_LOCALE) == 0)
{
/* Load the CTYPE data now. */
int old_needed = locale->needed;
locale->needed = 0;
locale = load_locale (LC_CTYPE,
locale->name,
locale->repertoire_name,
charmap, locale);
locale->needed = old_needed;
}
if ((locale->avail & CTYPE_LOCALE) != 0
&& ((translit = find_translit (locale,
charmap, wch))
!= NULL))
/* The CTYPE data contains a matching
transliteration. */
{
int i;
for (i = 0; translit[i] != 0; ++i)
{
char utmp[10];
snprintf (utmp, sizeof (utmp), "U%08X",
translit[i]);
seq = charmap_find_value (charmap, utmp,
9);
assert (seq != NULL);
ADDS (seq->bytes, seq->nbytes);
}
continue;
}
}
#endif /* NO_TRANSLITERATION */
/* Not a known name. */
illegal_string = 1;
}
}
if (seq != NULL)
ADDS (seq->bytes, seq->nbytes);
get_string_U_char (locale, charmap, repertoire, wch,
&lrb, &illegal_string);
continue;
}
}
/* We now have the symbolic name in buf[startidx] to
buf[bufact-1]. Now find out the value for this character
/* We now have the symbolic name in lrb.buf[startidx] to
lrb.buf[lrb.act-1]. Now find out the value for this character
in the charmap as well as in the repertoire map (in this
order). */
seq = charmap_find_value (charmap, &buf[startidx],
bufact - startidx);
seq = charmap_find_value (charmap, &lrb.buf[startidx],
lrb.act - startidx);
if (seq == NULL)
{
/* This name is not in the charmap. */
lr_error (lr, _("symbol `%.*s' not in charmap"),
(int) (bufact - startidx), &buf[startidx]);
illegal_string = 1;
(int) (lrb.act - startidx), &lrb.buf[startidx]);
illegal_string = true;
}
if (return_widestr)
@@ -816,8 +842,8 @@ get_string (struct linereader *lr, const struct charmap_t *charmap,
wch = seq->ucs4;
else
{
wch = repertoire_find_value (repertoire, &buf[startidx],
bufact - startidx);
wch = repertoire_find_value (repertoire, &lrb.buf[startidx],
lrb.act - startidx);
if (seq != NULL)
seq->ucs4 = wch;
}
@@ -826,30 +852,30 @@ get_string (struct linereader *lr, const struct charmap_t *charmap,
{
/* This name is not in the repertoire map. */
lr_error (lr, _("symbol `%.*s' not in repertoire map"),
(int) (bufact - startidx), &buf[startidx]);
illegal_string = 1;
(int) (lrb.act - startidx), &lrb.buf[startidx]);
illegal_string = true;
}
else
ADDWC (wch);
}
/* Now forget about the name we just added. */
bufact = startidx;
lrb.act = startidx;
/* And copy the bytes. */
if (seq != NULL)
ADDS (seq->bytes, seq->nbytes);
adds (&lrb, seq->bytes, seq->nbytes);
}
if (ch == '\n' || ch == EOF)
{
lr_error (lr, _("unterminated string"));
illegal_string = 1;
illegal_string = true;
}
if (illegal_string)
{
free (buf);
free (lrb.buf);
free (buf2);
lr->token.val.str.startmb = NULL;
lr->token.val.str.lenmb = 0;
@@ -859,7 +885,7 @@ get_string (struct linereader *lr, const struct charmap_t *charmap,
return &lr->token;
}
ADDC ('\0');
addc (&lrb, '\0');
if (return_widestr)
{
@@ -870,8 +896,7 @@ get_string (struct linereader *lr, const struct charmap_t *charmap,
}
}
lr->token.val.str.startmb = xrealloc (buf, bufact);
lr->token.val.str.lenmb = bufact;
lr_buffer_to_token (&lrb, lr);
return &lr->token;
}