xref: /vim-8.2.3635/src/charset.c (revision ed37d9b3)
1 /* vi:set ts=8 sts=4 sw=4 noet:
2  *
3  * VIM - Vi IMproved	by Bram Moolenaar
4  *
5  * Do ":help uganda"  in Vim to read copying and usage conditions.
6  * Do ":help credits" in Vim to see a list of people who contributed.
7  * See README.txt for an overview of the Vim source code.
8  */
9 
10 #include "vim.h"
11 
12 #if defined(HAVE_WCHAR_H)
13 # include <wchar.h>	    // for towupper() and towlower()
14 #endif
15 static int win_nolbr_chartabsize(win_T *wp, char_u *s, colnr_T col, int *headp);
16 
17 static unsigned nr2hex(unsigned c);
18 
19 static int    chartab_initialized = FALSE;
20 
21 // b_chartab[] is an array of 32 bytes, each bit representing one of the
22 // characters 0-255.
23 #define SET_CHARTAB(buf, c) (buf)->b_chartab[(unsigned)(c) >> 3] |= (1 << ((c) & 0x7))
24 #define RESET_CHARTAB(buf, c) (buf)->b_chartab[(unsigned)(c) >> 3] &= ~(1 << ((c) & 0x7))
25 #define GET_CHARTAB(buf, c) ((buf)->b_chartab[(unsigned)(c) >> 3] & (1 << ((c) & 0x7)))
26 
27 // table used below, see init_chartab() for an explanation
28 static char_u	g_chartab[256];
29 
30 /*
31  * Flags for g_chartab[].
32  */
33 #define CT_CELL_MASK	0x07	// mask: nr of display cells (1, 2 or 4)
34 #define CT_PRINT_CHAR	0x10	// flag: set for printable chars
35 #define CT_ID_CHAR	0x20	// flag: set for ID chars
36 #define CT_FNAME_CHAR	0x40	// flag: set for file name chars
37 
38 static int in_win_border(win_T *wp, colnr_T vcol);
39 
40 /*
41  * Fill g_chartab[].  Also fills curbuf->b_chartab[] with flags for keyword
42  * characters for current buffer.
43  *
44  * Depends on the option settings 'iskeyword', 'isident', 'isfname',
45  * 'isprint' and 'encoding'.
46  *
47  * The index in g_chartab[] depends on 'encoding':
48  * - For non-multi-byte index with the byte (same as the character).
49  * - For DBCS index with the first byte.
50  * - For UTF-8 index with the character (when first byte is up to 0x80 it is
51  *   the same as the character, if the first byte is 0x80 and above it depends
52  *   on further bytes).
53  *
54  * The contents of g_chartab[]:
55  * - The lower two bits, masked by CT_CELL_MASK, give the number of display
56  *   cells the character occupies (1 or 2).  Not valid for UTF-8 above 0x80.
57  * - CT_PRINT_CHAR bit is set when the character is printable (no need to
58  *   translate the character before displaying it).  Note that only DBCS
59  *   characters can have 2 display cells and still be printable.
60  * - CT_FNAME_CHAR bit is set when the character can be in a file name.
61  * - CT_ID_CHAR bit is set when the character can be in an identifier.
62  *
63  * Return FAIL if 'iskeyword', 'isident', 'isfname' or 'isprint' option has an
64  * error, OK otherwise.
65  */
66     int
67 init_chartab(void)
68 {
69     return buf_init_chartab(curbuf, TRUE);
70 }
71 
72     int
73 buf_init_chartab(
74     buf_T	*buf,
75     int		global)		// FALSE: only set buf->b_chartab[]
76 {
77     int		c;
78     int		c2;
79     char_u	*p;
80     int		i;
81     int		tilde;
82     int		do_isalpha;
83 
84     if (global)
85     {
86 	/*
87 	 * Set the default size for printable characters:
88 	 * From <Space> to '~' is 1 (printable), others are 2 (not printable).
89 	 * This also inits all 'isident' and 'isfname' flags to FALSE.
90 	 *
91 	 * EBCDIC: all chars below ' ' are not printable, all others are
92 	 * printable.
93 	 */
94 	c = 0;
95 	while (c < ' ')
96 	    g_chartab[c++] = (dy_flags & DY_UHEX) ? 4 : 2;
97 #ifdef EBCDIC
98 	while (c < 255)
99 #else
100 	while (c <= '~')
101 #endif
102 	    g_chartab[c++] = 1 + CT_PRINT_CHAR;
103 	while (c < 256)
104 	{
105 	    // UTF-8: bytes 0xa0 - 0xff are printable (latin1)
106 	    if (enc_utf8 && c >= 0xa0)
107 		g_chartab[c++] = CT_PRINT_CHAR + 1;
108 	    // euc-jp characters starting with 0x8e are single width
109 	    else if (enc_dbcs == DBCS_JPNU && c == 0x8e)
110 		g_chartab[c++] = CT_PRINT_CHAR + 1;
111 	    // other double-byte chars can be printable AND double-width
112 	    else if (enc_dbcs != 0 && MB_BYTE2LEN(c) == 2)
113 		g_chartab[c++] = CT_PRINT_CHAR + 2;
114 	    else
115 		// the rest is unprintable by default
116 		g_chartab[c++] = (dy_flags & DY_UHEX) ? 4 : 2;
117 	}
118 
119 	// Assume that every multi-byte char is a filename character.
120 	for (c = 1; c < 256; ++c)
121 	    if ((enc_dbcs != 0 && MB_BYTE2LEN(c) > 1)
122 		    || (enc_dbcs == DBCS_JPNU && c == 0x8e)
123 		    || (enc_utf8 && c >= 0xa0))
124 		g_chartab[c] |= CT_FNAME_CHAR;
125     }
126 
127     /*
128      * Init word char flags all to FALSE
129      */
130     CLEAR_FIELD(buf->b_chartab);
131     if (enc_dbcs != 0)
132 	for (c = 0; c < 256; ++c)
133 	{
134 	    // double-byte characters are probably word characters
135 	    if (MB_BYTE2LEN(c) == 2)
136 		SET_CHARTAB(buf, c);
137 	}
138 
139 #ifdef FEAT_LISP
140     /*
141      * In lisp mode the '-' character is included in keywords.
142      */
143     if (buf->b_p_lisp)
144 	SET_CHARTAB(buf, '-');
145 #endif
146 
147     // Walk through the 'isident', 'iskeyword', 'isfname' and 'isprint'
148     // options Each option is a list of characters, character numbers or
149     // ranges, separated by commas, e.g.: "200-210,x,#-178,-"
150     for (i = global ? 0 : 3; i <= 3; ++i)
151     {
152 	if (i == 0)
153 	    p = p_isi;		// first round: 'isident'
154 	else if (i == 1)
155 	    p = p_isp;		// second round: 'isprint'
156 	else if (i == 2)
157 	    p = p_isf;		// third round: 'isfname'
158 	else	// i == 3
159 	    p = buf->b_p_isk;	// fourth round: 'iskeyword'
160 
161 	while (*p)
162 	{
163 	    tilde = FALSE;
164 	    do_isalpha = FALSE;
165 	    if (*p == '^' && p[1] != NUL)
166 	    {
167 		tilde = TRUE;
168 		++p;
169 	    }
170 	    if (VIM_ISDIGIT(*p))
171 		c = getdigits(&p);
172 	    else
173 		 if (has_mbyte)
174 		c = mb_ptr2char_adv(&p);
175 	    else
176 		c = *p++;
177 	    c2 = -1;
178 	    if (*p == '-' && p[1] != NUL)
179 	    {
180 		++p;
181 		if (VIM_ISDIGIT(*p))
182 		    c2 = getdigits(&p);
183 		else
184 		     if (has_mbyte)
185 		    c2 = mb_ptr2char_adv(&p);
186 		else
187 		    c2 = *p++;
188 	    }
189 	    if (c <= 0 || c >= 256 || (c2 < c && c2 != -1) || c2 >= 256
190 						 || !(*p == NUL || *p == ','))
191 		return FAIL;
192 
193 	    if (c2 == -1)	// not a range
194 	    {
195 		/*
196 		 * A single '@' (not "@-@"):
197 		 * Decide on letters being ID/printable/keyword chars with
198 		 * standard function isalpha(). This takes care of locale for
199 		 * single-byte characters).
200 		 */
201 		if (c == '@')
202 		{
203 		    do_isalpha = TRUE;
204 		    c = 1;
205 		    c2 = 255;
206 		}
207 		else
208 		    c2 = c;
209 	    }
210 	    while (c <= c2)
211 	    {
212 		// Use the MB_ functions here, because isalpha() doesn't
213 		// work properly when 'encoding' is "latin1" and the locale is
214 		// "C".
215 		if (!do_isalpha || MB_ISLOWER(c) || MB_ISUPPER(c))
216 		{
217 		    if (i == 0)			// (re)set ID flag
218 		    {
219 			if (tilde)
220 			    g_chartab[c] &= ~CT_ID_CHAR;
221 			else
222 			    g_chartab[c] |= CT_ID_CHAR;
223 		    }
224 		    else if (i == 1)		// (re)set printable
225 		    {
226 			if ((c < ' '
227 #ifndef EBCDIC
228 				    || c > '~'
229 #endif
230 				// For double-byte we keep the cell width, so
231 				// that we can detect it from the first byte.
232 			    ) && !(enc_dbcs && MB_BYTE2LEN(c) == 2))
233 			{
234 			    if (tilde)
235 			    {
236 				g_chartab[c] = (g_chartab[c] & ~CT_CELL_MASK)
237 					     + ((dy_flags & DY_UHEX) ? 4 : 2);
238 				g_chartab[c] &= ~CT_PRINT_CHAR;
239 			    }
240 			    else
241 			    {
242 				g_chartab[c] = (g_chartab[c] & ~CT_CELL_MASK) + 1;
243 				g_chartab[c] |= CT_PRINT_CHAR;
244 			    }
245 			}
246 		    }
247 		    else if (i == 2)		// (re)set fname flag
248 		    {
249 			if (tilde)
250 			    g_chartab[c] &= ~CT_FNAME_CHAR;
251 			else
252 			    g_chartab[c] |= CT_FNAME_CHAR;
253 		    }
254 		    else // i == 3		 (re)set keyword flag
255 		    {
256 			if (tilde)
257 			    RESET_CHARTAB(buf, c);
258 			else
259 			    SET_CHARTAB(buf, c);
260 		    }
261 		}
262 		++c;
263 	    }
264 
265 	    c = *p;
266 	    p = skip_to_option_part(p);
267 	    if (c == ',' && *p == NUL)
268 		// Trailing comma is not allowed.
269 		return FAIL;
270 	}
271     }
272     chartab_initialized = TRUE;
273     return OK;
274 }
275 
276 /*
277  * Translate any special characters in buf[bufsize] in-place.
278  * The result is a string with only printable characters, but if there is not
279  * enough room, not all characters will be translated.
280  */
281     void
282 trans_characters(
283     char_u	*buf,
284     int		bufsize)
285 {
286     int		len;		// length of string needing translation
287     int		room;		// room in buffer after string
288     char_u	*trs;		// translated character
289     int		trs_len;	// length of trs[]
290 
291     len = (int)STRLEN(buf);
292     room = bufsize - len;
293     while (*buf != 0)
294     {
295 	// Assume a multi-byte character doesn't need translation.
296 	if (has_mbyte && (trs_len = (*mb_ptr2len)(buf)) > 1)
297 	    len -= trs_len;
298 	else
299 	{
300 	    trs = transchar_byte(*buf);
301 	    trs_len = (int)STRLEN(trs);
302 	    if (trs_len > 1)
303 	    {
304 		room -= trs_len - 1;
305 		if (room <= 0)
306 		    return;
307 		mch_memmove(buf + trs_len, buf + 1, (size_t)len);
308 	    }
309 	    mch_memmove(buf, trs, (size_t)trs_len);
310 	    --len;
311 	}
312 	buf += trs_len;
313     }
314 }
315 
316 /*
317  * Translate a string into allocated memory, replacing special chars with
318  * printable chars.  Returns NULL when out of memory.
319  */
320     char_u *
321 transstr(char_u *s)
322 {
323     char_u	*res;
324     char_u	*p;
325     int		l, len, c;
326     char_u	hexbuf[11];
327 
328     if (has_mbyte)
329     {
330 	// Compute the length of the result, taking account of unprintable
331 	// multi-byte characters.
332 	len = 0;
333 	p = s;
334 	while (*p != NUL)
335 	{
336 	    if ((l = (*mb_ptr2len)(p)) > 1)
337 	    {
338 		c = (*mb_ptr2char)(p);
339 		p += l;
340 		if (vim_isprintc(c))
341 		    len += l;
342 		else
343 		{
344 		    transchar_hex(hexbuf, c);
345 		    len += (int)STRLEN(hexbuf);
346 		}
347 	    }
348 	    else
349 	    {
350 		l = byte2cells(*p++);
351 		if (l > 0)
352 		    len += l;
353 		else
354 		    len += 4;	// illegal byte sequence
355 	    }
356 	}
357 	res = alloc(len + 1);
358     }
359     else
360 	res = alloc(vim_strsize(s) + 1);
361     if (res != NULL)
362     {
363 	*res = NUL;
364 	p = s;
365 	while (*p != NUL)
366 	{
367 	    if (has_mbyte && (l = (*mb_ptr2len)(p)) > 1)
368 	    {
369 		c = (*mb_ptr2char)(p);
370 		if (vim_isprintc(c))
371 		    STRNCAT(res, p, l);	// append printable multi-byte char
372 		else
373 		    transchar_hex(res + STRLEN(res), c);
374 		p += l;
375 	    }
376 	    else
377 		STRCAT(res, transchar_byte(*p++));
378 	}
379     }
380     return res;
381 }
382 
383 /*
384  * Convert the string "str[orglen]" to do ignore-case comparing.  Uses the
385  * current locale.
386  * When "buf" is NULL returns an allocated string (NULL for out-of-memory).
387  * Otherwise puts the result in "buf[buflen]".
388  */
389     char_u *
390 str_foldcase(
391     char_u	*str,
392     int		orglen,
393     char_u	*buf,
394     int		buflen)
395 {
396     garray_T	ga;
397     int		i;
398     int		len = orglen;
399 
400 #define GA_CHAR(i)  ((char_u *)ga.ga_data)[i]
401 #define GA_PTR(i)   ((char_u *)ga.ga_data + i)
402 #define STR_CHAR(i)  (buf == NULL ? GA_CHAR(i) : buf[i])
403 #define STR_PTR(i)   (buf == NULL ? GA_PTR(i) : buf + i)
404 
405     // Copy "str" into "buf" or allocated memory, unmodified.
406     if (buf == NULL)
407     {
408 	ga_init2(&ga, 1, 10);
409 	if (ga_grow(&ga, len + 1) == FAIL)
410 	    return NULL;
411 	mch_memmove(ga.ga_data, str, (size_t)len);
412 	ga.ga_len = len;
413     }
414     else
415     {
416 	if (len >= buflen)	    // Ugly!
417 	    len = buflen - 1;
418 	mch_memmove(buf, str, (size_t)len);
419     }
420     if (buf == NULL)
421 	GA_CHAR(len) = NUL;
422     else
423 	buf[len] = NUL;
424 
425     // Make each character lower case.
426     i = 0;
427     while (STR_CHAR(i) != NUL)
428     {
429 	if (enc_utf8 || (has_mbyte && MB_BYTE2LEN(STR_CHAR(i)) > 1))
430 	{
431 	    if (enc_utf8)
432 	    {
433 		int	c = utf_ptr2char(STR_PTR(i));
434 		int	olen = utf_ptr2len(STR_PTR(i));
435 		int	lc = utf_tolower(c);
436 
437 		// Only replace the character when it is not an invalid
438 		// sequence (ASCII character or more than one byte) and
439 		// utf_tolower() doesn't return the original character.
440 		if ((c < 0x80 || olen > 1) && c != lc)
441 		{
442 		    int	    nlen = utf_char2len(lc);
443 
444 		    // If the byte length changes need to shift the following
445 		    // characters forward or backward.
446 		    if (olen != nlen)
447 		    {
448 			if (nlen > olen)
449 			{
450 			    if (buf == NULL
451 				    ? ga_grow(&ga, nlen - olen + 1) == FAIL
452 				    : len + nlen - olen >= buflen)
453 			    {
454 				// out of memory, keep old char
455 				lc = c;
456 				nlen = olen;
457 			    }
458 			}
459 			if (olen != nlen)
460 			{
461 			    if (buf == NULL)
462 			    {
463 				STRMOVE(GA_PTR(i) + nlen, GA_PTR(i) + olen);
464 				ga.ga_len += nlen - olen;
465 			    }
466 			    else
467 			    {
468 				STRMOVE(buf + i + nlen, buf + i + olen);
469 				len += nlen - olen;
470 			    }
471 			}
472 		    }
473 		    (void)utf_char2bytes(lc, STR_PTR(i));
474 		}
475 	    }
476 	    // skip to next multi-byte char
477 	    i += (*mb_ptr2len)(STR_PTR(i));
478 	}
479 	else
480 	{
481 	    if (buf == NULL)
482 		GA_CHAR(i) = TOLOWER_LOC(GA_CHAR(i));
483 	    else
484 		buf[i] = TOLOWER_LOC(buf[i]);
485 	    ++i;
486 	}
487     }
488 
489     if (buf == NULL)
490 	return (char_u *)ga.ga_data;
491     return buf;
492 }
493 
494 /*
495  * Catch 22: g_chartab[] can't be initialized before the options are
496  * initialized, and initializing options may cause transchar() to be called!
497  * When chartab_initialized == FALSE don't use g_chartab[].
498  * Does NOT work for multi-byte characters, c must be <= 255.
499  * Also doesn't work for the first byte of a multi-byte, "c" must be a
500  * character!
501  */
502 static char_u	transchar_buf[7];
503 
504     char_u *
505 transchar(int c)
506 {
507     int			i;
508 
509     i = 0;
510     if (IS_SPECIAL(c))	    // special key code, display as ~@ char
511     {
512 	transchar_buf[0] = '~';
513 	transchar_buf[1] = '@';
514 	i = 2;
515 	c = K_SECOND(c);
516     }
517 
518     if ((!chartab_initialized && (
519 #ifdef EBCDIC
520 		    (c >= 64 && c < 255)
521 #else
522 		    (c >= ' ' && c <= '~')
523 #endif
524 		)) || (c < 256 && vim_isprintc_strict(c)))
525     {
526 	// printable character
527 	transchar_buf[i] = c;
528 	transchar_buf[i + 1] = NUL;
529     }
530     else
531 	transchar_nonprint(transchar_buf + i, c);
532     return transchar_buf;
533 }
534 
535 /*
536  * Like transchar(), but called with a byte instead of a character.  Checks
537  * for an illegal UTF-8 byte.
538  */
539     char_u *
540 transchar_byte(int c)
541 {
542     if (enc_utf8 && c >= 0x80)
543     {
544 	transchar_nonprint(transchar_buf, c);
545 	return transchar_buf;
546     }
547     return transchar(c);
548 }
549 
550 /*
551  * Convert non-printable character to two or more printable characters in
552  * "buf[]".  "buf" needs to be able to hold five bytes.
553  * Does NOT work for multi-byte characters, c must be <= 255.
554  */
555     void
556 transchar_nonprint(char_u *buf, int c)
557 {
558     if (c == NL)
559 	c = NUL;		// we use newline in place of a NUL
560     else if (c == CAR && get_fileformat(curbuf) == EOL_MAC)
561 	c = NL;			// we use CR in place of  NL in this case
562 
563     if (dy_flags & DY_UHEX)		// 'display' has "uhex"
564 	transchar_hex(buf, c);
565 
566 #ifdef EBCDIC
567     // For EBCDIC only the characters 0-63 and 255 are not printable
568     else if (CtrlChar(c) != 0 || c == DEL)
569 #else
570     else if (c <= 0x7f)			// 0x00 - 0x1f and 0x7f
571 #endif
572     {
573 	buf[0] = '^';
574 #ifdef EBCDIC
575 	if (c == DEL)
576 	    buf[1] = '?';		// DEL displayed as ^?
577 	else
578 	    buf[1] = CtrlChar(c);
579 #else
580 	buf[1] = c ^ 0x40;		// DEL displayed as ^?
581 #endif
582 
583 	buf[2] = NUL;
584     }
585     else if (enc_utf8 && c >= 0x80)
586     {
587 	transchar_hex(buf, c);
588     }
589 #ifndef EBCDIC
590     else if (c >= ' ' + 0x80 && c <= '~' + 0x80)    // 0xa0 - 0xfe
591     {
592 	buf[0] = '|';
593 	buf[1] = c - 0x80;
594 	buf[2] = NUL;
595     }
596 #else
597     else if (c < 64)
598     {
599 	buf[0] = '~';
600 	buf[1] = MetaChar(c);
601 	buf[2] = NUL;
602     }
603 #endif
604     else					    // 0x80 - 0x9f and 0xff
605     {
606 	/*
607 	 * TODO: EBCDIC I don't know what to do with this chars, so I display
608 	 * them as '~?' for now
609 	 */
610 	buf[0] = '~';
611 #ifdef EBCDIC
612 	buf[1] = '?';			// 0xff displayed as ~?
613 #else
614 	buf[1] = (c - 0x80) ^ 0x40;	// 0xff displayed as ~?
615 #endif
616 	buf[2] = NUL;
617     }
618 }
619 
620     void
621 transchar_hex(char_u *buf, int c)
622 {
623     int		i = 0;
624 
625     buf[0] = '<';
626     if (c > 255)
627     {
628 	buf[++i] = nr2hex((unsigned)c >> 12);
629 	buf[++i] = nr2hex((unsigned)c >> 8);
630     }
631     buf[++i] = nr2hex((unsigned)c >> 4);
632     buf[++i] = nr2hex((unsigned)c);
633     buf[++i] = '>';
634     buf[++i] = NUL;
635 }
636 
637 /*
638  * Convert the lower 4 bits of byte "c" to its hex character.
639  * Lower case letters are used to avoid the confusion of <F1> being 0xf1 or
640  * function key 1.
641  */
642     static unsigned
643 nr2hex(unsigned c)
644 {
645     if ((c & 0xf) <= 9)
646 	return (c & 0xf) + '0';
647     return (c & 0xf) - 10 + 'a';
648 }
649 
650 /*
651  * Return number of display cells occupied by byte "b".
652  * Caller must make sure 0 <= b <= 255.
653  * For multi-byte mode "b" must be the first byte of a character.
654  * A TAB is counted as two cells: "^I".
655  * For UTF-8 mode this will return 0 for bytes >= 0x80, because the number of
656  * cells depends on further bytes.
657  */
658     int
659 byte2cells(int b)
660 {
661     if (enc_utf8 && b >= 0x80)
662 	return 0;
663     return (g_chartab[b] & CT_CELL_MASK);
664 }
665 
666 /*
667  * Return number of display cells occupied by character "c".
668  * "c" can be a special key (negative number) in which case 3 or 4 is returned.
669  * A TAB is counted as two cells: "^I" or four: "<09>".
670  */
671     int
672 char2cells(int c)
673 {
674     if (IS_SPECIAL(c))
675 	return char2cells(K_SECOND(c)) + 2;
676     if (c >= 0x80)
677     {
678 	// UTF-8: above 0x80 need to check the value
679 	if (enc_utf8)
680 	    return utf_char2cells(c);
681 	// DBCS: double-byte means double-width, except for euc-jp with first
682 	// byte 0x8e
683 	if (enc_dbcs != 0 && c >= 0x100)
684 	{
685 	    if (enc_dbcs == DBCS_JPNU && ((unsigned)c >> 8) == 0x8e)
686 		return 1;
687 	    return 2;
688 	}
689     }
690     return (g_chartab[c & 0xff] & CT_CELL_MASK);
691 }
692 
693 /*
694  * Return number of display cells occupied by character at "*p".
695  * A TAB is counted as two cells: "^I" or four: "<09>".
696  */
697     int
698 ptr2cells(char_u *p)
699 {
700     // For UTF-8 we need to look at more bytes if the first byte is >= 0x80.
701     if (enc_utf8 && *p >= 0x80)
702 	return utf_ptr2cells(p);
703     // For DBCS we can tell the cell count from the first byte.
704     return (g_chartab[*p] & CT_CELL_MASK);
705 }
706 
707 /*
708  * Return the number of character cells string "s" will take on the screen,
709  * counting TABs as two characters: "^I".
710  */
711     int
712 vim_strsize(char_u *s)
713 {
714     return vim_strnsize(s, (int)MAXCOL);
715 }
716 
717 /*
718  * Return the number of character cells string "s[len]" will take on the
719  * screen, counting TABs as two characters: "^I".
720  */
721     int
722 vim_strnsize(char_u *s, int len)
723 {
724     int		size = 0;
725 
726     while (*s != NUL && --len >= 0)
727 	if (has_mbyte)
728 	{
729 	    int	    l = (*mb_ptr2len)(s);
730 
731 	    size += ptr2cells(s);
732 	    s += l;
733 	    len -= l - 1;
734 	}
735 	else
736 	    size += byte2cells(*s++);
737 
738     return size;
739 }
740 
741 /*
742  * Return the number of characters 'c' will take on the screen, taking
743  * into account the size of a tab.
744  * Use a define to make it fast, this is used very often!!!
745  * Also see getvcol() below.
746  */
747 
748 #ifdef FEAT_VARTABS
749 # define RET_WIN_BUF_CHARTABSIZE(wp, buf, p, col) \
750     if (*(p) == TAB && (!(wp)->w_p_list || lcs_tab1)) \
751     { \
752 	return tabstop_padding(col, (buf)->b_p_ts, (buf)->b_p_vts_array); \
753     } \
754     else \
755 	return ptr2cells(p);
756 #else
757 # define RET_WIN_BUF_CHARTABSIZE(wp, buf, p, col) \
758     if (*(p) == TAB && (!(wp)->w_p_list || lcs_tab1)) \
759     { \
760 	int ts; \
761 	ts = (buf)->b_p_ts; \
762 	return (int)(ts - (col % ts)); \
763     } \
764     else \
765 	return ptr2cells(p);
766 #endif
767 
768     int
769 chartabsize(char_u *p, colnr_T col)
770 {
771     RET_WIN_BUF_CHARTABSIZE(curwin, curbuf, p, col)
772 }
773 
774 #ifdef FEAT_LINEBREAK
775     static int
776 win_chartabsize(win_T *wp, char_u *p, colnr_T col)
777 {
778     RET_WIN_BUF_CHARTABSIZE(wp, wp->w_buffer, p, col)
779 }
780 #endif
781 
782 /*
783  * Return the number of characters the string 's' will take on the screen,
784  * taking into account the size of a tab.
785  */
786     int
787 linetabsize(char_u *s)
788 {
789     return linetabsize_col(0, s);
790 }
791 
792 /*
793  * Like linetabsize(), but starting at column "startcol".
794  */
795     int
796 linetabsize_col(int startcol, char_u *s)
797 {
798     colnr_T	col = startcol;
799     char_u	*line = s; // pointer to start of line, for breakindent
800 
801     while (*s != NUL)
802 	col += lbr_chartabsize_adv(line, &s, col);
803     return (int)col;
804 }
805 
806 /*
807  * Like linetabsize(), but for a given window instead of the current one.
808  */
809     int
810 win_linetabsize(win_T *wp, char_u *line, colnr_T len)
811 {
812     colnr_T	col = 0;
813     char_u	*s;
814 
815     for (s = line; *s != NUL && (len == MAXCOL || s < line + len);
816 								MB_PTR_ADV(s))
817 	col += win_lbr_chartabsize(wp, line, s, col, NULL);
818     return (int)col;
819 }
820 
821 /*
822  * Return TRUE if 'c' is a normal identifier character:
823  * Letters and characters from the 'isident' option.
824  */
825     int
826 vim_isIDc(int c)
827 {
828     return (c > 0 && c < 0x100 && (g_chartab[c] & CT_ID_CHAR));
829 }
830 
831 /*
832  * return TRUE if 'c' is a keyword character: Letters and characters from
833  * 'iskeyword' option for the current buffer.
834  * For multi-byte characters mb_get_class() is used (builtin rules).
835  */
836     int
837 vim_iswordc(int c)
838 {
839     return vim_iswordc_buf(c, curbuf);
840 }
841 
842     int
843 vim_iswordc_buf(int c, buf_T *buf)
844 {
845     if (c >= 0x100)
846     {
847 	if (enc_dbcs != 0)
848 	    return dbcs_class((unsigned)c >> 8, (unsigned)(c & 0xff)) >= 2;
849 	if (enc_utf8)
850 	    return utf_class_buf(c, buf) >= 2;
851 	return FALSE;
852     }
853     return (c > 0 && GET_CHARTAB(buf, c) != 0);
854 }
855 
856 /*
857  * Just like vim_iswordc() but uses a pointer to the (multi-byte) character.
858  */
859     int
860 vim_iswordp(char_u *p)
861 {
862     return vim_iswordp_buf(p, curbuf);
863 }
864 
865     int
866 vim_iswordp_buf(char_u *p, buf_T *buf)
867 {
868     int	c = *p;
869 
870     if (has_mbyte && MB_BYTE2LEN(c) > 1)
871 	c = (*mb_ptr2char)(p);
872     return vim_iswordc_buf(c, buf);
873 }
874 
875 /*
876  * return TRUE if 'c' is a valid file-name character
877  * Assume characters above 0x100 are valid (multi-byte).
878  */
879     int
880 vim_isfilec(int c)
881 {
882     return (c >= 0x100 || (c > 0 && (g_chartab[c] & CT_FNAME_CHAR)));
883 }
884 
885 /*
886  * return TRUE if 'c' is a valid file-name character or a wildcard character
887  * Assume characters above 0x100 are valid (multi-byte).
888  * Explicitly interpret ']' as a wildcard character as mch_has_wildcard("]")
889  * returns false.
890  */
891     int
892 vim_isfilec_or_wc(int c)
893 {
894     char_u buf[2];
895 
896     buf[0] = (char_u)c;
897     buf[1] = NUL;
898     return vim_isfilec(c) || c == ']' || mch_has_wildcard(buf);
899 }
900 
901 /*
902  * Return TRUE if 'c' is a printable character.
903  * Assume characters above 0x100 are printable (multi-byte), except for
904  * Unicode.
905  */
906     int
907 vim_isprintc(int c)
908 {
909     if (enc_utf8 && c >= 0x100)
910 	return utf_printable(c);
911     return (c >= 0x100 || (c > 0 && (g_chartab[c] & CT_PRINT_CHAR)));
912 }
913 
914 /*
915  * Strict version of vim_isprintc(c), don't return TRUE if "c" is the head
916  * byte of a double-byte character.
917  */
918     int
919 vim_isprintc_strict(int c)
920 {
921     if (enc_dbcs != 0 && c < 0x100 && MB_BYTE2LEN(c) > 1)
922 	return FALSE;
923     if (enc_utf8 && c >= 0x100)
924 	return utf_printable(c);
925     return (c >= 0x100 || (c > 0 && (g_chartab[c] & CT_PRINT_CHAR)));
926 }
927 
928 /*
929  * like chartabsize(), but also check for line breaks on the screen
930  */
931     int
932 lbr_chartabsize(
933     char_u		*line UNUSED, // start of the line
934     unsigned char	*s,
935     colnr_T		col)
936 {
937 #ifdef FEAT_LINEBREAK
938     if (!curwin->w_p_lbr && *get_showbreak_value(curwin) == NUL
939 							   && !curwin->w_p_bri)
940     {
941 #endif
942 	if (curwin->w_p_wrap)
943 	    return win_nolbr_chartabsize(curwin, s, col, NULL);
944 	RET_WIN_BUF_CHARTABSIZE(curwin, curbuf, s, col)
945 #ifdef FEAT_LINEBREAK
946     }
947     return win_lbr_chartabsize(curwin, line == NULL ? s : line, s, col, NULL);
948 #endif
949 }
950 
951 /*
952  * Call lbr_chartabsize() and advance the pointer.
953  */
954     int
955 lbr_chartabsize_adv(
956     char_u	*line, // start of the line
957     char_u	**s,
958     colnr_T	col)
959 {
960     int		retval;
961 
962     retval = lbr_chartabsize(line, *s, col);
963     MB_PTR_ADV(*s);
964     return retval;
965 }
966 
967 /*
968  * This function is used very often, keep it fast!!!!
969  *
970  * If "headp" not NULL, set *headp to the size of what we for 'showbreak'
971  * string at start of line.  Warning: *headp is only set if it's a non-zero
972  * value, init to 0 before calling.
973  */
974     int
975 win_lbr_chartabsize(
976     win_T	*wp,
977     char_u	*line UNUSED, // start of the line
978     char_u	*s,
979     colnr_T	col,
980     int		*headp UNUSED)
981 {
982 #ifdef FEAT_LINEBREAK
983     int		c;
984     int		size;
985     colnr_T	col2;
986     colnr_T	col_adj = 0; // col + screen size of tab
987     colnr_T	colmax;
988     int		added;
989     int		mb_added = 0;
990     int		numberextra;
991     char_u	*ps;
992     int		tab_corr = (*s == TAB);
993     int		n;
994     char_u	*sbr;
995 
996     /*
997      * No 'linebreak', 'showbreak' and 'breakindent': return quickly.
998      */
999     if (!wp->w_p_lbr && !wp->w_p_bri && *get_showbreak_value(wp) == NUL)
1000 #endif
1001     {
1002 	if (wp->w_p_wrap)
1003 	    return win_nolbr_chartabsize(wp, s, col, headp);
1004 	RET_WIN_BUF_CHARTABSIZE(wp, wp->w_buffer, s, col)
1005     }
1006 
1007 #ifdef FEAT_LINEBREAK
1008     /*
1009      * First get normal size, without 'linebreak'
1010      */
1011     size = win_chartabsize(wp, s, col);
1012     c = *s;
1013     if (tab_corr)
1014 	col_adj = size - 1;
1015 
1016     /*
1017      * If 'linebreak' set check at a blank before a non-blank if the line
1018      * needs a break here
1019      */
1020     if (wp->w_p_lbr
1021 	    && VIM_ISBREAK(c)
1022 	    && !VIM_ISBREAK((int)s[1])
1023 	    && wp->w_p_wrap
1024 	    && wp->w_width != 0)
1025     {
1026 	/*
1027 	 * Count all characters from first non-blank after a blank up to next
1028 	 * non-blank after a blank.
1029 	 */
1030 	numberextra = win_col_off(wp);
1031 	col2 = col;
1032 	colmax = (colnr_T)(wp->w_width - numberextra - col_adj);
1033 	if (col >= colmax)
1034 	{
1035 	    colmax += col_adj;
1036 	    n = colmax +  win_col_off2(wp);
1037 	    if (n > 0)
1038 		colmax += (((col - colmax) / n) + 1) * n - col_adj;
1039 	}
1040 
1041 	for (;;)
1042 	{
1043 	    ps = s;
1044 	    MB_PTR_ADV(s);
1045 	    c = *s;
1046 	    if (!(c != NUL
1047 		    && (VIM_ISBREAK(c)
1048 			|| (!VIM_ISBREAK(c)
1049 			    && (col2 == col || !VIM_ISBREAK((int)*ps))))))
1050 		break;
1051 
1052 	    col2 += win_chartabsize(wp, s, col2);
1053 	    if (col2 >= colmax)		// doesn't fit
1054 	    {
1055 		size = colmax - col + col_adj;
1056 		break;
1057 	    }
1058 	}
1059     }
1060     else if (has_mbyte && size == 2 && MB_BYTE2LEN(*s) > 1
1061 				    && wp->w_p_wrap && in_win_border(wp, col))
1062     {
1063 	++size;		// Count the ">" in the last column.
1064 	mb_added = 1;
1065     }
1066 
1067     /*
1068      * May have to add something for 'breakindent' and/or 'showbreak'
1069      * string at start of line.
1070      * Set *headp to the size of what we add.
1071      */
1072     added = 0;
1073     sbr = get_showbreak_value(wp);
1074     if ((*sbr != NUL || wp->w_p_bri) && wp->w_p_wrap && col != 0)
1075     {
1076 	colnr_T sbrlen = 0;
1077 	int	numberwidth = win_col_off(wp);
1078 
1079 	numberextra = numberwidth;
1080 	col += numberextra + mb_added;
1081 	if (col >= (colnr_T)wp->w_width)
1082 	{
1083 	    col -= wp->w_width;
1084 	    numberextra = wp->w_width - (numberextra - win_col_off2(wp));
1085 	    if (col >= numberextra && numberextra > 0)
1086 		col %= numberextra;
1087 	    if (*sbr != NUL)
1088 	    {
1089 		sbrlen = (colnr_T)MB_CHARLEN(sbr);
1090 		if (col >= sbrlen)
1091 		    col -= sbrlen;
1092 	    }
1093 	    if (col >= numberextra && numberextra > 0)
1094 		col = col % numberextra;
1095 	    else if (col > 0 && numberextra > 0)
1096 		col += numberwidth - win_col_off2(wp);
1097 
1098 	    numberwidth -= win_col_off2(wp);
1099 	}
1100 	if (col == 0 || col + size + sbrlen > (colnr_T)wp->w_width)
1101 	{
1102 	    added = 0;
1103 	    if (*sbr != NUL)
1104 	    {
1105 		if (size + sbrlen + numberwidth > (colnr_T)wp->w_width)
1106 		{
1107 		    // calculate effective window width
1108 		    int width = (colnr_T)wp->w_width - sbrlen - numberwidth;
1109 		    int prev_width = col
1110 				 ? ((colnr_T)wp->w_width - (sbrlen + col)) : 0;
1111 
1112 		    if (width <= 0)
1113 			width = (colnr_T)1;
1114 		    added += ((size - prev_width) / width) * vim_strsize(sbr);
1115 		    if ((size - prev_width) % width)
1116 			// wrapped, add another length of 'sbr'
1117 			added += vim_strsize(sbr);
1118 		}
1119 		else
1120 		    added += vim_strsize(sbr);
1121 	    }
1122 	    if (wp->w_p_bri)
1123 		added += get_breakindent_win(wp, line);
1124 
1125 	    size += added;
1126 	    if (col != 0)
1127 		added = 0;
1128 	}
1129     }
1130     if (headp != NULL)
1131 	*headp = added + mb_added;
1132     return size;
1133 #endif
1134 }
1135 
1136 /*
1137  * Like win_lbr_chartabsize(), except that we know 'linebreak' is off and
1138  * 'wrap' is on.  This means we need to check for a double-byte character that
1139  * doesn't fit at the end of the screen line.
1140  */
1141     static int
1142 win_nolbr_chartabsize(
1143     win_T	*wp,
1144     char_u	*s,
1145     colnr_T	col,
1146     int		*headp)
1147 {
1148     int		n;
1149 
1150     if (*s == TAB && (!wp->w_p_list || lcs_tab1))
1151     {
1152 # ifdef FEAT_VARTABS
1153 	return tabstop_padding(col, wp->w_buffer->b_p_ts,
1154 				    wp->w_buffer->b_p_vts_array);
1155 # else
1156 	n = wp->w_buffer->b_p_ts;
1157 	return (int)(n - (col % n));
1158 # endif
1159     }
1160     n = ptr2cells(s);
1161     // Add one cell for a double-width character in the last column of the
1162     // window, displayed with a ">".
1163     if (n == 2 && MB_BYTE2LEN(*s) > 1 && in_win_border(wp, col))
1164     {
1165 	if (headp != NULL)
1166 	    *headp = 1;
1167 	return 3;
1168     }
1169     return n;
1170 }
1171 
1172 /*
1173  * Return TRUE if virtual column "vcol" is in the rightmost column of window
1174  * "wp".
1175  */
1176     static int
1177 in_win_border(win_T *wp, colnr_T vcol)
1178 {
1179     int		width1;		// width of first line (after line number)
1180     int		width2;		// width of further lines
1181 
1182     if (wp->w_width == 0)	// there is no border
1183 	return FALSE;
1184     width1 = wp->w_width - win_col_off(wp);
1185     if ((int)vcol < width1 - 1)
1186 	return FALSE;
1187     if ((int)vcol == width1 - 1)
1188 	return TRUE;
1189     width2 = width1 + win_col_off2(wp);
1190     if (width2 <= 0)
1191 	return FALSE;
1192     return ((vcol - width1) % width2 == width2 - 1);
1193 }
1194 
1195 /*
1196  * Get virtual column number of pos.
1197  *  start: on the first position of this character (TAB, ctrl)
1198  * cursor: where the cursor is on this character (first char, except for TAB)
1199  *    end: on the last position of this character (TAB, ctrl)
1200  *
1201  * This is used very often, keep it fast!
1202  */
1203     void
1204 getvcol(
1205     win_T	*wp,
1206     pos_T	*pos,
1207     colnr_T	*start,
1208     colnr_T	*cursor,
1209     colnr_T	*end)
1210 {
1211     colnr_T	vcol;
1212     char_u	*ptr;		// points to current char
1213     char_u	*posptr;	// points to char at pos->col
1214     char_u	*line;		// start of the line
1215     int		incr;
1216     int		head;
1217 #ifdef FEAT_VARTABS
1218     int		*vts = wp->w_buffer->b_p_vts_array;
1219 #endif
1220     int		ts = wp->w_buffer->b_p_ts;
1221     int		c;
1222 
1223     vcol = 0;
1224     line = ptr = ml_get_buf(wp->w_buffer, pos->lnum, FALSE);
1225     if (pos->col == MAXCOL)
1226 	posptr = NULL;  // continue until the NUL
1227     else
1228     {
1229 	// Special check for an empty line, which can happen on exit, when
1230 	// ml_get_buf() always returns an empty string.
1231 	if (*ptr == NUL)
1232 	    pos->col = 0;
1233 	posptr = ptr + pos->col;
1234 	if (has_mbyte)
1235 	    // always start on the first byte
1236 	    posptr -= (*mb_head_off)(line, posptr);
1237     }
1238 
1239     /*
1240      * This function is used very often, do some speed optimizations.
1241      * When 'list', 'linebreak', 'showbreak' and 'breakindent' are not set
1242      * use a simple loop.
1243      * Also use this when 'list' is set but tabs take their normal size.
1244      */
1245     if ((!wp->w_p_list || lcs_tab1 != NUL)
1246 #ifdef FEAT_LINEBREAK
1247 	    && !wp->w_p_lbr && *get_showbreak_value(wp) == NUL && !wp->w_p_bri
1248 #endif
1249        )
1250     {
1251 	for (;;)
1252 	{
1253 	    head = 0;
1254 	    c = *ptr;
1255 	    // make sure we don't go past the end of the line
1256 	    if (c == NUL)
1257 	    {
1258 		incr = 1;	// NUL at end of line only takes one column
1259 		break;
1260 	    }
1261 	    // A tab gets expanded, depending on the current column
1262 	    if (c == TAB)
1263 #ifdef FEAT_VARTABS
1264 		incr = tabstop_padding(vcol, ts, vts);
1265 #else
1266 		incr = ts - (vcol % ts);
1267 #endif
1268 	    else
1269 	    {
1270 		if (has_mbyte)
1271 		{
1272 		    // For utf-8, if the byte is >= 0x80, need to look at
1273 		    // further bytes to find the cell width.
1274 		    if (enc_utf8 && c >= 0x80)
1275 			incr = utf_ptr2cells(ptr);
1276 		    else
1277 			incr = g_chartab[c] & CT_CELL_MASK;
1278 
1279 		    // If a double-cell char doesn't fit at the end of a line
1280 		    // it wraps to the next line, it's like this char is three
1281 		    // cells wide.
1282 		    if (incr == 2 && wp->w_p_wrap && MB_BYTE2LEN(*ptr) > 1
1283 			    && in_win_border(wp, vcol))
1284 		    {
1285 			++incr;
1286 			head = 1;
1287 		    }
1288 		}
1289 		else
1290 		    incr = g_chartab[c] & CT_CELL_MASK;
1291 	    }
1292 
1293 	    if (posptr != NULL && ptr >= posptr) // character at pos->col
1294 		break;
1295 
1296 	    vcol += incr;
1297 	    MB_PTR_ADV(ptr);
1298 	}
1299     }
1300     else
1301     {
1302 	for (;;)
1303 	{
1304 	    // A tab gets expanded, depending on the current column
1305 	    head = 0;
1306 	    incr = win_lbr_chartabsize(wp, line, ptr, vcol, &head);
1307 	    // make sure we don't go past the end of the line
1308 	    if (*ptr == NUL)
1309 	    {
1310 		incr = 1;	// NUL at end of line only takes one column
1311 		break;
1312 	    }
1313 
1314 	    if (posptr != NULL && ptr >= posptr) // character at pos->col
1315 		break;
1316 
1317 	    vcol += incr;
1318 	    MB_PTR_ADV(ptr);
1319 	}
1320     }
1321     if (start != NULL)
1322 	*start = vcol + head;
1323     if (end != NULL)
1324 	*end = vcol + incr - 1;
1325     if (cursor != NULL)
1326     {
1327 	if (*ptr == TAB
1328 		&& (State & NORMAL)
1329 		&& !wp->w_p_list
1330 		&& !virtual_active()
1331 		&& !(VIsual_active
1332 				&& (*p_sel == 'e' || LTOREQ_POS(*pos, VIsual)))
1333 		)
1334 	    *cursor = vcol + incr - 1;	    // cursor at end
1335 	else
1336 	    *cursor = vcol + head;	    // cursor at start
1337     }
1338 }
1339 
1340 /*
1341  * Get virtual cursor column in the current window, pretending 'list' is off.
1342  */
1343     colnr_T
1344 getvcol_nolist(pos_T *posp)
1345 {
1346     int		list_save = curwin->w_p_list;
1347     colnr_T	vcol;
1348 
1349     curwin->w_p_list = FALSE;
1350     if (posp->coladd)
1351 	getvvcol(curwin, posp, NULL, &vcol, NULL);
1352     else
1353 	getvcol(curwin, posp, NULL, &vcol, NULL);
1354     curwin->w_p_list = list_save;
1355     return vcol;
1356 }
1357 
1358 /*
1359  * Get virtual column in virtual mode.
1360  */
1361     void
1362 getvvcol(
1363     win_T	*wp,
1364     pos_T	*pos,
1365     colnr_T	*start,
1366     colnr_T	*cursor,
1367     colnr_T	*end)
1368 {
1369     colnr_T	col;
1370     colnr_T	coladd;
1371     colnr_T	endadd;
1372     char_u	*ptr;
1373 
1374     if (virtual_active())
1375     {
1376 	// For virtual mode, only want one value
1377 	getvcol(wp, pos, &col, NULL, NULL);
1378 
1379 	coladd = pos->coladd;
1380 	endadd = 0;
1381 	// Cannot put the cursor on part of a wide character.
1382 	ptr = ml_get_buf(wp->w_buffer, pos->lnum, FALSE);
1383 	if (pos->col < (colnr_T)STRLEN(ptr))
1384 	{
1385 	    int c = (*mb_ptr2char)(ptr + pos->col);
1386 
1387 	    if (c != TAB && vim_isprintc(c))
1388 	    {
1389 		endadd = (colnr_T)(char2cells(c) - 1);
1390 		if (coladd > endadd)	// past end of line
1391 		    endadd = 0;
1392 		else
1393 		    coladd = 0;
1394 	    }
1395 	}
1396 	col += coladd;
1397 	if (start != NULL)
1398 	    *start = col;
1399 	if (cursor != NULL)
1400 	    *cursor = col;
1401 	if (end != NULL)
1402 	    *end = col + endadd;
1403     }
1404     else
1405 	getvcol(wp, pos, start, cursor, end);
1406 }
1407 
1408 /*
1409  * Get the leftmost and rightmost virtual column of pos1 and pos2.
1410  * Used for Visual block mode.
1411  */
1412     void
1413 getvcols(
1414     win_T	*wp,
1415     pos_T	*pos1,
1416     pos_T	*pos2,
1417     colnr_T	*left,
1418     colnr_T	*right)
1419 {
1420     colnr_T	from1, from2, to1, to2;
1421 
1422     if (LT_POSP(pos1, pos2))
1423     {
1424 	getvvcol(wp, pos1, &from1, NULL, &to1);
1425 	getvvcol(wp, pos2, &from2, NULL, &to2);
1426     }
1427     else
1428     {
1429 	getvvcol(wp, pos2, &from1, NULL, &to1);
1430 	getvvcol(wp, pos1, &from2, NULL, &to2);
1431     }
1432     if (from2 < from1)
1433 	*left = from2;
1434     else
1435 	*left = from1;
1436     if (to2 > to1)
1437     {
1438 	if (*p_sel == 'e' && from2 - 1 >= to1)
1439 	    *right = from2 - 1;
1440 	else
1441 	    *right = to2;
1442     }
1443     else
1444 	*right = to1;
1445 }
1446 
1447 /*
1448  * skipwhite: skip over ' ' and '\t'.
1449  */
1450     char_u *
1451 skipwhite(char_u *q)
1452 {
1453     char_u	*p = q;
1454 
1455     while (VIM_ISWHITE(*p)) // skip to next non-white
1456 	++p;
1457     return p;
1458 }
1459 
1460 /*
1461  * getwhitecols: return the number of whitespace
1462  * columns (bytes) at the start of a given line
1463  */
1464     int
1465 getwhitecols_curline()
1466 {
1467     return getwhitecols(ml_get_curline());
1468 }
1469 
1470     int
1471 getwhitecols(char_u *p)
1472 {
1473     return skipwhite(p) - p;
1474 }
1475 
1476 /*
1477  * skip over digits
1478  */
1479     char_u *
1480 skipdigits(char_u *q)
1481 {
1482     char_u	*p = q;
1483 
1484     while (VIM_ISDIGIT(*p))	// skip to next non-digit
1485 	++p;
1486     return p;
1487 }
1488 
1489 #if defined(FEAT_SYN_HL) || defined(FEAT_SPELL) || defined(PROTO)
1490 /*
1491  * skip over binary digits
1492  */
1493     char_u *
1494 skipbin(char_u *q)
1495 {
1496     char_u	*p = q;
1497 
1498     while (vim_isbdigit(*p))	// skip to next non-digit
1499 	++p;
1500     return p;
1501 }
1502 
1503 /*
1504  * skip over digits and hex characters
1505  */
1506     char_u *
1507 skiphex(char_u *q)
1508 {
1509     char_u	*p = q;
1510 
1511     while (vim_isxdigit(*p))	// skip to next non-digit
1512 	++p;
1513     return p;
1514 }
1515 #endif
1516 
1517 /*
1518  * skip to bin digit (or NUL after the string)
1519  */
1520     char_u *
1521 skiptobin(char_u *q)
1522 {
1523     char_u	*p = q;
1524 
1525     while (*p != NUL && !vim_isbdigit(*p))	// skip to next digit
1526 	++p;
1527     return p;
1528 }
1529 
1530 /*
1531  * skip to digit (or NUL after the string)
1532  */
1533     char_u *
1534 skiptodigit(char_u *q)
1535 {
1536     char_u	*p = q;
1537 
1538     while (*p != NUL && !VIM_ISDIGIT(*p))	// skip to next digit
1539 	++p;
1540     return p;
1541 }
1542 
1543 /*
1544  * skip to hex character (or NUL after the string)
1545  */
1546     char_u *
1547 skiptohex(char_u *q)
1548 {
1549     char_u	*p = q;
1550 
1551     while (*p != NUL && !vim_isxdigit(*p))	// skip to next digit
1552 	++p;
1553     return p;
1554 }
1555 
1556 /*
1557  * Variant of isdigit() that can handle characters > 0x100.
1558  * We don't use isdigit() here, because on some systems it also considers
1559  * superscript 1 to be a digit.
1560  * Use the VIM_ISDIGIT() macro for simple arguments.
1561  */
1562     int
1563 vim_isdigit(int c)
1564 {
1565     return (c >= '0' && c <= '9');
1566 }
1567 
1568 /*
1569  * Variant of isxdigit() that can handle characters > 0x100.
1570  * We don't use isxdigit() here, because on some systems it also considers
1571  * superscript 1 to be a digit.
1572  */
1573     int
1574 vim_isxdigit(int c)
1575 {
1576     return (c >= '0' && c <= '9')
1577 	|| (c >= 'a' && c <= 'f')
1578 	|| (c >= 'A' && c <= 'F');
1579 }
1580 
1581 /*
1582  * Corollary of vim_isdigit and vim_isxdigit() that can handle
1583  * characters > 0x100.
1584  */
1585     int
1586 vim_isbdigit(int c)
1587 {
1588     return (c == '0' || c == '1');
1589 }
1590 
1591 /*
1592  * Vim's own character class functions.  These exist because many library
1593  * islower()/toupper() etc. do not work properly: they crash when used with
1594  * invalid values or can't handle latin1 when the locale is C.
1595  * Speed is most important here.
1596  */
1597 #define LATIN1LOWER 'l'
1598 #define LATIN1UPPER 'U'
1599 
1600 static char_u latin1flags[257] = "                                                                 UUUUUUUUUUUUUUUUUUUUUUUUUU      llllllllllllllllllllllllll                                                                     UUUUUUUUUUUUUUUUUUUUUUU UUUUUUUllllllllllllllllllllllll llllllll";
1601 static char_u latin1upper[257] = "                                 !\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`ABCDEFGHIJKLMNOPQRSTUVWXYZ{|}~\x7f\x80\x81\x82\x83\x84\x85\x86\x87\x88\x89\x8a\x8b\x8c\x8d\x8e\x8f\x90\x91\x92\x93\x94\x95\x96\x97\x98\x99\x9a\x9b\x9c\x9d\x9e\x9f\xa0\xa1\xa2\xa3\xa4\xa5\xa6\xa7\xa8\xa9\xaa\xab\xac\xad\xae\xaf\xb0\xb1\xb2\xb3\xb4\xb5\xb6\xb7\xb8\xb9\xba\xbb\xbc\xbd\xbe\xbf\xc0\xc1\xc2\xc3\xc4\xc5\xc6\xc7\xc8\xc9\xca\xcb\xcc\xcd\xce\xcf\xd0\xd1\xd2\xd3\xd4\xd5\xd6\xd7\xd8\xd9\xda\xdb\xdc\xdd\xde\xdf\xc0\xc1\xc2\xc3\xc4\xc5\xc6\xc7\xc8\xc9\xca\xcb\xcc\xcd\xce\xcf\xd0\xd1\xd2\xd3\xd4\xd5\xd6\xf7\xd8\xd9\xda\xdb\xdc\xdd\xde\xff";
1602 static char_u latin1lower[257] = "                                 !\"#$%&'()*+,-./0123456789:;<=>?@abcdefghijklmnopqrstuvwxyz[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~\x7f\x80\x81\x82\x83\x84\x85\x86\x87\x88\x89\x8a\x8b\x8c\x8d\x8e\x8f\x90\x91\x92\x93\x94\x95\x96\x97\x98\x99\x9a\x9b\x9c\x9d\x9e\x9f\xa0\xa1\xa2\xa3\xa4\xa5\xa6\xa7\xa8\xa9\xaa\xab\xac\xad\xae\xaf\xb0\xb1\xb2\xb3\xb4\xb5\xb6\xb7\xb8\xb9\xba\xbb\xbc\xbd\xbe\xbf\xe0\xe1\xe2\xe3\xe4\xe5\xe6\xe7\xe8\xe9\xea\xeb\xec\xed\xee\xef\xf0\xf1\xf2\xf3\xf4\xf5\xf6\xd7\xf8\xf9\xfa\xfb\xfc\xfd\xfe\xdf\xe0\xe1\xe2\xe3\xe4\xe5\xe6\xe7\xe8\xe9\xea\xeb\xec\xed\xee\xef\xf0\xf1\xf2\xf3\xf4\xf5\xf6\xf7\xf8\xf9\xfa\xfb\xfc\xfd\xfe\xff";
1603 
1604     int
1605 vim_islower(int c)
1606 {
1607     if (c <= '@')
1608 	return FALSE;
1609     if (c >= 0x80)
1610     {
1611 	if (enc_utf8)
1612 	    return utf_islower(c);
1613 	if (c >= 0x100)
1614 	{
1615 #ifdef HAVE_ISWLOWER
1616 	    if (has_mbyte)
1617 		return iswlower(c);
1618 #endif
1619 	    // islower() can't handle these chars and may crash
1620 	    return FALSE;
1621 	}
1622 	if (enc_latin1like)
1623 	    return (latin1flags[c] & LATIN1LOWER) == LATIN1LOWER;
1624     }
1625     return islower(c);
1626 }
1627 
1628     int
1629 vim_isupper(int c)
1630 {
1631     if (c <= '@')
1632 	return FALSE;
1633     if (c >= 0x80)
1634     {
1635 	if (enc_utf8)
1636 	    return utf_isupper(c);
1637 	if (c >= 0x100)
1638 	{
1639 #ifdef HAVE_ISWUPPER
1640 	    if (has_mbyte)
1641 		return iswupper(c);
1642 #endif
1643 	    // islower() can't handle these chars and may crash
1644 	    return FALSE;
1645 	}
1646 	if (enc_latin1like)
1647 	    return (latin1flags[c] & LATIN1UPPER) == LATIN1UPPER;
1648     }
1649     return isupper(c);
1650 }
1651 
1652     int
1653 vim_toupper(int c)
1654 {
1655     if (c <= '@')
1656 	return c;
1657     if (c >= 0x80 || !(cmp_flags & CMP_KEEPASCII))
1658     {
1659 	if (enc_utf8)
1660 	    return utf_toupper(c);
1661 	if (c >= 0x100)
1662 	{
1663 #ifdef HAVE_TOWUPPER
1664 	    if (has_mbyte)
1665 		return towupper(c);
1666 #endif
1667 	    // toupper() can't handle these chars and may crash
1668 	    return c;
1669 	}
1670 	if (enc_latin1like)
1671 	    return latin1upper[c];
1672     }
1673     if (c < 0x80 && (cmp_flags & CMP_KEEPASCII))
1674 	return TOUPPER_ASC(c);
1675     return TOUPPER_LOC(c);
1676 }
1677 
1678     int
1679 vim_tolower(int c)
1680 {
1681     if (c <= '@')
1682 	return c;
1683     if (c >= 0x80 || !(cmp_flags & CMP_KEEPASCII))
1684     {
1685 	if (enc_utf8)
1686 	    return utf_tolower(c);
1687 	if (c >= 0x100)
1688 	{
1689 #ifdef HAVE_TOWLOWER
1690 	    if (has_mbyte)
1691 		return towlower(c);
1692 #endif
1693 	    // tolower() can't handle these chars and may crash
1694 	    return c;
1695 	}
1696 	if (enc_latin1like)
1697 	    return latin1lower[c];
1698     }
1699     if (c < 0x80 && (cmp_flags & CMP_KEEPASCII))
1700 	return TOLOWER_ASC(c);
1701     return TOLOWER_LOC(c);
1702 }
1703 
1704 /*
1705  * skiptowhite: skip over text until ' ' or '\t' or NUL.
1706  */
1707     char_u *
1708 skiptowhite(char_u *p)
1709 {
1710     while (*p != ' ' && *p != '\t' && *p != NUL)
1711 	++p;
1712     return p;
1713 }
1714 
1715 /*
1716  * skiptowhite_esc: Like skiptowhite(), but also skip escaped chars
1717  */
1718     char_u *
1719 skiptowhite_esc(char_u *p)
1720 {
1721     while (*p != ' ' && *p != '\t' && *p != NUL)
1722     {
1723 	if ((*p == '\\' || *p == Ctrl_V) && *(p + 1) != NUL)
1724 	    ++p;
1725 	++p;
1726     }
1727     return p;
1728 }
1729 
1730 /*
1731  * Getdigits: Get a number from a string and skip over it.
1732  * Note: the argument is a pointer to a char_u pointer!
1733  */
1734     long
1735 getdigits(char_u **pp)
1736 {
1737     char_u	*p;
1738     long	retval;
1739 
1740     p = *pp;
1741     retval = atol((char *)p);
1742     if (*p == '-')		// skip negative sign
1743 	++p;
1744     p = skipdigits(p);		// skip to next non-digit
1745     *pp = p;
1746     return retval;
1747 }
1748 
1749 /*
1750  * Return TRUE if "lbuf" is empty or only contains blanks.
1751  */
1752     int
1753 vim_isblankline(char_u *lbuf)
1754 {
1755     char_u	*p;
1756 
1757     p = skipwhite(lbuf);
1758     return (*p == NUL || *p == '\r' || *p == '\n');
1759 }
1760 
1761 /*
1762  * Convert a string into a long and/or unsigned long, taking care of
1763  * hexadecimal, octal, and binary numbers.  Accepts a '-' sign.
1764  * If "prep" is not NULL, returns a flag to indicate the type of the number:
1765  *  0	    decimal
1766  *  '0'	    octal
1767  *  'B'	    bin
1768  *  'b'	    bin
1769  *  'X'	    hex
1770  *  'x'	    hex
1771  * If "len" is not NULL, the length of the number in characters is returned.
1772  * If "nptr" is not NULL, the signed result is returned in it.
1773  * If "unptr" is not NULL, the unsigned result is returned in it.
1774  * If "what" contains STR2NR_BIN recognize binary numbers
1775  * If "what" contains STR2NR_OCT recognize octal numbers
1776  * If "what" contains STR2NR_HEX recognize hex numbers
1777  * If "what" contains STR2NR_FORCE always assume bin/oct/hex.
1778  * If "what" contains STR2NR_QUOTE ignore embedded single quotes
1779  * If maxlen > 0, check at a maximum maxlen chars.
1780  * If strict is TRUE, check the number strictly. return *len = 0 if fail.
1781  */
1782     void
1783 vim_str2nr(
1784     char_u		*start,
1785     int			*prep,	    // return: type of number 0 = decimal, 'x'
1786 				    // or 'X' is hex, '0' = octal, 'b' or 'B'
1787 				    // is bin
1788     int			*len,	    // return: detected length of number
1789     int			what,	    // what numbers to recognize
1790     varnumber_T		*nptr,	    // return: signed result
1791     uvarnumber_T	*unptr,	    // return: unsigned result
1792     int			maxlen,     // max length of string to check
1793     int			strict)     // check strictly
1794 {
1795     char_u	    *ptr = start;
1796     int		    pre = 0;		// default is decimal
1797     int		    negative = FALSE;
1798     uvarnumber_T    un = 0;
1799     int		    n;
1800 
1801     if (len != NULL)
1802 	*len = 0;
1803 
1804     if (ptr[0] == '-')
1805     {
1806 	negative = TRUE;
1807 	++ptr;
1808     }
1809 
1810     // Recognize hex, octal, and bin.
1811     if (ptr[0] == '0' && ptr[1] != '8' && ptr[1] != '9'
1812 					       && (maxlen == 0 || maxlen > 1))
1813     {
1814 	pre = ptr[1];
1815 	if ((what & STR2NR_HEX)
1816 		&& (pre == 'X' || pre == 'x') && vim_isxdigit(ptr[2])
1817 		&& (maxlen == 0 || maxlen > 2))
1818 	    // hexadecimal
1819 	    ptr += 2;
1820 	else if ((what & STR2NR_BIN)
1821 		&& (pre == 'B' || pre == 'b') && vim_isbdigit(ptr[2])
1822 		&& (maxlen == 0 || maxlen > 2))
1823 	    // binary
1824 	    ptr += 2;
1825 	else
1826 	{
1827 	    // decimal or octal, default is decimal
1828 	    pre = 0;
1829 	    if (what & STR2NR_OCT)
1830 	    {
1831 		// Don't interpret "0", "08" or "0129" as octal.
1832 		for (n = 1; n != maxlen && VIM_ISDIGIT(ptr[n]); ++n)
1833 		{
1834 		    if (ptr[n] > '7')
1835 		    {
1836 			pre = 0;	// can't be octal
1837 			break;
1838 		    }
1839 		    pre = '0';	// assume octal
1840 		}
1841 	    }
1842 	}
1843     }
1844 
1845     // Do the conversion manually to avoid sscanf() quirks.
1846     n = 1;
1847     if (pre == 'B' || pre == 'b'
1848 			     || ((what & STR2NR_BIN) && (what & STR2NR_FORCE)))
1849     {
1850 	// bin
1851 	if (pre != 0)
1852 	    n += 2;	    // skip over "0b"
1853 	while ('0' <= *ptr && *ptr <= '1')
1854 	{
1855 	    // avoid ubsan error for overflow
1856 	    if (un <= UVARNUM_MAX / 2)
1857 		un = 2 * un + (uvarnumber_T)(*ptr - '0');
1858 	    else
1859 		un = UVARNUM_MAX;
1860 	    ++ptr;
1861 	    if (n++ == maxlen)
1862 		break;
1863 	    if ((what & STR2NR_QUOTE) && *ptr == '\''
1864 					     && '0' <= ptr[1] && ptr[1] <= '1')
1865 	    {
1866 		++ptr;
1867 		if (n++ == maxlen)
1868 		    break;
1869 	    }
1870 	}
1871     }
1872     else if (pre == '0' || ((what & STR2NR_OCT) && (what & STR2NR_FORCE)))
1873     {
1874 	// octal
1875 	while ('0' <= *ptr && *ptr <= '7')
1876 	{
1877 	    // avoid ubsan error for overflow
1878 	    if (un <= UVARNUM_MAX / 8)
1879 		un = 8 * un + (uvarnumber_T)(*ptr - '0');
1880 	    else
1881 		un = UVARNUM_MAX;
1882 	    ++ptr;
1883 	    if (n++ == maxlen)
1884 		break;
1885 	    if ((what & STR2NR_QUOTE) && *ptr == '\''
1886 					     && '0' <= ptr[1] && ptr[1] <= '7')
1887 	    {
1888 		++ptr;
1889 		if (n++ == maxlen)
1890 		    break;
1891 	    }
1892 	}
1893     }
1894     else if (pre != 0 || ((what & STR2NR_HEX) && (what & STR2NR_FORCE)))
1895     {
1896 	// hex
1897 	if (pre != 0)
1898 	    n += 2;	    // skip over "0x"
1899 	while (vim_isxdigit(*ptr))
1900 	{
1901 	    // avoid ubsan error for overflow
1902 	    if (un <= UVARNUM_MAX / 16)
1903 		un = 16 * un + (uvarnumber_T)hex2nr(*ptr);
1904 	    else
1905 		un = UVARNUM_MAX;
1906 	    ++ptr;
1907 	    if (n++ == maxlen)
1908 		break;
1909 	    if ((what & STR2NR_QUOTE) && *ptr == '\'' && vim_isxdigit(ptr[1]))
1910 	    {
1911 		++ptr;
1912 		if (n++ == maxlen)
1913 		    break;
1914 	    }
1915 	}
1916     }
1917     else
1918     {
1919 	// decimal
1920 	while (VIM_ISDIGIT(*ptr))
1921 	{
1922 	    uvarnumber_T    digit = (uvarnumber_T)(*ptr - '0');
1923 
1924 	    // avoid ubsan error for overflow
1925 	    if (un < UVARNUM_MAX / 10
1926 		    || (un == UVARNUM_MAX / 10 && digit <= UVARNUM_MAX % 10))
1927 		un = 10 * un + digit;
1928 	    else
1929 		un = UVARNUM_MAX;
1930 	    ++ptr;
1931 	    if (n++ == maxlen)
1932 		break;
1933 	    if ((what & STR2NR_QUOTE) && *ptr == '\'' && VIM_ISDIGIT(ptr[1]))
1934 	    {
1935 		++ptr;
1936 		if (n++ == maxlen)
1937 		    break;
1938 	    }
1939 	}
1940     }
1941 
1942     // Check for an alphanumeric character immediately following, that is
1943     // most likely a typo.
1944     if (strict && n - 1 != maxlen && ASCII_ISALNUM(*ptr))
1945 	return;
1946 
1947     if (prep != NULL)
1948 	*prep = pre;
1949     if (len != NULL)
1950 	*len = (int)(ptr - start);
1951     if (nptr != NULL)
1952     {
1953 	if (negative)   // account for leading '-' for decimal numbers
1954 	{
1955 	    // avoid ubsan error for overflow
1956 	    if (un > VARNUM_MAX)
1957 		*nptr = VARNUM_MIN;
1958 	    else
1959 		*nptr = -(varnumber_T)un;
1960 	}
1961 	else
1962 	{
1963 	    if (un > VARNUM_MAX)
1964 		un = VARNUM_MAX;
1965 	    *nptr = (varnumber_T)un;
1966 	}
1967     }
1968     if (unptr != NULL)
1969 	*unptr = un;
1970 }
1971 
1972 /*
1973  * Return the value of a single hex character.
1974  * Only valid when the argument is '0' - '9', 'A' - 'F' or 'a' - 'f'.
1975  */
1976     int
1977 hex2nr(int c)
1978 {
1979     if (c >= 'a' && c <= 'f')
1980 	return c - 'a' + 10;
1981     if (c >= 'A' && c <= 'F')
1982 	return c - 'A' + 10;
1983     return c - '0';
1984 }
1985 
1986 #if defined(FEAT_TERMRESPONSE) || defined(FEAT_GUI_GTK) || defined(PROTO)
1987 /*
1988  * Convert two hex characters to a byte.
1989  * Return -1 if one of the characters is not hex.
1990  */
1991     int
1992 hexhex2nr(char_u *p)
1993 {
1994     if (!vim_isxdigit(p[0]) || !vim_isxdigit(p[1]))
1995 	return -1;
1996     return (hex2nr(p[0]) << 4) + hex2nr(p[1]);
1997 }
1998 #endif
1999 
2000 /*
2001  * Return TRUE if "str" starts with a backslash that should be removed.
2002  * For MS-DOS, MSWIN and OS/2 this is only done when the character after the
2003  * backslash is not a normal file name character.
2004  * '$' is a valid file name character, we don't remove the backslash before
2005  * it.  This means it is not possible to use an environment variable after a
2006  * backslash.  "C:\$VIM\doc" is taken literally, only "$VIM\doc" works.
2007  * Although "\ name" is valid, the backslash in "Program\ files" must be
2008  * removed.  Assume a file name doesn't start with a space.
2009  * For multi-byte names, never remove a backslash before a non-ascii
2010  * character, assume that all multi-byte characters are valid file name
2011  * characters.
2012  */
2013     int
2014 rem_backslash(char_u *str)
2015 {
2016 #ifdef BACKSLASH_IN_FILENAME
2017     return (str[0] == '\\'
2018 	    && str[1] < 0x80
2019 	    && (str[1] == ' '
2020 		|| (str[1] != NUL
2021 		    && str[1] != '*'
2022 		    && str[1] != '?'
2023 		    && !vim_isfilec(str[1]))));
2024 #else
2025     return (str[0] == '\\' && str[1] != NUL);
2026 #endif
2027 }
2028 
2029 /*
2030  * Halve the number of backslashes in a file name argument.
2031  * For MS-DOS we only do this if the character after the backslash
2032  * is not a normal file character.
2033  */
2034     void
2035 backslash_halve(char_u *p)
2036 {
2037     for ( ; *p; ++p)
2038 	if (rem_backslash(p))
2039 	    STRMOVE(p, p + 1);
2040 }
2041 
2042 /*
2043  * backslash_halve() plus save the result in allocated memory.
2044  * However, returns "p" when out of memory.
2045  */
2046     char_u *
2047 backslash_halve_save(char_u *p)
2048 {
2049     char_u	*res;
2050 
2051     res = vim_strsave(p);
2052     if (res == NULL)
2053 	return p;
2054     backslash_halve(res);
2055     return res;
2056 }
2057 
2058 #if (defined(EBCDIC) && defined(FEAT_POSTSCRIPT)) || defined(PROTO)
2059 /*
2060  * Table for EBCDIC to ASCII conversion unashamedly taken from xxd.c!
2061  * The first 64 entries have been added to map control characters defined in
2062  * ascii.h
2063  */
2064 static char_u ebcdic2ascii_tab[256] =
2065 {
2066     0000, 0001, 0002, 0003, 0004, 0011, 0006, 0177,
2067     0010, 0011, 0012, 0013, 0014, 0015, 0016, 0017,
2068     0020, 0021, 0022, 0023, 0024, 0012, 0010, 0027,
2069     0030, 0031, 0032, 0033, 0033, 0035, 0036, 0037,
2070     0040, 0041, 0042, 0043, 0044, 0045, 0046, 0047,
2071     0050, 0051, 0052, 0053, 0054, 0055, 0056, 0057,
2072     0060, 0061, 0062, 0063, 0064, 0065, 0066, 0067,
2073     0070, 0071, 0072, 0073, 0074, 0075, 0076, 0077,
2074     0040, 0240, 0241, 0242, 0243, 0244, 0245, 0246,
2075     0247, 0250, 0325, 0056, 0074, 0050, 0053, 0174,
2076     0046, 0251, 0252, 0253, 0254, 0255, 0256, 0257,
2077     0260, 0261, 0041, 0044, 0052, 0051, 0073, 0176,
2078     0055, 0057, 0262, 0263, 0264, 0265, 0266, 0267,
2079     0270, 0271, 0313, 0054, 0045, 0137, 0076, 0077,
2080     0272, 0273, 0274, 0275, 0276, 0277, 0300, 0301,
2081     0302, 0140, 0072, 0043, 0100, 0047, 0075, 0042,
2082     0303, 0141, 0142, 0143, 0144, 0145, 0146, 0147,
2083     0150, 0151, 0304, 0305, 0306, 0307, 0310, 0311,
2084     0312, 0152, 0153, 0154, 0155, 0156, 0157, 0160,
2085     0161, 0162, 0136, 0314, 0315, 0316, 0317, 0320,
2086     0321, 0345, 0163, 0164, 0165, 0166, 0167, 0170,
2087     0171, 0172, 0322, 0323, 0324, 0133, 0326, 0327,
2088     0330, 0331, 0332, 0333, 0334, 0335, 0336, 0337,
2089     0340, 0341, 0342, 0343, 0344, 0135, 0346, 0347,
2090     0173, 0101, 0102, 0103, 0104, 0105, 0106, 0107,
2091     0110, 0111, 0350, 0351, 0352, 0353, 0354, 0355,
2092     0175, 0112, 0113, 0114, 0115, 0116, 0117, 0120,
2093     0121, 0122, 0356, 0357, 0360, 0361, 0362, 0363,
2094     0134, 0237, 0123, 0124, 0125, 0126, 0127, 0130,
2095     0131, 0132, 0364, 0365, 0366, 0367, 0370, 0371,
2096     0060, 0061, 0062, 0063, 0064, 0065, 0066, 0067,
2097     0070, 0071, 0372, 0373, 0374, 0375, 0376, 0377
2098 };
2099 
2100 /*
2101  * Convert a buffer worth of characters from EBCDIC to ASCII.  Only useful if
2102  * wanting 7-bit ASCII characters out the other end.
2103  */
2104     void
2105 ebcdic2ascii(char_u *buffer, int len)
2106 {
2107     int		i;
2108 
2109     for (i = 0; i < len; i++)
2110 	buffer[i] = ebcdic2ascii_tab[buffer[i]];
2111 }
2112 #endif
2113