Home | History | Annotate | Line # | Download | only in src
      1 /*	$NetBSD: encoding.c,v 1.13 2026/06/10 20:54:16 christos Exp $	*/
      2 
      3 /*
      4  * Copyright (c) Ian F. Darwin 1986-1995.
      5  * Software written by Ian F. Darwin and others;
      6  * maintained 1995-present by Christos Zoulas and others.
      7  *
      8  * Redistribution and use in source and binary forms, with or without
      9  * modification, are permitted provided that the following conditions
     10  * are met:
     11  * 1. Redistributions of source code must retain the above copyright
     12  *    notice immediately at the beginning of the file, without modification,
     13  *    this list of conditions, and the following disclaimer.
     14  * 2. Redistributions in binary form must reproduce the above copyright
     15  *    notice, this list of conditions and the following disclaimer in the
     16  *    documentation and/or other materials provided with the distribution.
     17  *
     18  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
     19  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
     20  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
     21  * ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE FOR
     22  * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
     23  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
     24  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
     25  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
     26  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
     27  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
     28  * SUCH DAMAGE.
     29  */
     30 /*
     31  * Encoding -- determine the character encoding of a text file.
     32  *
     33  * Joerg Wunsch <joerg (at) freebsd.org> wrote the original support for 8-bit
     34  * international characters.
     35  */
     36 
     37 #include "file.h"
     38 
     39 #ifndef	lint
     40 #if 0
     41 FILE_RCSID("@(#)$File: encoding.c,v 1.45 2026/06/02 17:19:02 christos Exp $")
     42 #else
     43 __RCSID("$NetBSD: encoding.c,v 1.13 2026/06/10 20:54:16 christos Exp $");
     44 #endif
     45 #endif	/* lint */
     46 
     47 #include "magic.h"
     48 #include <string.h>
     49 #include <stdlib.h>
     50 
     51 
     52 file_private int looks_ascii(const unsigned char *, size_t, file_unichar_t *,
     53     size_t *);
     54 file_private int looks_utf8_with_BOM(const unsigned char *, size_t, file_unichar_t *,
     55     size_t *);
     56 file_private int looks_utf7(const unsigned char *, size_t, file_unichar_t *,
     57     size_t *);
     58 file_private int looks_ucs16(const unsigned char *, size_t, file_unichar_t *,
     59     size_t *);
     60 file_private int looks_ucs32(const unsigned char *, size_t, file_unichar_t *,
     61     size_t *);
     62 file_private int looks_latin1(const unsigned char *, size_t, file_unichar_t *,
     63     size_t *);
     64 file_private int looks_extended(const unsigned char *, size_t, file_unichar_t *,
     65     size_t *);
     66 file_private void from_ebcdic(const unsigned char *, size_t, unsigned char *);
     67 
     68 #ifdef DEBUG_ENCODING
     69 #define DPRINTF(a) printf a
     70 #else
     71 #define DPRINTF(a)
     72 #endif
     73 
     74 /*
     75  * Try to determine whether text is in some character code we can
     76  * identify.  Each of these tests, if it succeeds, will leave
     77  * the text converted into one-file_unichar_t-per-character Unicode in
     78  * ubuf, and the number of characters converted in ulen.
     79  */
     80 file_protected int
     81 file_encoding(struct magic_set *ms, const struct buffer *b,
     82     file_unichar_t **ubuf, size_t *ulen, const char **code,
     83     const char **code_mime, const char **type)
     84 {
     85 	const unsigned char *buf = CAST(const unsigned char *, b->fbuf);
     86 	size_t nbytes = b->flen;
     87 	size_t mlen;
     88 	int rv = 1, ucs_type;
     89 	file_unichar_t *udefbuf;
     90 	size_t udeflen;
     91 
     92 	if (buf == NULL)
     93 		return 0;
     94 	if (ubuf == NULL)
     95 		ubuf = &udefbuf;
     96 	if (ulen == NULL)
     97 		ulen = &udeflen;
     98 
     99 	*type = "text";
    100 	*ulen = 0;
    101 	*code = "unknown";
    102 	*code_mime = "binary";
    103 
    104 	if (nbytes > ms->encoding_max)
    105 		nbytes = ms->encoding_max;
    106 
    107 	mlen = (nbytes + 1) * sizeof((*ubuf)[0]);
    108 	*ubuf = CAST(file_unichar_t *, calloc(CAST(size_t, 1), mlen));
    109 	if (*ubuf == NULL) {
    110 		file_oomem(ms, mlen);
    111 		goto done;
    112 	}
    113 	if (looks_ascii(buf, nbytes, *ubuf, ulen)) {
    114 		if (looks_utf7(buf, nbytes, *ubuf, ulen) > 0) {
    115 			DPRINTF(("utf-7 %" SIZE_T_FORMAT "u\n", *ulen));
    116 			*code = "Unicode text, UTF-7";
    117 			*code_mime = "utf-7";
    118 		} else {
    119 			DPRINTF(("ascii %" SIZE_T_FORMAT "u\n", *ulen));
    120 			*code = "ASCII";
    121 			*code_mime = "us-ascii";
    122 		}
    123 	} else if (looks_utf8_with_BOM(buf, nbytes, *ubuf, ulen) > 0) {
    124 		DPRINTF(("utf8/bom %" SIZE_T_FORMAT "u\n", *ulen));
    125 		*code = "Unicode text, UTF-8 (with BOM)";
    126 		*code_mime = "utf-8";
    127 	} else if (file_looks_utf8(buf, nbytes, *ubuf, ulen) > 1) {
    128 		DPRINTF(("utf8 %" SIZE_T_FORMAT "u\n", *ulen));
    129 		*code = "Unicode text, UTF-8";
    130 		*code_mime = "utf-8";
    131 	} else if ((ucs_type = looks_ucs32(buf, nbytes, *ubuf, ulen)) != 0) {
    132 		if (ucs_type == 1) {
    133 			*code = "Unicode text, UTF-32, little-endian";
    134 			*code_mime = "utf-32le";
    135 		} else {
    136 			*code = "Unicode text, UTF-32, big-endian";
    137 			*code_mime = "utf-32be";
    138 		}
    139 		DPRINTF(("ucs32 %" SIZE_T_FORMAT "u\n", *ulen));
    140 	} else if ((ucs_type = looks_ucs16(buf, nbytes, *ubuf, ulen)) != 0) {
    141 		if (ucs_type == 1) {
    142 			*code = "Unicode text, UTF-16, little-endian";
    143 			*code_mime = "utf-16le";
    144 		} else {
    145 			*code = "Unicode text, UTF-16, big-endian";
    146 			*code_mime = "utf-16be";
    147 		}
    148 		DPRINTF(("ucs16 %" SIZE_T_FORMAT "u\n", *ulen));
    149 	} else if (looks_latin1(buf, nbytes, *ubuf, ulen)) {
    150 		DPRINTF(("latin1 %" SIZE_T_FORMAT "u\n", *ulen));
    151 		*code = "ISO-8859";
    152 		*code_mime = "iso-8859-1";
    153 	} else if (looks_extended(buf, nbytes, *ubuf, ulen)) {
    154 		DPRINTF(("extended %" SIZE_T_FORMAT "u\n", *ulen));
    155 		*code = "Non-ISO extended-ASCII";
    156 		*code_mime = "unknown-8bit";
    157 	} else {
    158 		unsigned char *nbuf;
    159 
    160 		mlen = (nbytes + 1) * sizeof(nbuf[0]);
    161 		if ((nbuf = CAST(unsigned char *, malloc(mlen))) == NULL) {
    162 			file_oomem(ms, mlen);
    163 			goto done;
    164 		}
    165 		from_ebcdic(buf, nbytes, nbuf);
    166 
    167 		if (looks_ascii(nbuf, nbytes, *ubuf, ulen)) {
    168 			DPRINTF(("ebcdic %" SIZE_T_FORMAT "u\n", *ulen));
    169 			*code = "EBCDIC";
    170 			*code_mime = "ebcdic";
    171 		} else if (looks_latin1(nbuf, nbytes, *ubuf, ulen)) {
    172 			DPRINTF(("ebcdic/international %" SIZE_T_FORMAT "u\n",
    173 			    *ulen));
    174 			*code = "International EBCDIC";
    175 			*code_mime = "ebcdic";
    176 		} else { /* Doesn't look like text at all */
    177 			DPRINTF(("binary\n"));
    178 			rv = 0;
    179 			*type = "binary";
    180 		}
    181 		free(nbuf);
    182 	}
    183 
    184  done:
    185 	if (ubuf == &udefbuf)
    186 		free(udefbuf);
    187 
    188 	return rv;
    189 }
    190 
    191 /*
    192  * This table reflects a particular philosophy about what constitutes
    193  * "text," and there is room for disagreement about it.
    194  *
    195  * Version 3.31 of the file command considered a file to be ASCII if
    196  * each of its characters was approved by either the isascii() or
    197  * isalpha() function.  On most systems, this would mean that any
    198  * file consisting only of characters in the range 0x00 ... 0x7F
    199  * would be called ASCII text, but many systems might reasonably
    200  * consider some characters outside this range to be alphabetic,
    201  * so the file command would call such characters ASCII.  It might
    202  * have been more accurate to call this "considered textual on the
    203  * local system" than "ASCII."
    204  *
    205  * It considered a file to be "International language text" if each
    206  * of its characters was either an ASCII printing character (according
    207  * to the real ASCII standard, not the above test), a character in
    208  * the range 0x80 ... 0xFF, or one of the following control characters:
    209  * backspace, tab, line feed, vertical tab, form feed, carriage return,
    210  * escape.  No attempt was made to determine the language in which files
    211  * of this type were written.
    212  *
    213  *
    214  * The table below considers a file to be ASCII if all of its characters
    215  * are either ASCII printing characters (again, according to the X3.4
    216  * standard, not isascii()) or any of the following controls: bell,
    217  * backspace, tab, line feed, form feed, carriage return, esc, nextline.
    218  *
    219  * I include bell because some programs (particularly shell scripts)
    220  * use it literally, even though it is rare in normal text.  I exclude
    221  * vertical tab because it never seems to be used in real text.  I also
    222  * include, with hesitation, the X3.64/ECMA-43 control nextline (0x85),
    223  * because that's what the dd EBCDIC->ASCII table maps the EBCDIC newline
    224  * character to.  It might be more appropriate to include it in the 8859
    225  * set instead of the ASCII set, but it's got to be included in *something*
    226  * we recognize or EBCDIC files aren't going to be considered textual.
    227  * Some old Unix source files use SO/SI (^N/^O) to shift between Greek
    228  * and Latin characters, so these should possibly be allowed.  But they
    229  * make a real mess on VT100-style displays if they're not paired properly,
    230  * so we are probably better off not calling them text.
    231  *
    232  * A file is considered to be ISO-8859 text if its characters are all
    233  * either ASCII, according to the above definition, or printing characters
    234  * from the ISO-8859 8-bit extension, characters 0xA0 ... 0xFF.
    235  *
    236  * Finally, a file is considered to be international text from some other
    237  * character code if its characters are all either ISO-8859 (according to
    238  * the above definition) or characters in the range 0x80 ... 0x9F, which
    239  * ISO-8859 considers to be control characters but the IBM PC and Macintosh
    240  * consider to be printing characters.
    241  */
    242 
    243 #define F 0   /* character never appears in text */
    244 #define T 1   /* character appears in plain ASCII text */
    245 #define I 2   /* character appears in ISO-8859 text */
    246 #define X 3   /* character appears in non-ISO extended ASCII (Mac, IBM PC) */
    247 
    248 /*
    249  * SUB (substitute character ^Z) was used as EOF in DOS and early Windows
    250  * NEL (next line 0x85) is considered in ECMAScript as whitespace
    251  */
    252 file_private char text_chars[256] = {
    253 	/*                  BEL BS HT LF VT FF CR    */
    254 	F, F, F, F, F, F, F, T, T, T, T, T, T, T, F, F,  /* 0x0X */
    255 	/*                           SUB ESC          */
    256 	F, F, F, F, F, F, F, F, F, F, T, T, F, F, F, F,  /* 0x1X */
    257 	T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T,  /* 0x2X */
    258 	T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T,  /* 0x3X */
    259 	T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T,  /* 0x4X */
    260 	T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T,  /* 0x5X */
    261 	T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T,  /* 0x6X */
    262 	T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, F,  /* 0x7X */
    263 	/*            NEL                            */
    264 	X, X, X, X, X, T, X, X, X, X, X, X, X, X, X, X,  /* 0x8X */
    265 	X, X, X, X, X, X, X, X, X, X, X, X, X, X, X, X,  /* 0x9X */
    266 	I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I,  /* 0xaX */
    267 	I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I,  /* 0xbX */
    268 	I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I,  /* 0xcX */
    269 	I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I,  /* 0xdX */
    270 	I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I,  /* 0xeX */
    271 	I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I   /* 0xfX */
    272 };
    273 
    274 #define LOOKS(NAME, COND) \
    275 file_private int \
    276 looks_ ## NAME(const unsigned char *buf, size_t nbytes, file_unichar_t *ubuf, \
    277     size_t *ulen) \
    278 { \
    279 	size_t i; \
    280 \
    281 	*ulen = 0; \
    282 \
    283 	for (i = 0; i < nbytes; i++) { \
    284 		int t = text_chars[buf[i]]; \
    285 \
    286 		if (COND) \
    287 			return 0; \
    288 \
    289 		ubuf[(*ulen)++] = buf[i]; \
    290 	} \
    291 	return 1; \
    292 }
    293 
    294 LOOKS(ascii, t != T)
    295 LOOKS(latin1, t != T && t != I)
    296 LOOKS(extended, t != T && t != I && t != X)
    297 
    298 /*
    299  * Decide whether some text looks like UTF-8. Returns:
    300  *
    301  *     -1: invalid UTF-8
    302  *      0: uses odd control characters, so doesn't look like text
    303  *      1: 7-bit text
    304  *      2: definitely UTF-8 text (valid high-bit set bytes)
    305  *
    306  * If ubuf is non-NULL on entry, text is decoded into ubuf, *ulen;
    307  * ubuf must be big enough!
    308  */
    309 
    310 // from: https://golang.org/src/unicode/utf8/utf8.go
    311 
    312 #define	XX 0xF1 // invalid: size 1
    313 #define	AS 0xF0 // ASCII: size 1
    314 #define	S1 0x02 // accept 0, size 2
    315 #define	S2 0x13 // accept 1, size 3
    316 #define	S3 0x03 // accept 0, size 3
    317 #define	S4 0x23 // accept 2, size 3
    318 #define	S5 0x34 // accept 3, size 4
    319 #define	S6 0x04 // accept 0, size 4
    320 #define	S7 0x44 // accept 4, size 4
    321 
    322 #define LOCB 0x80
    323 #define HICB 0xBF
    324 
    325 // first is information about the first byte in a UTF-8 sequence.
    326 static const uint8_t first[] = {
    327     //   1   2   3   4   5   6   7   8   9   A   B   C   D   E   F
    328     AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, // 0x00-0x0F
    329     AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, // 0x10-0x1F
    330     AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, // 0x20-0x2F
    331     AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, // 0x30-0x3F
    332     AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, // 0x40-0x4F
    333     AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, // 0x50-0x5F
    334     AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, // 0x60-0x6F
    335     AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, AS, // 0x70-0x7F
    336     //   1   2   3   4   5   6   7   8   9   A   B   C   D   E   F
    337     XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, // 0x80-0x8F
    338     XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, // 0x90-0x9F
    339     XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, // 0xA0-0xAF
    340     XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, // 0xB0-0xBF
    341     XX, XX, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, // 0xC0-0xCF
    342     S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, S1, // 0xD0-0xDF
    343     S2, S3, S3, S3, S3, S3, S3, S3, S3, S3, S3, S3, S3, S4, S3, S3, // 0xE0-0xEF
    344     S5, S6, S6, S6, S7, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, XX, // 0xF0-0xFF
    345 };
    346 
    347 // acceptRange gives the range of valid values for the second byte in a UTF-8
    348 // sequence.
    349 static struct accept_range {
    350 	uint8_t lo; // lowest value for second byte.
    351 	uint8_t hi; // highest value for second byte.
    352 } accept_ranges[16] = {
    353 // acceptRanges has size 16 to avoid bounds checks in the code that uses it.
    354 	{ LOCB, HICB },
    355 	{ 0xA0, HICB },
    356 	{ LOCB, 0x9F },
    357 	{ 0x90, HICB },
    358 	{ LOCB, 0x8F },
    359 };
    360 
    361 file_protected int
    362 file_looks_utf8(const unsigned char *buf, size_t nbytes, file_unichar_t *ubuf,
    363     size_t *ulen)
    364 {
    365 	size_t i;
    366 	int n;
    367 	file_unichar_t c;
    368 	int gotone = 0, ctrl = 0;
    369 
    370 	if (ubuf)
    371 		*ulen = 0;
    372 
    373 	for (i = 0; i < nbytes; i++) {
    374 		if ((buf[i] & 0x80) == 0) {	   /* 0xxxxxxx is plain ASCII */
    375 			/*
    376 			 * Even if the whole file is valid UTF-8 sequences,
    377 			 * still reject it if it uses weird control characters.
    378 			 */
    379 
    380 			if (text_chars[buf[i]] != T)
    381 				ctrl = 1;
    382 
    383 			if (ubuf)
    384 				ubuf[(*ulen)++] = buf[i];
    385 		} else if ((buf[i] & 0x40) == 0) { /* 10xxxxxx never 1st byte */
    386 			return -1;
    387 		} else {			   /* 11xxxxxx begins UTF-8 */
    388 			int following;
    389 			uint8_t x = first[buf[i]];
    390 			const struct accept_range *ar =
    391 			    &accept_ranges[(unsigned int)x >> 4];
    392 			if (x == XX)
    393 				return -1;
    394 
    395 			if ((buf[i] & 0x20) == 0) {		/* 110xxxxx */
    396 				c = buf[i] & 0x1f;
    397 				following = 1;
    398 			} else if ((buf[i] & 0x10) == 0) {	/* 1110xxxx */
    399 				c = buf[i] & 0x0f;
    400 				following = 2;
    401 			} else if ((buf[i] & 0x08) == 0) {	/* 11110xxx */
    402 				c = buf[i] & 0x07;
    403 				following = 3;
    404 			} else if ((buf[i] & 0x04) == 0) {	/* 111110xx */
    405 				c = buf[i] & 0x03;
    406 				following = 4;
    407 			} else if ((buf[i] & 0x02) == 0) {	/* 1111110x */
    408 				c = buf[i] & 0x01;
    409 				following = 5;
    410 			} else
    411 				return -1;
    412 
    413 			for (n = 0; n < following; n++) {
    414 				i++;
    415 				if (i >= nbytes)
    416 					goto done;
    417 
    418 				if (n == 0 &&
    419 				     (buf[i] < ar->lo || buf[i] > ar->hi))
    420 					return -1;
    421 
    422 				if ((buf[i] & 0x80) == 0 || (buf[i] & 0x40))
    423 					return -1;
    424 
    425 				c = (c << 6) + (buf[i] & 0x3f);
    426 			}
    427 
    428 			if (ubuf)
    429 				ubuf[(*ulen)++] = c;
    430 			gotone = 1;
    431 		}
    432 	}
    433 done:
    434 	return ctrl ? 0 : (gotone ? 2 : 1);
    435 }
    436 
    437 /*
    438  * Decide whether some text looks like UTF-8 with BOM. If there is no
    439  * BOM, return -1; otherwise return the result of looks_utf8 on the
    440  * rest of the text.
    441  */
    442 file_private int
    443 looks_utf8_with_BOM(const unsigned char *buf, size_t nbytes,
    444     file_unichar_t *ubuf, size_t *ulen)
    445 {
    446 	if (nbytes > 3 && buf[0] == 0xef && buf[1] == 0xbb && buf[2] == 0xbf)
    447 		return file_looks_utf8(buf + 3, nbytes - 3, ubuf, ulen);
    448 	else
    449 		return -1;
    450 }
    451 
    452 file_private int
    453 looks_utf7(const unsigned char *buf, size_t nbytes, file_unichar_t *ubuf,
    454     size_t *ulen)
    455 {
    456 	if (nbytes > 4 && buf[0] == '+' && buf[1] == '/' && buf[2] == 'v')
    457 		switch (buf[3]) {
    458 		case '8':
    459 		case '9':
    460 		case '+':
    461 		case '/':
    462 			if (ubuf)
    463 				*ulen = 0;
    464 			return 1;
    465 		default:
    466 			return -1;
    467 		}
    468 	else
    469 		return -1;
    470 }
    471 
    472 #define UCS16_NOCHAR(c) ((c) >= 0xfdd0 && (c) <= 0xfdef)
    473 #define UCS16_HISURR(c) ((c) >= 0xd800 && (c) <= 0xdbff)
    474 #define UCS16_LOSURR(c) ((c) >= 0xdc00 && (c) <= 0xdfff)
    475 
    476 file_private int
    477 looks_ucs16(const unsigned char *bf, size_t nbytes, file_unichar_t *ubf,
    478     size_t *ulen)
    479 {
    480 	int bigend;
    481 	uint32_t hi;
    482 	size_t i;
    483 
    484 	if (nbytes < 2)
    485 		return 0;
    486 
    487 	if (bf[0] == 0xff && bf[1] == 0xfe)
    488 		bigend = 0;
    489 	else if (bf[0] == 0xfe && bf[1] == 0xff)
    490 		bigend = 1;
    491 	else
    492 		return 0;
    493 
    494 	*ulen = 0;
    495 	hi = 0;
    496 
    497 	for (i = 2; i + 1 < nbytes; i += 2) {
    498 		uint32_t uc;
    499 
    500 		if (bigend)
    501 			uc = CAST(uint32_t,
    502 			    bf[i + 1] | (CAST(file_unichar_t, bf[i]) << 8));
    503 		else
    504 			uc = CAST(uint32_t,
    505 			    bf[i] | (CAST(file_unichar_t, bf[i + 1]) << 8));
    506 
    507 		uc &= 0xffff;
    508 
    509 		switch (uc) {
    510 		case 0xfffe:
    511 		case 0xffff:
    512 			return 0;
    513 		default:
    514 			if (UCS16_NOCHAR(uc))
    515 				return 0;
    516 			break;
    517 		}
    518 		if (hi) {
    519 			if (!UCS16_LOSURR(uc))
    520 				return 0;
    521 			uc = 0x10000 + 0x400 * (hi - 1) + (uc - 0xdc00);
    522 			hi = 0;
    523 		}
    524 		if (uc < 128 && text_chars[CAST(size_t, uc)] != T)
    525 			return 0;
    526 		ubf[(*ulen)++] = uc;
    527 		if (UCS16_HISURR(uc))
    528 			hi = uc - 0xd800 + 1;
    529 		if (UCS16_LOSURR(uc))
    530 			return 0;
    531 	}
    532 
    533 	return 1 + bigend;
    534 }
    535 
    536 file_private int
    537 looks_ucs32(const unsigned char *bf, size_t nbytes, file_unichar_t *ubf,
    538     size_t *ulen)
    539 {
    540 	int bigend;
    541 	size_t i;
    542 
    543 	if (nbytes < 4)
    544 		return 0;
    545 
    546 	if (bf[0] == 0xff && bf[1] == 0xfe && bf[2] == 0 && bf[3] == 0)
    547 		bigend = 0;
    548 	else if (bf[0] == 0 && bf[1] == 0 && bf[2] == 0xfe && bf[3] == 0xff)
    549 		bigend = 1;
    550 	else
    551 		return 0;
    552 
    553 	*ulen = 0;
    554 
    555 	for (i = 4; i + 3 < nbytes; i += 4) {
    556 		/* XXX fix to properly handle chars > 65536 */
    557 
    558 		if (bigend)
    559 			ubf[(*ulen)++] = CAST(file_unichar_t, bf[i + 3])
    560 			    | (CAST(file_unichar_t, bf[i + 2]) << 8)
    561 			    | (CAST(file_unichar_t, bf[i + 1]) << 16)
    562 			    | (CAST(file_unichar_t, bf[i]) << 24);
    563 		else
    564 			ubf[(*ulen)++] = CAST(file_unichar_t, bf[i + 0])
    565 			    | (CAST(file_unichar_t, bf[i + 1]) << 8)
    566 			    | (CAST(file_unichar_t, bf[i + 2]) << 16)
    567 			    | (CAST(file_unichar_t, bf[i + 3]) << 24);
    568 
    569 		if (ubf[*ulen - 1] == 0xfffe)
    570 			return 0;
    571 		if (ubf[*ulen - 1] < 128 &&
    572 		    text_chars[CAST(size_t, ubf[*ulen - 1])] != T)
    573 			return 0;
    574 	}
    575 
    576 	return 1 + bigend;
    577 }
    578 #undef F
    579 #undef T
    580 #undef I
    581 #undef X
    582 
    583 /*
    584  * This table maps each EBCDIC character to an (8-bit extended) ASCII
    585  * character, as specified in the rationale for the dd(1) command in
    586  * draft 11.2 (September, 1991) of the POSIX P1003.2 standard.
    587  *
    588  * Unfortunately it does not seem to correspond exactly to any of the
    589  * five variants of EBCDIC documented in IBM's _Enterprise Systems
    590  * Architecture/390: Principles of Operation_, SA22-7201-06, Seventh
    591  * Edition, July, 1999, pp. I-1 - I-4.
    592  *
    593  * Fortunately, though, all versions of EBCDIC, including this one, agree
    594  * on most of the printing characters that also appear in (7-bit) ASCII.
    595  * Of these, only '|', '!', '~', '^', '[', and ']' are in question at all.
    596  *
    597  * Fortunately too, there is general agreement that codes 0x00 through
    598  * 0x3F represent control characters, 0x41 a nonbreaking space, and the
    599  * remainder printing characters.
    600  *
    601  * This is sufficient to allow us to identify EBCDIC text and to distinguish
    602  * between old-style and internationalized examples of text.
    603  */
    604 
    605 file_private unsigned char ebcdic_to_ascii[] = {
    606   0,   1,   2,   3, 156,   9, 134, 127, 151, 141, 142,  11,  12,  13,  14,  15,
    607  16,  17,  18,  19, 157, 133,   8, 135,  24,  25, 146, 143,  28,  29,  30,  31,
    608 128, 129, 130, 131, 132,  10,  23,  27, 136, 137, 138, 139, 140,   5,   6,   7,
    609 144, 145,  22, 147, 148, 149, 150,   4, 152, 153, 154, 155,  20,  21, 158,  26,
    610 ' ', 160, 161, 162, 163, 164, 165, 166, 167, 168, 213, '.', '<', '(', '+', '|',
    611 '&', 169, 170, 171, 172, 173, 174, 175, 176, 177, '!', '$', '*', ')', ';', '~',
    612 '-', '/', 178, 179, 180, 181, 182, 183, 184, 185, 203, ',', '%', '_', '>', '?',
    613 186, 187, 188, 189, 190, 191, 192, 193, 194, '`', ':', '#', '@', '\'','=', '"',
    614 195, 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 196, 197, 198, 199, 200, 201,
    615 202, 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', '^', 204, 205, 206, 207, 208,
    616 209, 229, 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 210, 211, 212, '[', 214, 215,
    617 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227, 228, ']', 230, 231,
    618 '{', 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 232, 233, 234, 235, 236, 237,
    619 '}', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 238, 239, 240, 241, 242, 243,
    620 '\\',159, 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', 244, 245, 246, 247, 248, 249,
    621 '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 250, 251, 252, 253, 254, 255
    622 };
    623 
    624 #ifdef notdef
    625 /*
    626  * The following EBCDIC-to-ASCII table may relate more closely to reality,
    627  * or at least to modern reality.  It comes from
    628  *
    629  *   http://ftp.s390.ibm.com/products/oe/bpxqp9.html
    630  *
    631  * and maps the characters of EBCDIC code page 1047 (the code used for
    632  * Unix-derived software on IBM's 390 systems) to the corresponding
    633  * characters from ISO 8859-1.
    634  *
    635  * If this table is used instead of the above one, some of the special
    636  * cases for the NEL character can be taken out of the code.
    637  */
    638 
    639 file_private unsigned char ebcdic_1047_to_8859[] = {
    640 0x00,0x01,0x02,0x03,0x9C,0x09,0x86,0x7F,0x97,0x8D,0x8E,0x0B,0x0C,0x0D,0x0E,0x0F,
    641 0x10,0x11,0x12,0x13,0x9D,0x0A,0x08,0x87,0x18,0x19,0x92,0x8F,0x1C,0x1D,0x1E,0x1F,
    642 0x80,0x81,0x82,0x83,0x84,0x85,0x17,0x1B,0x88,0x89,0x8A,0x8B,0x8C,0x05,0x06,0x07,
    643 0x90,0x91,0x16,0x93,0x94,0x95,0x96,0x04,0x98,0x99,0x9A,0x9B,0x14,0x15,0x9E,0x1A,
    644 0x20,0xA0,0xE2,0xE4,0xE0,0xE1,0xE3,0xE5,0xE7,0xF1,0xA2,0x2E,0x3C,0x28,0x2B,0x7C,
    645 0x26,0xE9,0xEA,0xEB,0xE8,0xED,0xEE,0xEF,0xEC,0xDF,0x21,0x24,0x2A,0x29,0x3B,0x5E,
    646 0x2D,0x2F,0xC2,0xC4,0xC0,0xC1,0xC3,0xC5,0xC7,0xD1,0xA6,0x2C,0x25,0x5F,0x3E,0x3F,
    647 0xF8,0xC9,0xCA,0xCB,0xC8,0xCD,0xCE,0xCF,0xCC,0x60,0x3A,0x23,0x40,0x27,0x3D,0x22,
    648 0xD8,0x61,0x62,0x63,0x64,0x65,0x66,0x67,0x68,0x69,0xAB,0xBB,0xF0,0xFD,0xFE,0xB1,
    649 0xB0,0x6A,0x6B,0x6C,0x6D,0x6E,0x6F,0x70,0x71,0x72,0xAA,0xBA,0xE6,0xB8,0xC6,0xA4,
    650 0xB5,0x7E,0x73,0x74,0x75,0x76,0x77,0x78,0x79,0x7A,0xA1,0xBF,0xD0,0x5B,0xDE,0xAE,
    651 0xAC,0xA3,0xA5,0xB7,0xA9,0xA7,0xB6,0xBC,0xBD,0xBE,0xDD,0xA8,0xAF,0x5D,0xB4,0xD7,
    652 0x7B,0x41,0x42,0x43,0x44,0x45,0x46,0x47,0x48,0x49,0xAD,0xF4,0xF6,0xF2,0xF3,0xF5,
    653 0x7D,0x4A,0x4B,0x4C,0x4D,0x4E,0x4F,0x50,0x51,0x52,0xB9,0xFB,0xFC,0xF9,0xFA,0xFF,
    654 0x5C,0xF7,0x53,0x54,0x55,0x56,0x57,0x58,0x59,0x5A,0xB2,0xD4,0xD6,0xD2,0xD3,0xD5,
    655 0x30,0x31,0x32,0x33,0x34,0x35,0x36,0x37,0x38,0x39,0xB3,0xDB,0xDC,0xD9,0xDA,0x9F
    656 };
    657 #endif
    658 
    659 /*
    660  * Copy buf[0 ... nbytes-1] into out[], translating EBCDIC to ASCII.
    661  */
    662 file_private void
    663 from_ebcdic(const unsigned char *buf, size_t nbytes, unsigned char *out)
    664 {
    665 	size_t i;
    666 
    667 	for (i = 0; i < nbytes; i++) {
    668 		out[i] = ebcdic_to_ascii[buf[i]];
    669 	}
    670 }
    671