2 * Copyright (C) 1999-2001, 2008, 2016 Free Software Foundation, Inc.
3 * This file is part of the GNU LIBICONV Library.
5 * The GNU LIBICONV Library is free software; you can redistribute it
6 * and/or modify it under the terms of the GNU Library General Public
7 * License as published by the Free Software Foundation; either version 2
8 * of the License, or (at your option) any later version.
10 * The GNU LIBICONV Library is distributed in the hope that it will be
11 * useful, but WITHOUT ANY WARRANTY; without even the implied warranty of
12 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
13 * Library General Public License for more details.
15 * You should have received a copy of the GNU Library General Public
16 * License along with the GNU LIBICONV Library; see the file COPYING.LIB.
17 * If not, see <http://www.gnu.org/licenses/>.
24 /* Specification: RFC 1922 */
31 * The state is composed of one of the following values
34 #define STATE_TWOBYTE 1
36 * and one of the following values, << 8
39 #define STATE2_DESIGNATED_GB2312 1
40 #define STATE2_DESIGNATED_CNS11643_1 2
42 * and one of the following values, << 16
45 #define STATE3_DESIGNATED_CNS11643_2 1
48 unsigned int state1 = state & 0xff, state2 = (state >> 8) & 0xff, state3 = state >> 16
49 #define COMBINE_STATE \
50 state = (state3 << 16) | (state2 << 8) | state1
53 iso2022_cn_mbtowc (conv_t conv, ucs4_t *pwc, const unsigned char *s, size_t n)
55 state_t state = conv->istate;
67 state2 = STATE2_DESIGNATED_GB2312;
74 state2 = STATE2_DESIGNATED_CNS11643_1;
83 state3 = STATE3_DESIGNATED_CNS11643_2;
95 case STATE3_DESIGNATED_CNS11643_2:
96 if (s[2] < 0x80 && s[3] < 0x80) {
97 int ret = cns11643_2_mbtowc(conv,pwc,s+2,2);
100 if (ret != 2) abort();
102 conv->istate = state;
112 if (state2 != STATE2_DESIGNATED_GB2312 && state2 != STATE2_DESIGNATED_CNS11643_1)
114 state1 = STATE_TWOBYTE;
121 state1 = STATE_ASCII;
132 int ret = ascii_mbtowc(conv,pwc,s,1);
133 if (ret == RET_ILSEQ)
135 if (ret != 1) abort();
136 if (*pwc == 0x000a || *pwc == 0x000d) {
137 state2 = STATE2_NONE; state3 = STATE3_NONE;
140 conv->istate = state;
147 if (s[0] < 0x80 && s[1] < 0x80) {
152 case STATE2_DESIGNATED_GB2312:
153 ret = gb2312_mbtowc(conv,pwc,s,2); break;
154 case STATE2_DESIGNATED_CNS11643_1:
155 ret = cns11643_1_mbtowc(conv,pwc,s,2); break;
158 if (ret == RET_ILSEQ)
160 if (ret != 2) abort();
162 conv->istate = state;
171 conv->istate = state;
172 return RET_TOOFEW(count);
176 conv->istate = state;
177 return RET_SHIFT_ILSEQ(count);
181 iso2022_cn_wctomb (conv_t conv, unsigned char *r, ucs4_t wc, size_t n)
183 state_t state = conv->ostate;
185 unsigned char buf[3];
188 /* There is no need to handle Unicode 3.1 tag characters and to look for
189 "zh-CN" or "zh-TW" tags, because GB2312 and CNS11643 are disjoint. */
192 ret = ascii_wctomb(conv,buf,wc,1);
193 if (ret != RET_ILUNI) {
194 if (ret != 1) abort();
196 int count = (state1 == STATE_ASCII ? 1 : 2);
199 if (state1 != STATE_ASCII) {
202 state1 = STATE_ASCII;
205 if (wc == 0x000a || wc == 0x000d) {
206 state2 = STATE2_NONE; state3 = STATE3_NONE;
209 conv->ostate = state;
214 /* Try GB 2312-1980. */
215 ret = gb2312_wctomb(conv,buf,wc,2);
216 if (ret != RET_ILUNI) {
217 if (ret != 2) abort();
218 if (buf[0] < 0x80 && buf[1] < 0x80) {
219 int count = (state2 == STATE2_DESIGNATED_GB2312 ? 0 : 4) + (state1 == STATE_TWOBYTE ? 0 : 1) + 2;
222 if (state2 != STATE2_DESIGNATED_GB2312) {
228 state2 = STATE2_DESIGNATED_GB2312;
230 if (state1 != STATE_TWOBYTE) {
233 state1 = STATE_TWOBYTE;
238 conv->ostate = state;
243 ret = cns11643_wctomb(conv,buf,wc,3);
244 if (ret != RET_ILUNI) {
245 if (ret != 3) abort();
247 /* Try CNS 11643-1992 Plane 1. */
248 if (buf[0] == 1 && buf[1] < 0x80 && buf[2] < 0x80) {
249 int count = (state2 == STATE2_DESIGNATED_CNS11643_1 ? 0 : 4) + (state1 == STATE_TWOBYTE ? 0 : 1) + 2;
252 if (state2 != STATE2_DESIGNATED_CNS11643_1) {
258 state2 = STATE2_DESIGNATED_CNS11643_1;
260 if (state1 != STATE_TWOBYTE) {
263 state1 = STATE_TWOBYTE;
268 conv->ostate = state;
272 /* Try CNS 11643-1992 Plane 2. */
273 if (buf[0] == 2 && buf[1] < 0x80 && buf[2] < 0x80) {
274 int count = (state3 == STATE3_DESIGNATED_CNS11643_2 ? 0 : 4) + 4;
277 if (state3 != STATE3_DESIGNATED_CNS11643_2) {
283 state3 = STATE3_DESIGNATED_CNS11643_2;
290 conv->ostate = state;
299 iso2022_cn_reset (conv_t conv, unsigned char *r, size_t n)
301 state_t state = conv->ostate;
305 if (state1 != STATE_ASCII) {
309 /* conv->ostate = 0; will be done by the caller */
317 #undef STATE3_DESIGNATED_CNS11643_2
319 #undef STATE2_DESIGNATED_CNS11643_1
320 #undef STATE2_DESIGNATED_GB2312