771
|
1 /* Header for encoding conversion functions; coding-system object.
|
|
2 #### rename me to coding-system.h
|
428
|
3 Copyright (C) 1991, 1995 Free Software Foundation, Inc.
|
|
4 Copyright (C) 1995 Sun Microsystems, Inc.
|
793
|
5 Copyright (C) 2000, 2001, 2002 Ben Wing.
|
428
|
6
|
|
7 This file is part of XEmacs.
|
|
8
|
|
9 XEmacs is free software; you can redistribute it and/or modify it
|
|
10 under the terms of the GNU General Public License as published by the
|
|
11 Free Software Foundation; either version 2, or (at your option) any
|
|
12 later version.
|
|
13
|
|
14 XEmacs is distributed in the hope that it will be useful, but WITHOUT
|
|
15 ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
|
|
16 FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License
|
|
17 for more details.
|
|
18
|
|
19 You should have received a copy of the GNU General Public License
|
|
20 along with XEmacs; see the file COPYING. If not, write to
|
|
21 the Free Software Foundation, Inc., 59 Temple Place - Suite 330,
|
|
22 Boston, MA 02111-1307, USA. */
|
|
23
|
|
24 /* Synched up with: Mule 2.3. Not in FSF. */
|
|
25
|
771
|
26 /* Authorship:
|
|
27
|
|
28 Current primary author: Ben Wing <ben@xemacs.org>
|
|
29
|
|
30 Written by Ben Wing <ben@xemacs.org> for XEmacs, 1995, loosely based
|
|
31 on code written 91.10.09 by K.Handa <handa@etl.go.jp>.
|
|
32 Rewritten again 2000-2001 by Ben Wing to support properly
|
|
33 abstracted coding systems.
|
|
34 September 2001: Finished last part of abstraction, the detection
|
|
35 mechanism.
|
|
36 */
|
428
|
37
|
440
|
38 #ifndef INCLUDED_file_coding_h_
|
|
39 #define INCLUDED_file_coding_h_
|
428
|
40
|
771
|
41 /* Capsule description of the different structures, what their purpose is,
|
|
42 how they fit together, and where various bits of data are stored.
|
|
43
|
|
44 A "coding system" is an algorithm for converting data in one format into
|
|
45 data in another format. Currently most of the coding systems we have
|
|
46 created concern internationalized text, and convert between the XEmacs
|
|
47 internal format for multilingual text, and various external
|
|
48 representations of such text. However, any such conversion is possible,
|
|
49 for example, compressing or uncompressing text using the gzip algorithm.
|
|
50 All coding systems provide both encode and decode routines, so that the
|
|
51 conversion can go both ways.
|
|
52
|
|
53 The way we handle this is by dividing the various potential coding
|
|
54 systems into types, analogous to classes in C++. Each coding system
|
|
55 type encompasses a series of related coding systems that it can
|
|
56 implement, and it has properties which control how exactly the encoding
|
|
57 works. A particular set of values for each of the properties makes up a
|
|
58 "coding system", and specifies one particular encoding. A `struct
|
|
59 Lisp_Coding_System' object encapsulates those settings -- its type, the
|
|
60 values chosen for all properties of that type, a name for the coding
|
|
61 system, some documentation.
|
|
62
|
|
63 In addition, there are of course methods associated with a coding system
|
|
64 type, implementing the encoding, decoding, etc. These are stored in a
|
|
65 `struct coding_system_methods' object, one per coding-system type, which
|
|
66 contains mostly function pointers. This is retrievable from the
|
|
67 coding-system object (i.e. the struct Lisp_Coding_System), which has a
|
|
68 pointer to it.
|
|
69
|
|
70 In order to actually use a coding system to do an encoding or decoding
|
|
71 operation, you need to use a coding Lstream.
|
|
72
|
|
73 Now let's look more at attached data. All coding systems have certain
|
|
74 common data fields -- name, type, documentation, etc. -- as well as a
|
|
75 bunch more that are defined by the coding system type. To handle this
|
|
76 cleanly, each coding system type defines a structure that holds just the
|
|
77 fields of data particular to it, and calls it e.g. `struct
|
|
78 iso2022_coding_system' for coding system type `iso2022'. When the
|
|
79 memory block holding the coding system object is created, it is sized
|
|
80 such that it can hold both the struct Lisp_Coding_System and the struct
|
|
81 iso2022_coding_system (or whatever) directly following it. (This is a
|
|
82 common trick; another possibility is to have a void * pointer in the
|
|
83 struct Lisp_Coding_System, which points to another memory block holding
|
|
84 the struct iso2022_coding_system.) A macro is provided
|
|
85 (CODING_SYSTEM_TYPE_DATA) to retrieve a pointer of the right type to the
|
|
86 type-specific data contained within the overall `struct
|
|
87 Lisp_Coding_System' block.
|
|
88
|
|
89 Lstreams, similarly, are objects of type `struct lstream' holding data
|
|
90 about the stream operation (how much data has been read or written, any
|
|
91 buffered data, any error conditions, etc.), and like coding systems have
|
|
92 different types. They have a structure called `Lstream_implementation',
|
|
93 one per lstream type, exactly analogous to `struct
|
|
94 coding_system_methods'. In addition, they have type-specific data
|
|
95 (specifying, e.g., the file number, FILE *, memory location, other
|
|
96 lstream, etc. to read the data from or write it to, and for conversion
|
|
97 processes, the current state of the process -- are we decoding ASCII or
|
|
98 Kanji characters? are we in the middle of a processing an escape
|
|
99 sequence? etc.). This type-specific data is stored in a structure
|
|
100 named `struct coding_stream'. Just like for coding systems, the
|
|
101 type-independent data in the `struct lstream' and the type-dependent
|
|
102 data in the `struct coding_stream' are stored together in the same
|
|
103 memory block.
|
428
|
104
|
771
|
105 Now things get a bit tricky. The `struct coding_stream' is
|
|
106 type-specific from the point of view of an lstream, but not from the
|
|
107 point of view of a coding system. It contains only general data about
|
|
108 the conversion process, e.g. the name of the coding system used for
|
|
109 conversion, the lstream that we take data from or write it to (depending
|
|
110 on whether this was created as a read stream or a write stream), a
|
|
111 buffer to hold extra data we retrieved but can't send on yet, some
|
|
112 flags, etc. It also needs some data specific to the particular coding
|
|
113 system and thus to the particular operation going on. This data is held
|
|
114 in a structure named (e.g.) `struct iso2022_coding_stream', and it's
|
|
115 held in a separate memory block and pointed to by the generic `struct
|
|
116 coding_stream'. It's not glommed into a single memory block both
|
|
117 because that would require making changes to the generic lstream code
|
|
118 and more importantly because the coding system used in a particular
|
|
119 coding lstream can be changed at any point during the lifetime of the
|
|
120 lstream, and possibly multiple times. (For example, it can be set using
|
|
121 the Lisp primitives `set-process-input-coding-system' and
|
|
122 `set-console-tty-input-coding-system', as well as getting set when a
|
|
123 conversion operation was started with coding system `undecided' and the
|
|
124 correct coding system was then detected.)
|
428
|
125
|
771
|
126 IMPORTANT NOTE: There are at least two ancillary data structures
|
|
127 associated with a coding system type. (There may also be detection data;
|
|
128 see elsewhere.) It's important, when writing a coding system type, to
|
|
129 keep straight which type of data goes where. In particular, `struct
|
|
130 foo_coding_system' is attached to the coding system object itself. This
|
|
131 is a permanent object and there's only one per coding system. It's
|
|
132 created once, usually at init time, and never destroyed. So, `struct
|
|
133 foo_coding_system' should in general not contain dynamic data! (Just
|
|
134 data describing the properties of the coding system.) In particular,
|
|
135 *NO* data about any conversion in progress. There may be many
|
|
136 conversions going on simultaneously using a particular coding system,
|
|
137 and by storing conversion data in the coding system, these conversions
|
|
138 will overwrite each other's data.
|
|
139
|
|
140 Instead, use the lstream object, whose purpose is to encapsulate a
|
|
141 particular conversion and all associated data. From the lstream object,
|
|
142 you can get the struct coding_stream using something like
|
|
143
|
|
144 struct coding_stream *str = LSTREAM_TYPE_DATA (lstr, coding);
|
|
145
|
|
146 But usually this structure is already passed to you as one of the
|
|
147 parameters of the method being invoked.
|
|
148
|
|
149 From the struct coding_stream, you can retrieve the
|
|
150 coding-system-type-specific data using something like
|
|
151
|
|
152 struct foo_coding_stream *data = CODING_STREAM_TYPE_DATA (str, foo);
|
|
153
|
|
154 Then, use this structure to hold all data relevant to the particular
|
|
155 conversion being done.
|
|
156
|
|
157 Initialize this structure whenever init_coding_stream_method is called
|
|
158 (this may happen more than once), and finalize it (free resources, etc.)
|
|
159 when finalize_coding_stream_method is called.
|
|
160 */
|
|
161
|
|
162 struct coding_stream;
|
|
163 struct detection_state;
|
|
164
|
1204
|
165 extern const struct sized_memory_description coding_system_methods_description;
|
771
|
166
|
|
167 struct coding_system_methods;
|
|
168
|
|
169 enum source_sink_type
|
428
|
170 {
|
771
|
171 DECODES_CHARACTER_TO_BYTE,
|
|
172 DECODES_BYTE_TO_BYTE,
|
|
173 DECODES_BYTE_TO_CHARACTER,
|
|
174 DECODES_CHARACTER_TO_CHARACTER
|
428
|
175 };
|
|
176
|
|
177 enum eol_type
|
|
178 {
|
|
179 EOL_LF,
|
|
180 EOL_CRLF,
|
771
|
181 EOL_CR,
|
|
182 EOL_AUTODETECT,
|
428
|
183 };
|
|
184
|
|
185 struct Lisp_Coding_System
|
|
186 {
|
|
187 struct lcrecord_header header;
|
771
|
188 struct coding_system_methods *methods;
|
428
|
189
|
771
|
190 /* If true, this is an internal coding system, which will not show up in
|
|
191 coding-system-list unless a special parameter is given to it. */
|
|
192 int internal_p;
|
|
193
|
1204
|
194 #define CODING_SYSTEM_SLOT_DECLARATION
|
|
195 #define MARKED_SLOT(x) Lisp_Object x;
|
|
196 #include "coding-system-slots.h"
|
771
|
197
|
1204
|
198 /* Eol type requested by user. See comment about EOL junk in
|
|
199 coding-system-slots.h. */
|
771
|
200 enum eol_type eol_type;
|
428
|
201
|
771
|
202 /* type-specific extra data attached to a coding_system */
|
|
203 char data[1];
|
428
|
204 };
|
|
205 typedef struct Lisp_Coding_System Lisp_Coding_System;
|
|
206
|
440
|
207 DECLARE_LRECORD (coding_system, Lisp_Coding_System);
|
|
208 #define XCODING_SYSTEM(x) XRECORD (x, coding_system, Lisp_Coding_System)
|
617
|
209 #define wrap_coding_system(p) wrap_record (p, coding_system)
|
428
|
210 #define CODING_SYSTEMP(x) RECORDP (x, coding_system)
|
|
211 #define CHECK_CODING_SYSTEM(x) CHECK_RECORD (x, coding_system)
|
|
212 #define CONCHECK_CODING_SYSTEM(x) CONCHECK_RECORD (x, coding_system)
|
|
213
|
1204
|
214 enum coding_system_variant
|
|
215 {
|
|
216 no_conversion_coding_system,
|
|
217 convert_eol_coding_system,
|
|
218 undecided_coding_system,
|
|
219 chain_coding_system,
|
|
220 text_file_wrapper_coding_system,
|
|
221 internal_coding_system,
|
|
222 gzip_coding_system,
|
|
223 mswindows_multibyte_to_unicode_coding_system,
|
|
224 mswindows_multibyte_coding_system,
|
|
225 iso2022_coding_system,
|
|
226 ccl_coding_system,
|
|
227 shift_jis_coding_system,
|
|
228 big5_coding_system,
|
|
229 unicode_coding_system,
|
|
230 };
|
|
231
|
771
|
232 struct coding_system_methods
|
|
233 {
|
|
234 Lisp_Object type;
|
|
235 Lisp_Object predicate_symbol;
|
|
236
|
1204
|
237 /* Type expressed as an enum, needed for KKCC marking of the
|
|
238 type-specific lstream data; copied into the struct coding_stream. */
|
|
239
|
|
240 enum coding_system_variant enumtype;
|
|
241
|
771
|
242 /* Implementation specific methods: */
|
|
243
|
|
244 /* Init method: Initialize coding-system data. Optional. */
|
|
245 void (*init_method) (Lisp_Object coding_system);
|
|
246
|
|
247 /* Mark method: Mark any Lisp objects in the type-specific data
|
|
248 attached to the coding-system object. Optional. */
|
|
249 void (*mark_method) (Lisp_Object coding_system);
|
|
250
|
|
251 /* Print method: Print the type-specific properties of this coding
|
|
252 system, as part of `print'-ing the object. If this method is defined
|
|
253 and prints anything, it should print a space as the first thing it
|
|
254 does. Optional. */
|
|
255 void (*print_method) (Lisp_Object cs, Lisp_Object printcharfun,
|
|
256 int escapeflag);
|
|
257
|
|
258 /* Canonicalize method: Convert this coding system to another one; called
|
|
259 once, at creation time, after all properties have been parsed. The
|
|
260 returned value should be a coding system created with
|
|
261 make_internal_coding_system() (passing the existing coding system as the
|
|
262 first argument), and will become the coding system returned by
|
|
263 `make-coding-system'. Optional.
|
|
264
|
|
265 NOTE: There are *three* different uses of "canonical" or "canonicalize"
|
|
266 w.r.t. coding systems, and it's important to keep them straight.
|
|
267
|
|
268 1. The canonicalize method. Used to specify a different coding
|
|
269 system, used when doing conversions, in place of the actual coding
|
|
270 system itself. Stored in the CANONICAL field of a coding system.
|
|
271
|
|
272 2. The canonicalize-after-coding method. Used to return the encoding
|
|
273 that was "actually" used to decode some text, such that this
|
|
274 particular encoding can be used to encode the text again with the
|
|
275 expectation that the result will be the same as the original encoding.
|
|
276 Particularly important with auto-detecting coding systems.
|
|
277
|
|
278 3. From the perspective of aliases, a "canonical" coding system is one
|
|
279 that's not an alias to some other coding system, and "canonicalization"
|
|
280 is the process of traversing the alias pointers to find the canonical
|
|
281 coding system that's equivalent to the alias.
|
|
282 */
|
|
283 Lisp_Object (*canonicalize_method) (Lisp_Object coding_system);
|
|
284
|
|
285 /* Canonicalize after coding method: Convert this coding system to
|
|
286 another one, after coding (usually decoding) has finished. This is
|
|
287 meant to be used by auto-detecting coding systems, which should return
|
|
288 the actually detected coding system. Optional. */
|
|
289 Lisp_Object (*canonicalize_after_coding_method)
|
|
290 (struct coding_stream *str);
|
|
291
|
|
292 /* Convert method: Decode or encode the data in SRC of size N, writing
|
|
293 the results into the Dynarr DST. If the conversion_end_type method
|
|
294 indicates that the source is characters (as opposed to bytes), you are
|
|
295 guaranteed to get only whole characters in the data in SRC/N. STR, a
|
|
296 struct coding_stream, stores all necessary state and other info about
|
|
297 the conversion. Coding-specific state (struct TYPE_coding_stream) can
|
|
298 be retrieved from STR using CODING_STREAM_TYPE_DATA(). Return value
|
|
299 indicates the number of bytes of the *INPUT* that were converted (not
|
|
300 the number of bytes written to the Dynarr!). This can be less than
|
|
301 the total amount of input passed in; if so, the remainder is
|
|
302 considered "rejected" and will appear again at the beginning of the
|
|
303 data passed in the next time the convert method is called. When EOF
|
|
304 is returned on the other end and there's no more data, the convert
|
|
305 method will be called one last time, STR->eof set and the passed-in
|
|
306 data will consist only of any rejected data from the previous
|
|
307 call. (At this point, file handles and similar resources can be
|
|
308 closed, but do NOT arbitrarily free data structures in the
|
|
309 type-specific data, because there are operations that can be done on
|
|
310 closed streams to query the results of the processing -- specifically,
|
|
311 for coding streams, there's the canonicalize_after_coding() method.)
|
|
312 Required. */
|
|
313 Bytecount (*convert_method) (struct coding_stream *str,
|
|
314 const unsigned char *src,
|
|
315 unsigned_char_dynarr *dst, Bytecount n);
|
|
316
|
|
317 /* Coding mark method: Mark any Lisp objects in the type-specific data
|
|
318 attached to `struct coding_stream'. Optional. */
|
|
319 void (*mark_coding_stream_method) (struct coding_stream *str);
|
|
320
|
|
321 /* Init coding stream method: Initialize the type-specific data attached
|
|
322 to the coding stream (i.e. in struct TYPE_coding_stream), when the
|
|
323 coding stream is opened. The type-specific data will be zeroed out.
|
|
324 Optional. */
|
|
325 void (*init_coding_stream_method) (struct coding_stream *str);
|
|
326
|
|
327 /* Rewind coding stream method: Reset any necessary type-specific data as
|
|
328 a result of the stream being rewound. Optional. */
|
|
329 void (*rewind_coding_stream_method) (struct coding_stream *str);
|
|
330
|
|
331 /* Finalize coding stream method: Clean up the type-specific data
|
|
332 attached to the coding stream (i.e. in struct TYPE_coding_stream).
|
|
333 Happens when the Lstream is deleted using Lstream_delete() or is
|
|
334 garbage-collected. Most streams are deleted after they've been used,
|
|
335 so it's less likely (but still possible) that allocated data will
|
|
336 stick around until GC time. (File handles can also be closed when EOF
|
|
337 is signalled; but some data must stick around after this point, for
|
|
338 the benefit of canonicalize_after_coding. See the convert method.)
|
|
339 Called only once (NOT called at disksave time). Optional. */
|
|
340 void (*finalize_coding_stream_method) (struct coding_stream *str);
|
|
341
|
|
342 /* Finalize method: Clean up type-specific data (e.g. free allocated
|
|
343 data) attached to the coding system (i.e. in struct
|
|
344 TYPE_coding_system), when the coding system is about to be garbage
|
|
345 collected. (Currently not called.) Called only once (NOT called at
|
|
346 disksave time). Optional. */
|
|
347 void (*finalize_method) (Lisp_Object codesys);
|
|
348
|
|
349 /* Conversion end type method: Does this coding system encode bytes ->
|
|
350 characters, characters -> characters, bytes -> bytes, or
|
|
351 characters -> bytes?. Default is characters -> bytes. Optional. */
|
|
352 enum source_sink_type (*conversion_end_type_method) (Lisp_Object codesys);
|
|
353
|
|
354 /* Putprop method: Set the value of a type-specific property. If
|
|
355 the property name is unrecognized, return 0. If the value is disallowed
|
|
356 or erroneous, signal an error. Currently called only at creation time.
|
|
357 Optional. */
|
|
358 int (*putprop_method) (Lisp_Object codesys,
|
|
359 Lisp_Object key,
|
|
360 Lisp_Object value);
|
|
361
|
|
362 /* Getprop method: Return the value of a type-specific property. If
|
|
363 the property name is unrecognized, return Qunbound. Optional.
|
|
364 */
|
|
365 Lisp_Object (*getprop_method) (Lisp_Object coding_system,
|
|
366 Lisp_Object prop);
|
|
367
|
|
368 /* These next three are set as part of the call to
|
|
369 INITIALIZE_CODING_SYSTEM_TYPE_WITH_DATA. */
|
|
370
|
|
371 /* Description of the extra data (struct foo_coding_system) attached to a
|
1204
|
372 coding system, for pdump purposes. */
|
|
373 const struct sized_memory_description *extra_description;
|
771
|
374 /* size of struct foo_coding_system -- extra data associated with
|
|
375 the coding system */
|
|
376 int extra_data_size;
|
|
377 /* size of struct foo_coding_stream -- extra data associated with the
|
|
378 struct coding_stream, needed for each active coding process
|
|
379 using this coding system. note that we can have more than one
|
|
380 process active at once (simply by creating more than one coding
|
|
381 lstream using this coding system), so we can't store this data in
|
|
382 the coding system object. */
|
|
383 int coding_data_size;
|
|
384 };
|
|
385
|
|
386 /***** Calling a coding-system method *****/
|
|
387
|
|
388 #define RAW_CODESYSMETH(cs, m) ((cs)->methods->m##_method)
|
|
389 #define HAS_CODESYSMETH_P(cs, m) (!!RAW_CODESYSMETH (cs, m))
|
|
390 #define CODESYSMETH(cs, m, args) (((cs)->methods->m##_method) args)
|
|
391
|
|
392 /* Call a void-returning coding-system method, if it exists. */
|
|
393 #define MAYBE_CODESYSMETH(cs, m, args) do { \
|
|
394 Lisp_Coding_System *maybe_codesysmeth_cs = (cs); \
|
|
395 if (HAS_CODESYSMETH_P (maybe_codesysmeth_cs, m)) \
|
|
396 CODESYSMETH (maybe_codesysmeth_cs, m, args); \
|
|
397 } while (0)
|
|
398
|
|
399 /* Call a coding-system method, if it exists, or return GIVEN.
|
|
400 NOTE: Multiply-evaluates CS. */
|
|
401 #define CODESYSMETH_OR_GIVEN(cs, m, args, given) \
|
|
402 (HAS_CODESYSMETH_P (cs, m) ? \
|
|
403 CODESYSMETH (cs, m, args) : (given))
|
|
404
|
|
405 #define XCODESYSMETH(cs, m, args) \
|
|
406 CODESYSMETH (XCODING_SYSTEM (cs), m, args)
|
|
407 #define MAYBE_XCODESYSMETH(cs, m, args) \
|
|
408 MAYBE_CODESYSMETH (XCODING_SYSTEM (cs), m, args)
|
|
409 #define XCODESYSMETH_OR_GIVEN(cs, m, args, given) \
|
|
410 CODESYSMETH_OR_GIVEN (XCODING_SYSTEM (cs), m, args, given)
|
|
411
|
|
412
|
|
413 /***** Defining new coding-system types *****/
|
|
414
|
1204
|
415 extern const struct sized_memory_description coding_system_empty_extra_description;
|
771
|
416
|
800
|
417 #ifdef ERROR_CHECK_TYPES
|
771
|
418 #define DECLARE_CODING_SYSTEM_TYPE(type) \
|
|
419 \
|
|
420 extern struct coding_system_methods * type##_coding_system_methods; \
|
826
|
421 DECLARE_INLINE_HEADER ( \
|
|
422 struct type##_coding_system * \
|
771
|
423 error_check_##type##_coding_system_data (Lisp_Coding_System *cs) \
|
826
|
424 ) \
|
771
|
425 { \
|
|
426 assert (CODING_SYSTEM_TYPE_P (cs, type)); \
|
|
427 /* Catch accidental use of INITIALIZE_CODING_SYSTEM_TYPE in place \
|
|
428 of INITIALIZE_CODING_SYSTEM_TYPE_WITH_DATA. */ \
|
|
429 assert (cs->methods->extra_data_size > 0); \
|
|
430 return (struct type##_coding_system *) cs->data; \
|
|
431 } \
|
|
432 \
|
826
|
433 DECLARE_INLINE_HEADER ( \
|
|
434 struct type##_coding_stream * \
|
771
|
435 error_check_##type##_coding_stream_data (struct coding_stream *s) \
|
826
|
436 ) \
|
771
|
437 { \
|
|
438 assert (XCODING_SYSTEM_TYPE_P (s->codesys, type)); \
|
|
439 return (struct type##_coding_stream *) s->data; \
|
|
440 } \
|
|
441 \
|
826
|
442 DECLARE_INLINE_HEADER ( \
|
|
443 Lisp_Coding_System * \
|
771
|
444 error_check_##type##_coding_system_type (Lisp_Object obj) \
|
826
|
445 ) \
|
771
|
446 { \
|
|
447 Lisp_Coding_System *cs = XCODING_SYSTEM (obj); \
|
|
448 assert (CODING_SYSTEM_TYPE_P (cs, type)); \
|
|
449 return cs; \
|
|
450 } \
|
|
451 \
|
|
452 DECLARE_NOTHING
|
|
453 #else
|
|
454 #define DECLARE_CODING_SYSTEM_TYPE(type) \
|
|
455 extern struct coding_system_methods * type##_coding_system_methods
|
800
|
456 #endif /* ERROR_CHECK_TYPES */
|
771
|
457
|
|
458 #define DEFINE_CODING_SYSTEM_TYPE(type) \
|
|
459 struct coding_system_methods * type##_coding_system_methods
|
|
460
|
1204
|
461 #define DEFINE_CODING_SYSTEM_TYPE_WITH_DATA(type) \
|
|
462 struct coding_system_methods * type##_coding_system_methods; \
|
|
463 static const struct sized_memory_description \
|
|
464 type##_coding_system_description_0 = { \
|
|
465 sizeof (struct type##_coding_system), \
|
|
466 type##_coding_system_description \
|
|
467 }
|
|
468
|
771
|
469 #define INITIALIZE_CODING_SYSTEM_TYPE(ty, pred_sym) do { \
|
|
470 ty##_coding_system_methods = \
|
|
471 xnew_and_zero (struct coding_system_methods); \
|
|
472 ty##_coding_system_methods->type = Q##ty; \
|
|
473 ty##_coding_system_methods->extra_description = \
|
1204
|
474 &coding_system_empty_extra_description; \
|
|
475 ty##_coding_system_methods->enumtype = ty##_coding_system; \
|
771
|
476 defsymbol_nodump (&ty##_coding_system_methods->predicate_symbol, \
|
|
477 pred_sym); \
|
|
478 add_entry_to_coding_system_type_list (ty##_coding_system_methods); \
|
|
479 dump_add_root_struct_ptr (&ty##_coding_system_methods, \
|
|
480 &coding_system_methods_description); \
|
|
481 } while (0)
|
|
482
|
|
483 #define REINITIALIZE_CODING_SYSTEM_TYPE(type) do { \
|
|
484 staticpro_nodump (&type##_coding_system_methods->predicate_symbol); \
|
|
485 } while (0)
|
|
486
|
|
487 /* This assumes the existence of two structures:
|
|
488
|
|
489 struct foo_coding_system (attached to the coding system)
|
|
490 struct foo_coding_stream (per coding process, attached to the
|
|
491 struct coding_stream)
|
1204
|
492 const struct memory_description foo_coding_system_description[]
|
|
493 (data description of struct foo_coding_system)
|
771
|
494
|
1204
|
495 For an example of how to do the description, see
|
771
|
496 chain_coding_system_description.
|
|
497 */
|
|
498 #define INITIALIZE_CODING_SYSTEM_TYPE_WITH_DATA(type, pred_sym) \
|
|
499 do { \
|
|
500 INITIALIZE_CODING_SYSTEM_TYPE (type, pred_sym); \
|
|
501 type##_coding_system_methods->extra_data_size = \
|
|
502 sizeof (struct type##_coding_system); \
|
|
503 type##_coding_system_methods->extra_description = \
|
1204
|
504 &type##_coding_system_description_0; \
|
771
|
505 type##_coding_system_methods->coding_data_size = \
|
|
506 sizeof (struct type##_coding_stream); \
|
|
507 } while (0)
|
|
508
|
|
509 /* Declare that coding-system-type TYPE has method METH; used in
|
|
510 initialization routines */
|
|
511 #define CODING_SYSTEM_HAS_METHOD(type, meth) \
|
|
512 (type##_coding_system_methods->meth##_method = type##_##meth)
|
|
513
|
|
514 /***** Macros for accessing coding-system types *****/
|
|
515
|
|
516 #define CODING_SYSTEM_TYPE_P(cs, type) \
|
|
517 ((cs)->methods == type##_coding_system_methods)
|
|
518 #define XCODING_SYSTEM_TYPE_P(cs, type) \
|
|
519 CODING_SYSTEM_TYPE_P (XCODING_SYSTEM (cs), type)
|
|
520
|
800
|
521 #ifdef ERROR_CHECK_TYPES
|
771
|
522 # define CODING_SYSTEM_TYPE_DATA(cs, type) \
|
|
523 error_check_##type##_coding_system_data (cs)
|
|
524 #else
|
|
525 # define CODING_SYSTEM_TYPE_DATA(cs, type) \
|
|
526 ((struct type##_coding_system *) \
|
|
527 (cs)->data)
|
|
528 #endif
|
|
529
|
|
530 #define XCODING_SYSTEM_TYPE_DATA(cs, type) \
|
|
531 CODING_SYSTEM_TYPE_DATA (XCODING_SYSTEM_OF_TYPE (cs, type), type)
|
|
532
|
800
|
533 #ifdef ERROR_CHECK_TYPES
|
771
|
534 # define XCODING_SYSTEM_OF_TYPE(x, type) \
|
|
535 error_check_##type##_coding_system_type (x)
|
|
536 # define XSETCODING_SYSTEM_OF_TYPE(x, p, type) do \
|
|
537 { \
|
793
|
538 x = wrap_coding_system (p); \
|
|
539 assert (CODING_SYSTEM_TYPEP (XCODING_SYSTEM (x), type)); \
|
771
|
540 } while (0)
|
|
541 #else
|
|
542 # define XCODING_SYSTEM_OF_TYPE(x, type) XCODING_SYSTEM (x)
|
793
|
543 # define XSETCODING_SYSTEM_OF_TYPE(x, p, type) do \
|
|
544 { \
|
|
545 x = wrap_coding_system (p); \
|
|
546 } while (0)
|
771
|
547 #endif /* ERROR_CHECK_TYPE_CHECK */
|
|
548
|
|
549 #define CODING_SYSTEM_TYPEP(x, type) \
|
|
550 (CODING_SYSTEMP (x) && CODING_SYSTEM_TYPE_P (XCODING_SYSTEM (x), type))
|
|
551 #define CHECK_CODING_SYSTEM_OF_TYPE(x, type) do { \
|
|
552 CHECK_CODING_SYSTEM (x); \
|
|
553 if (!CODING_SYSTEM_TYPE_P (XCODING_SYSTEM (x), type)) \
|
|
554 dead_wrong_type_argument \
|
|
555 (type##_coding_system_methods->predicate_symbol, x); \
|
|
556 } while (0)
|
|
557 #define CONCHECK_CODING_SYSTEM_OF_TYPE(x, type) do { \
|
|
558 CONCHECK_CODING_SYSTEM (x); \
|
|
559 if (!(CODING_SYSTEM_TYPEP (x, type))) \
|
|
560 x = wrong_type_argument \
|
|
561 (type##_coding_system_methods->predicate_symbol, x); \
|
|
562 } while (0)
|
|
563
|
|
564 #define CODING_SYSTEM_METHODS(codesys) ((codesys)->methods)
|
428
|
565 #define CODING_SYSTEM_NAME(codesys) ((codesys)->name)
|
771
|
566 #define CODING_SYSTEM_DESCRIPTION(codesys) ((codesys)->description)
|
|
567 #define CODING_SYSTEM_TYPE(codesys) ((codesys)->methods->type)
|
428
|
568 #define CODING_SYSTEM_MNEMONIC(codesys) ((codesys)->mnemonic)
|
771
|
569 #define CODING_SYSTEM_DOCUMENTATION(codesys) ((codesys)->documentation)
|
428
|
570 #define CODING_SYSTEM_POST_READ_CONVERSION(codesys) \
|
|
571 ((codesys)->post_read_conversion)
|
|
572 #define CODING_SYSTEM_PRE_WRITE_CONVERSION(codesys) \
|
|
573 ((codesys)->pre_write_conversion)
|
|
574 #define CODING_SYSTEM_EOL_TYPE(codesys) ((codesys)->eol_type)
|
771
|
575 #define CODING_SYSTEM_EOL_LF(codesys) ((codesys)->eol[EOL_LF])
|
|
576 #define CODING_SYSTEM_EOL_CRLF(codesys) ((codesys)->eol[EOL_CRLF])
|
|
577 #define CODING_SYSTEM_EOL_CR(codesys) ((codesys)->eol[EOL_CR])
|
|
578 #define CODING_SYSTEM_TEXT_FILE_WRAPPER(codesys) ((codesys)->text_file_wrapper)
|
|
579 #define CODING_SYSTEM_AUTO_EOL_WRAPPER(codesys) ((codesys)->auto_eol_wrapper)
|
|
580 #define CODING_SYSTEM_SUBSIDIARY_PARENT(codesys) ((codesys)->subsidiary_parent)
|
|
581 #define CODING_SYSTEM_CANONICAL(codesys) ((codesys)->canonical)
|
428
|
582
|
771
|
583 #define CODING_SYSTEM_CHAIN_CHAIN(codesys) \
|
|
584 (CODING_SYSTEM_TYPE_DATA (codesys, chain)->chain)
|
|
585 #define CODING_SYSTEM_CHAIN_COUNT(codesys) \
|
|
586 (CODING_SYSTEM_TYPE_DATA (codesys, chain)->count)
|
|
587 #define CODING_SYSTEM_CHAIN_CANONICALIZE_AFTER_CODING(codesys) \
|
|
588 (CODING_SYSTEM_TYPE_DATA (codesys, chain)->canonicalize_after_coding)
|
428
|
589
|
771
|
590 #define XCODING_SYSTEM_METHODS(codesys) \
|
|
591 CODING_SYSTEM_METHODS (XCODING_SYSTEM (codesys))
|
428
|
592 #define XCODING_SYSTEM_NAME(codesys) \
|
|
593 CODING_SYSTEM_NAME (XCODING_SYSTEM (codesys))
|
771
|
594 #define XCODING_SYSTEM_DESCRIPTION(codesys) \
|
|
595 CODING_SYSTEM_DESCRIPTION (XCODING_SYSTEM (codesys))
|
428
|
596 #define XCODING_SYSTEM_TYPE(codesys) \
|
|
597 CODING_SYSTEM_TYPE (XCODING_SYSTEM (codesys))
|
|
598 #define XCODING_SYSTEM_MNEMONIC(codesys) \
|
|
599 CODING_SYSTEM_MNEMONIC (XCODING_SYSTEM (codesys))
|
771
|
600 #define XCODING_SYSTEM_DOCUMENTATION(codesys) \
|
|
601 CODING_SYSTEM_DOCUMENTATION (XCODING_SYSTEM (codesys))
|
428
|
602 #define XCODING_SYSTEM_POST_READ_CONVERSION(codesys) \
|
|
603 CODING_SYSTEM_POST_READ_CONVERSION (XCODING_SYSTEM (codesys))
|
|
604 #define XCODING_SYSTEM_PRE_WRITE_CONVERSION(codesys) \
|
|
605 CODING_SYSTEM_PRE_WRITE_CONVERSION (XCODING_SYSTEM (codesys))
|
|
606 #define XCODING_SYSTEM_EOL_TYPE(codesys) \
|
|
607 CODING_SYSTEM_EOL_TYPE (XCODING_SYSTEM (codesys))
|
|
608 #define XCODING_SYSTEM_EOL_LF(codesys) \
|
|
609 CODING_SYSTEM_EOL_LF (XCODING_SYSTEM (codesys))
|
|
610 #define XCODING_SYSTEM_EOL_CRLF(codesys) \
|
|
611 CODING_SYSTEM_EOL_CRLF (XCODING_SYSTEM (codesys))
|
|
612 #define XCODING_SYSTEM_EOL_CR(codesys) \
|
|
613 CODING_SYSTEM_EOL_CR (XCODING_SYSTEM (codesys))
|
771
|
614 #define XCODING_SYSTEM_TEXT_FILE_WRAPPER(codesys) \
|
|
615 CODING_SYSTEM_TEXT_FILE_WRAPPER (XCODING_SYSTEM (codesys))
|
|
616 #define XCODING_SYSTEM_AUTO_EOL_WRAPPER(codesys) \
|
|
617 CODING_SYSTEM_AUTO_EOL_WRAPPER (XCODING_SYSTEM (codesys))
|
|
618 #define XCODING_SYSTEM_SUBSIDIARY_PARENT(codesys) \
|
|
619 CODING_SYSTEM_SUBSIDIARY_PARENT (XCODING_SYSTEM (codesys))
|
|
620 #define XCODING_SYSTEM_CANONICAL(codesys) \
|
|
621 CODING_SYSTEM_CANONICAL (XCODING_SYSTEM (codesys))
|
428
|
622
|
771
|
623 #define XCODING_SYSTEM_CHAIN_CHAIN(codesys) \
|
|
624 CODING_SYSTEM_CHAIN_CHAIN (XCODING_SYSTEM (codesys))
|
|
625 #define XCODING_SYSTEM_CHAIN_COUNT(codesys) \
|
|
626 CODING_SYSTEM_CHAIN_COUNT (XCODING_SYSTEM (codesys))
|
|
627 #define XCODING_SYSTEM_CHAIN_CANONICALIZE_AFTER_CODING(codesys) \
|
|
628 CODING_SYSTEM_CHAIN_CANONICALIZE_AFTER_CODING (XCODING_SYSTEM (codesys))
|
428
|
629
|
771
|
630 /**************************************************/
|
|
631 /* Detection */
|
|
632 /**************************************************/
|
428
|
633
|
771
|
634 #define MAX_DETECTOR_CATEGORIES 256
|
|
635 #define MAX_DETECTORS 64
|
428
|
636
|
771
|
637 #define MAX_BYTES_PROCESSED_FOR_DETECTION 65536
|
428
|
638
|
771
|
639 struct detection_state
|
428
|
640 {
|
771
|
641 int seen_non_ascii;
|
|
642 Bytecount bytes_seen;
|
428
|
643
|
771
|
644 char categories[MAX_DETECTOR_CATEGORIES];
|
|
645 Bytecount data_offset[MAX_DETECTORS];
|
|
646 /* ... more data follows; data_offset[detector_##TYPE] points to
|
|
647 the data for that type */
|
428
|
648 };
|
|
649
|
771
|
650 #define DETECTION_STATE_DATA(st, type) \
|
|
651 ((struct type##_detector *) \
|
|
652 ((char *) (st) + (st)->data_offset[detector_##type]))
|
428
|
653
|
448
|
654 /* Distinguishable categories of encodings.
|
|
655
|
|
656 This list determines the initial priority of the categories.
|
|
657
|
|
658 For better or worse, currently Mule files are encoded in 7-bit ISO 2022.
|
|
659 For this reason, under Mule ISO_7 gets highest priority.
|
|
660
|
|
661 Putting NO_CONVERSION second prevents "binary corruption" in the
|
|
662 default case in all but the (presumably) extremely rare case of a
|
|
663 binary file which contains redundant escape sequences but no 8-bit
|
|
664 characters.
|
|
665
|
|
666 The remaining priorities are based on perceived "internationalization
|
|
667 political correctness." An exception is UCS-4 at the bottom, since
|
|
668 basically everything is compatible with UCS-4, but it is likely to
|
|
669 be very rare as an external encoding. */
|
|
670
|
771
|
671 /* Macros to define code of control characters for ISO2022's functions. */
|
|
672 /* Used by the detection routines of other coding system types as well. */
|
|
673 /* code */ /* function */
|
|
674 #define ISO_CODE_LF 0x0A /* line-feed */
|
|
675 #define ISO_CODE_CR 0x0D /* carriage-return */
|
|
676 #define ISO_CODE_SO 0x0E /* shift-out */
|
|
677 #define ISO_CODE_SI 0x0F /* shift-in */
|
|
678 #define ISO_CODE_ESC 0x1B /* escape */
|
|
679 #define ISO_CODE_DEL 0x7F /* delete */
|
|
680 #define ISO_CODE_SS2 0x8E /* single-shift-2 */
|
|
681 #define ISO_CODE_SS3 0x8F /* single-shift-3 */
|
|
682 #define ISO_CODE_CSI 0x9B /* control-sequence-introduce */
|
|
683
|
|
684 enum detection_result
|
|
685 {
|
|
686 /* Basically means a magic cookie was seen indicating this type, or
|
|
687 something similar. */
|
|
688 DET_NEAR_CERTAINTY = 4,
|
|
689 DET_HIGHEST = 4,
|
|
690 /* Characteristics seen that are unlikely to be other coding system types
|
|
691 -- e.g. ISO-2022 escape sequences, or perhaps a consistent pattern of
|
|
692 alternating zero bytes in UTF-16, along with Unicode LF or CRLF
|
|
693 sequences at regular intervals. (Zero bytes are unlikely or impossible
|
|
694 in most text encodings.) */
|
|
695 DET_QUITE_PROBABLE = 3,
|
|
696 /* Strong or medium statistical likelihood. At least some
|
|
697 characteristics seen that match what's normally found in this encoding
|
|
698 -- e.g. in Shift-JIS, a number of two-byte Japanese character
|
|
699 sequences in the right range, and nothing out of range; or in Unicode,
|
|
700 much higher statistical variance in the odd bytes than in the even
|
|
701 bytes, or vice-versa (perhaps the presence of regular EOL sequences
|
|
702 would bump this too to DET_QUITE_PROBABLE). This is quite often a
|
|
703 statistical test. */
|
|
704 DET_SOMEWHAT_LIKELY = 2,
|
|
705 /* Weak statistical likelihood. Pretty much any features at all that
|
|
706 characterize this encoding, and nothing that rules against it. */
|
|
707 DET_SLIGHTLY_LIKELY = 1,
|
|
708 /* Default state. Perhaps it indicates pure ASCII or something similarly
|
|
709 vague seen in Shift-JIS, or, exactly as the level says, it might mean
|
|
710 in a statistical-based detector that the pros and cons are balanced
|
|
711 out. This is also the lowest level that will be accepted by the
|
|
712 auto-detector without asking the user: If all available detectors
|
|
713 report lower levels for all categories with attached coding systems,
|
|
714 the user will be shown the results and explicitly prompted for action.
|
|
715 The user will also be prompted if this is the highest available level
|
|
716 and more than one detector reports the level. (See below about the
|
|
717 consequent necessity of an "ASCII" detector, which will return level 1
|
|
718 or higher for most plain text files.) */
|
|
719 DET_AS_LIKELY_AS_UNLIKELY = 0,
|
|
720 /* Some characteristics seen that are unusual for this encoding --
|
|
721 e.g. unusual control characters in a plain-text encoding, lots of
|
|
722 8-bit characters, or little statistical variance in the odd and even
|
|
723 bytes in UTF-16. */
|
|
724 DET_SOMEWHAT_UNLIKELY = -1,
|
|
725 /* This indicates that there is very little chance the data is in the
|
|
726 right format; this is probably the lowest level you can get when
|
|
727 presenting random binary data to a text file, because there are no
|
|
728 "specific sequences" you can see that would totally rule out
|
|
729 recognition. */
|
|
730 DET_QUITE_IMPROBABLE = -2,
|
|
731 /* An erroneous sequence was seen. */
|
|
732 DET_NEARLY_IMPOSSIBLE = -3,
|
985
|
733 DET_LOWEST = -3,
|
771
|
734 };
|
|
735
|
|
736 extern int coding_detector_count;
|
|
737 extern int coding_detector_category_count;
|
|
738
|
|
739 struct detector_category
|
428
|
740 {
|
771
|
741 int id;
|
|
742 Lisp_Object sym;
|
|
743 };
|
|
744
|
|
745 typedef struct
|
|
746 {
|
|
747 Dynarr_declare (struct detector_category);
|
|
748 } detector_category_dynarr;
|
|
749
|
|
750 struct detector
|
|
751 {
|
|
752 int id;
|
|
753 detector_category_dynarr *cats;
|
|
754 Bytecount data_size;
|
|
755 /* Detect method: Required. */
|
|
756 void (*detect_method) (struct detection_state *st,
|
|
757 const unsigned char *src, Bytecount n);
|
|
758 /* Finalize detection state method: Clean up any allocated data in the
|
|
759 detection state. Called only once (NOT called at disksave time).
|
|
760 Optional. */
|
|
761 void (*finalize_detection_state_method) (struct detection_state *st);
|
428
|
762 };
|
|
763
|
771
|
764 /* Lvalue for a particular detection result -- detection state ST,
|
|
765 category CAT */
|
|
766 #define DET_RESULT(st, cat) ((st)->categories[detector_category_##cat])
|
|
767 /* In state ST, set all detection results associated with detector DET to
|
|
768 RESULT. */
|
|
769 #define SET_DET_RESULTS(st, det, result) \
|
|
770 set_detection_results (st, detector_##det, result)
|
|
771
|
|
772 typedef struct
|
|
773 {
|
|
774 Dynarr_declare (struct detector);
|
|
775 } detector_dynarr;
|
|
776
|
|
777 extern detector_dynarr *all_coding_detectors;
|
|
778
|
|
779 #define DEFINE_DETECTOR_CATEGORY(detector, cat) \
|
|
780 int detector_category_##cat
|
|
781 #define DECLARE_DETECTOR_CATEGORY(detector, cat) \
|
|
782 extern int detector_category_##cat
|
|
783 #define INITIALIZE_DETECTOR_CATEGORY(detector, cat) \
|
|
784 do { \
|
|
785 struct detector_category dog; \
|
|
786 xzero (dog); \
|
|
787 detector_category_##cat = coding_detector_category_count++; \
|
|
788 dump_add_opaque_int (&detector_category_##cat); \
|
|
789 dog.id = detector_category_##cat; \
|
|
790 dog.sym = Q##cat; \
|
|
791 Dynarr_add (Dynarr_at (all_coding_detectors, detector_##detector).cats, \
|
|
792 dog); \
|
|
793 } while (0)
|
|
794
|
|
795 #define DEFINE_DETECTOR(Detector) \
|
|
796 int detector_##Detector
|
|
797 #define DECLARE_DETECTOR(Detector) \
|
|
798 extern int detector_##Detector
|
|
799 #define INITIALIZE_DETECTOR(Detector) \
|
|
800 do { \
|
|
801 struct detector det; \
|
|
802 xzero (det); \
|
|
803 detector_##Detector = coding_detector_count++; \
|
|
804 dump_add_opaque_int (&detector_##Detector); \
|
|
805 det.id = detector_##Detector; \
|
|
806 det.cats = Dynarr_new2 (detector_category_dynarr, \
|
|
807 struct detector_category); \
|
|
808 det.data_size = sizeof (struct Detector##_detector); \
|
|
809 Dynarr_add (all_coding_detectors, det); \
|
|
810 } while (0)
|
|
811 #define DETECTOR_HAS_METHOD(Detector, Meth) \
|
|
812 Dynarr_at (all_coding_detectors, detector_##Detector).Meth##_method = \
|
802
|
813 Detector##_##Meth
|
771
|
814
|
|
815
|
|
816 /**************************************************/
|
|
817 /* Decoding/Encoding */
|
|
818 /**************************************************/
|
|
819
|
|
820 /* Is the source (SOURCEP == 1) or sink (SOURCEP == 0) when encoding specified
|
|
821 in characters? */
|
|
822
|
|
823 enum source_or_sink
|
|
824 {
|
|
825 CODING_SOURCE,
|
|
826 CODING_SINK
|
|
827 };
|
|
828
|
|
829 enum encode_decode
|
|
830 {
|
|
831 CODING_ENCODE,
|
|
832 CODING_DECODE
|
|
833 };
|
|
834
|
|
835 /* Data structure attached to an lstream of type `coding',
|
|
836 containing values specific to the coding process. Additional
|
|
837 data is stored in the DATA field below; the exact form of that data
|
|
838 is controlled by the type of the coding system that governs the
|
|
839 conversion (field CODESYS). CODESYS may be set at any time
|
|
840 throughout the lifetime of the lstream and possibly more than once.
|
|
841 See long comment above for more info. */
|
|
842
|
|
843 struct coding_stream
|
|
844 {
|
1204
|
845 /* Enumerated constant listing which type of console this is (TTY, X,
|
|
846 MS-Windows, etc.). This duplicates the method structure in
|
|
847 XCODING_SYSTEM (str->codesys)->methods->type, which formerly was the
|
|
848 only way to determine the coding system type. We need this constant
|
|
849 now for KKCC, so that it can be used in an XD_UNION clause to
|
|
850 determine the Lisp objects in the type-specific data. */
|
|
851 enum coding_system_variant type;
|
|
852
|
771
|
853 /* Coding system that governs the conversion. */
|
|
854 Lisp_Object codesys;
|
|
855 /* Original coding system, pre-canonicalization. */
|
|
856 Lisp_Object orig_codesys;
|
|
857
|
|
858 /* Back pointer to current stream. */
|
|
859 Lstream *us;
|
|
860
|
|
861 /* Stream that we read the unprocessed data from or write the processed
|
|
862 data to. */
|
|
863 Lstream *other_end;
|
|
864
|
|
865 /* In order to handle both reading to and writing from a coding stream,
|
|
866 we phrase the conversion methods like write methods -- we can
|
|
867 implement reading in terms of a write method but not vice-versa,
|
|
868 because the write method is forced to take only what it's given but
|
|
869 the read method can read more data from the other end if necessary.
|
|
870 On the other hand, the write method is free to generate all the data
|
|
871 it wants (and just write it to the other end), but the the read method
|
|
872 can return only as much as was asked for, so we need to implement our
|
|
873 own buffering. */
|
|
874
|
|
875 /* If we are reading, then we can return only a fixed amount of data, but
|
|
876 the converter is free to return as much as it wants, so we direct it
|
|
877 to store the data here and lop off chunks as we need them. If we are
|
|
878 writing, we use this because the converter takes a Dynarr but we are
|
|
879 supposed to write into a fixed buffer. (NOTE: This introduces an extra
|
|
880 memory copy.) */
|
|
881 unsigned_char_dynarr *convert_to;
|
|
882
|
|
883 /* The conversion method might reject some of the data -- this typically
|
|
884 includes partial characters, partial escape sequences, etc. When
|
|
885 writing, we just pass the rejection up to the Lstream module, and it
|
|
886 will buffer the data. When reading, however, we need to do the
|
|
887 buffering ourselves, and we put it here, combined with newly read
|
|
888 data. */
|
|
889 unsigned_char_dynarr *convert_from;
|
|
890
|
|
891 /* If set, this is the last chunk of data being processed. When this is
|
|
892 finished, output any necessary terminating control characters, escape
|
|
893 sequences, etc. */
|
|
894 unsigned int eof:1;
|
|
895
|
|
896 /* CH holds a partially built-up character. This is really part of the
|
|
897 state-dependent data and should be moved there. */
|
|
898 unsigned int ch;
|
|
899
|
|
900 /* Coding-system-specific data holding extra state about the
|
|
901 conversion. Logically a struct TYPE_coding_stream; a pointer
|
800
|
902 to such a struct, with (when ERROR_CHECK_TYPES is defined)
|
771
|
903 error-checking that this is really a structure of that type
|
|
904 (checking the corresponding coding system type) can be retrieved using
|
|
905 CODING_STREAM_TYPE_DATA(). Allocated at the same time that
|
|
906 CODESYS is set (which may occur at any time, even multiple times,
|
|
907 during the lifetime of the stream). The size comes from
|
|
908 methods->coding_data_size. */
|
|
909 void *data;
|
|
910
|
|
911 enum encode_decode direction;
|
|
912
|
800
|
913 /* If set, don't close the stream at the other end when being closed. */
|
|
914 unsigned int no_close_other:1;
|
802
|
915 /* If set, read only one byte at a time from other end to avoid any
|
|
916 possible blocking. */
|
|
917 unsigned int one_byte_at_a_time:1;
|
814
|
918 /* If set, and we're a read stream, we init char mode on ourselves as
|
|
919 necessary to prevent the caller from getting partial characters. (the
|
|
920 default) */
|
|
921 unsigned int set_char_mode_on_us_when_reading:1;
|
800
|
922
|
771
|
923 /* #### Temporary test */
|
|
924 unsigned int finalized:1;
|
|
925 };
|
|
926
|
|
927 #define CODING_STREAM_DATA(stream) LSTREAM_TYPE_DATA (stream, coding)
|
|
928
|
800
|
929 #ifdef ERROR_CHECK_TYPES
|
771
|
930 # define CODING_STREAM_TYPE_DATA(s, type) \
|
|
931 error_check_##type##_coding_stream_data (s)
|
|
932 #else
|
|
933 # define CODING_STREAM_TYPE_DATA(s, type) \
|
|
934 ((struct type##_coding_stream *) (s)->data)
|
|
935 #endif
|
|
936
|
|
937 /* C should be a binary character in the range 0 - 255; convert
|
|
938 to internal format and add to Dynarr DST. */
|
|
939
|
428
|
940 #ifdef MULE
|
771
|
941
|
|
942 #define DECODE_ADD_BINARY_CHAR(c, dst) \
|
|
943 do { \
|
826
|
944 if (byte_ascii_p (c)) \
|
771
|
945 Dynarr_add (dst, c); \
|
826
|
946 else if (byte_c1_p (c)) \
|
771
|
947 { \
|
|
948 Dynarr_add (dst, LEADING_BYTE_CONTROL_1); \
|
|
949 Dynarr_add (dst, c + 0x20); \
|
|
950 } \
|
|
951 else \
|
|
952 { \
|
|
953 Dynarr_add (dst, LEADING_BYTE_LATIN_ISO8859_1); \
|
|
954 Dynarr_add (dst, c); \
|
|
955 } \
|
|
956 } while (0)
|
|
957
|
|
958 #else /* not MULE */
|
|
959
|
|
960 #define DECODE_ADD_BINARY_CHAR(c, dst) \
|
|
961 do { \
|
|
962 Dynarr_add (dst, c); \
|
|
963 } while (0)
|
|
964
|
|
965 #endif /* MULE */
|
|
966
|
|
967 #define DECODE_OUTPUT_PARTIAL_CHAR(ch, dst) \
|
|
968 do { \
|
|
969 if (ch) \
|
|
970 { \
|
|
971 DECODE_ADD_BINARY_CHAR (ch, dst); \
|
|
972 ch = 0; \
|
|
973 } \
|
|
974 } while (0)
|
428
|
975
|
|
976 #ifdef MULE
|
|
977 /* Convert shift-JIS code (sj1, sj2) into internal string
|
|
978 representation (c1, c2). (The leading byte is assumed.) */
|
|
979
|
771
|
980 #define DECODE_SHIFT_JIS(sj1, sj2, c1, c2) \
|
428
|
981 do { \
|
|
982 int I1 = sj1, I2 = sj2; \
|
|
983 if (I2 >= 0x9f) \
|
|
984 c1 = (I1 << 1) - ((I1 >= 0xe0) ? 0xe0 : 0x60), \
|
|
985 c2 = I2 + 2; \
|
|
986 else \
|
|
987 c1 = (I1 << 1) - ((I1 >= 0xe0) ? 0xe1 : 0x61), \
|
|
988 c2 = I2 + ((I2 >= 0x7f) ? 0x60 : 0x61); \
|
|
989 } while (0)
|
|
990
|
|
991 /* Convert the internal string representation of a Shift-JIS character
|
|
992 (c1, c2) into Shift-JIS code (sj1, sj2). The leading byte is
|
|
993 assumed. */
|
|
994
|
771
|
995 #define ENCODE_SHIFT_JIS(c1, c2, sj1, sj2) \
|
428
|
996 do { \
|
|
997 int I1 = c1, I2 = c2; \
|
|
998 if (I1 & 1) \
|
|
999 sj1 = (I1 >> 1) + ((I1 < 0xdf) ? 0x31 : 0x71), \
|
|
1000 sj2 = I2 - ((I2 >= 0xe0) ? 0x60 : 0x61); \
|
|
1001 else \
|
|
1002 sj1 = (I1 >> 1) + ((I1 < 0xdf) ? 0x30 : 0x70), \
|
|
1003 sj2 = I2 - 2; \
|
|
1004 } while (0)
|
|
1005 #endif /* MULE */
|
|
1006
|
771
|
1007 DECLARE_CODING_SYSTEM_TYPE (no_conversion);
|
|
1008 DECLARE_CODING_SYSTEM_TYPE (convert_eol);
|
|
1009 #if 0
|
|
1010 DECLARE_CODING_SYSTEM_TYPE (text_file_wrapper);
|
|
1011 #endif /* 0 */
|
|
1012 DECLARE_CODING_SYSTEM_TYPE (undecided);
|
|
1013 DECLARE_CODING_SYSTEM_TYPE (chain);
|
|
1014
|
|
1015 #ifdef DEBUG_XEMACS
|
|
1016 DECLARE_CODING_SYSTEM_TYPE (internal);
|
|
1017 #endif
|
|
1018
|
|
1019 #ifdef MULE
|
|
1020 DECLARE_CODING_SYSTEM_TYPE (iso2022);
|
|
1021 DECLARE_CODING_SYSTEM_TYPE (ccl);
|
|
1022 DECLARE_CODING_SYSTEM_TYPE (shift_jis);
|
|
1023 DECLARE_CODING_SYSTEM_TYPE (big5);
|
|
1024 #endif
|
|
1025
|
|
1026 #ifdef HAVE_ZLIB
|
|
1027 DECLARE_CODING_SYSTEM_TYPE (gzip);
|
|
1028 #endif
|
428
|
1029
|
771
|
1030 DECLARE_CODING_SYSTEM_TYPE (unicode);
|
428
|
1031
|
771
|
1032 #ifdef HAVE_WIN32_CODING_SYSTEMS
|
|
1033 DECLARE_CODING_SYSTEM_TYPE (mswindows_multibyte_to_unicode);
|
|
1034 DECLARE_CODING_SYSTEM_TYPE (mswindows_multibyte);
|
428
|
1035 #endif
|
771
|
1036
|
|
1037 Lisp_Object coding_stream_detected_coding_system (Lstream *stream);
|
|
1038 Lisp_Object coding_stream_coding_system (Lstream *stream);
|
|
1039 void set_coding_stream_coding_system (Lstream *stream,
|
|
1040 Lisp_Object codesys);
|
|
1041 Lisp_Object detect_coding_stream (Lisp_Object stream);
|
867
|
1042 Ichar decode_big5_char (int o1, int o2);
|
771
|
1043 void add_entry_to_coding_system_type_list (struct coding_system_methods *m);
|
|
1044 Lisp_Object make_internal_coding_system (Lisp_Object existing,
|
|
1045 Char_ASCII *prefix,
|
|
1046 Lisp_Object type,
|
|
1047 Lisp_Object description,
|
|
1048 Lisp_Object props);
|
802
|
1049
|
814
|
1050 #define LSTREAM_FL_NO_CLOSE_OTHER (1 << 16)
|
|
1051 #define LSTREAM_FL_READ_ONE_BYTE_AT_A_TIME (1 << 17)
|
|
1052 #define LSTREAM_FL_NO_INIT_CHAR_MODE_WHEN_READING (1 << 18)
|
|
1053
|
771
|
1054 Lisp_Object make_coding_input_stream (Lstream *stream, Lisp_Object codesys,
|
800
|
1055 enum encode_decode direction,
|
802
|
1056 int flags);
|
771
|
1057 Lisp_Object make_coding_output_stream (Lstream *stream, Lisp_Object codesys,
|
800
|
1058 enum encode_decode direction,
|
802
|
1059 int flags);
|
771
|
1060 void set_detection_results (struct detection_state *st, int detector,
|
|
1061 int given);
|
428
|
1062
|
440
|
1063 #endif /* INCLUDED_file_coding_h_ */
|
|
1064
|