771
|
1 /* Header for encoding conversion functions; coding-system object.
|
|
2 #### rename me to coding-system.h
|
428
|
3 Copyright (C) 1991, 1995 Free Software Foundation, Inc.
|
|
4 Copyright (C) 1995 Sun Microsystems, Inc.
|
793
|
5 Copyright (C) 2000, 2001, 2002 Ben Wing.
|
428
|
6
|
|
7 This file is part of XEmacs.
|
|
8
|
|
9 XEmacs is free software; you can redistribute it and/or modify it
|
|
10 under the terms of the GNU General Public License as published by the
|
|
11 Free Software Foundation; either version 2, or (at your option) any
|
|
12 later version.
|
|
13
|
|
14 XEmacs is distributed in the hope that it will be useful, but WITHOUT
|
|
15 ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
|
|
16 FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License
|
|
17 for more details.
|
|
18
|
|
19 You should have received a copy of the GNU General Public License
|
|
20 along with XEmacs; see the file COPYING. If not, write to
|
|
21 the Free Software Foundation, Inc., 59 Temple Place - Suite 330,
|
|
22 Boston, MA 02111-1307, USA. */
|
|
23
|
|
24 /* Synched up with: Mule 2.3. Not in FSF. */
|
|
25
|
771
|
26 /* Authorship:
|
|
27
|
|
28 Current primary author: Ben Wing <ben@xemacs.org>
|
|
29
|
|
30 Written by Ben Wing <ben@xemacs.org> for XEmacs, 1995, loosely based
|
|
31 on code written 91.10.09 by K.Handa <handa@etl.go.jp>.
|
|
32 Rewritten again 2000-2001 by Ben Wing to support properly
|
|
33 abstracted coding systems.
|
|
34 September 2001: Finished last part of abstraction, the detection
|
|
35 mechanism.
|
|
36 */
|
428
|
37
|
440
|
38 #ifndef INCLUDED_file_coding_h_
|
|
39 #define INCLUDED_file_coding_h_
|
428
|
40
|
771
|
41 /* Capsule description of the different structures, what their purpose is,
|
|
42 how they fit together, and where various bits of data are stored.
|
|
43
|
2297
|
44 A "coding system" is an algorithm for converting stream data in one format
|
|
45 into stream data in another format. Currently most of the coding systems
|
|
46 we have created concern internationalized text, and convert between the
|
|
47 XEmacs internal format for multilingual text, and various external
|
771
|
48 representations of such text. However, any such conversion is possible,
|
|
49 for example, compressing or uncompressing text using the gzip algorithm.
|
|
50 All coding systems provide both encode and decode routines, so that the
|
2297
|
51 conversion can go both ways. Unfortunately encoding and decoding may not
|
|
52 be exact inverses, even for a specific instance of a coding system. Care
|
|
53 must be taken when this is not the case.
|
771
|
54
|
|
55 The way we handle this is by dividing the various potential coding
|
|
56 systems into types, analogous to classes in C++. Each coding system
|
|
57 type encompasses a series of related coding systems that it can
|
|
58 implement, and it has properties which control how exactly the encoding
|
|
59 works. A particular set of values for each of the properties makes up a
|
|
60 "coding system", and specifies one particular encoding. A `struct
|
|
61 Lisp_Coding_System' object encapsulates those settings -- its type, the
|
|
62 values chosen for all properties of that type, a name for the coding
|
|
63 system, some documentation.
|
|
64
|
|
65 In addition, there are of course methods associated with a coding system
|
|
66 type, implementing the encoding, decoding, etc. These are stored in a
|
|
67 `struct coding_system_methods' object, one per coding-system type, which
|
|
68 contains mostly function pointers. This is retrievable from the
|
|
69 coding-system object (i.e. the struct Lisp_Coding_System), which has a
|
|
70 pointer to it.
|
|
71
|
|
72 In order to actually use a coding system to do an encoding or decoding
|
|
73 operation, you need to use a coding Lstream.
|
|
74
|
|
75 Now let's look more at attached data. All coding systems have certain
|
|
76 common data fields -- name, type, documentation, etc. -- as well as a
|
|
77 bunch more that are defined by the coding system type. To handle this
|
|
78 cleanly, each coding system type defines a structure that holds just the
|
|
79 fields of data particular to it, and calls it e.g. `struct
|
|
80 iso2022_coding_system' for coding system type `iso2022'. When the
|
|
81 memory block holding the coding system object is created, it is sized
|
|
82 such that it can hold both the struct Lisp_Coding_System and the struct
|
|
83 iso2022_coding_system (or whatever) directly following it. (This is a
|
|
84 common trick; another possibility is to have a void * pointer in the
|
|
85 struct Lisp_Coding_System, which points to another memory block holding
|
|
86 the struct iso2022_coding_system.) A macro is provided
|
|
87 (CODING_SYSTEM_TYPE_DATA) to retrieve a pointer of the right type to the
|
|
88 type-specific data contained within the overall `struct
|
|
89 Lisp_Coding_System' block.
|
|
90
|
|
91 Lstreams, similarly, are objects of type `struct lstream' holding data
|
|
92 about the stream operation (how much data has been read or written, any
|
|
93 buffered data, any error conditions, etc.), and like coding systems have
|
|
94 different types. They have a structure called `Lstream_implementation',
|
|
95 one per lstream type, exactly analogous to `struct
|
|
96 coding_system_methods'. In addition, they have type-specific data
|
|
97 (specifying, e.g., the file number, FILE *, memory location, other
|
|
98 lstream, etc. to read the data from or write it to, and for conversion
|
|
99 processes, the current state of the process -- are we decoding ASCII or
|
|
100 Kanji characters? are we in the middle of a processing an escape
|
|
101 sequence? etc.). This type-specific data is stored in a structure
|
|
102 named `struct coding_stream'. Just like for coding systems, the
|
|
103 type-independent data in the `struct lstream' and the type-dependent
|
|
104 data in the `struct coding_stream' are stored together in the same
|
|
105 memory block.
|
428
|
106
|
771
|
107 Now things get a bit tricky. The `struct coding_stream' is
|
|
108 type-specific from the point of view of an lstream, but not from the
|
|
109 point of view of a coding system. It contains only general data about
|
|
110 the conversion process, e.g. the name of the coding system used for
|
|
111 conversion, the lstream that we take data from or write it to (depending
|
|
112 on whether this was created as a read stream or a write stream), a
|
|
113 buffer to hold extra data we retrieved but can't send on yet, some
|
|
114 flags, etc. It also needs some data specific to the particular coding
|
|
115 system and thus to the particular operation going on. This data is held
|
|
116 in a structure named (e.g.) `struct iso2022_coding_stream', and it's
|
|
117 held in a separate memory block and pointed to by the generic `struct
|
|
118 coding_stream'. It's not glommed into a single memory block both
|
|
119 because that would require making changes to the generic lstream code
|
|
120 and more importantly because the coding system used in a particular
|
|
121 coding lstream can be changed at any point during the lifetime of the
|
|
122 lstream, and possibly multiple times. (For example, it can be set using
|
|
123 the Lisp primitives `set-process-input-coding-system' and
|
|
124 `set-console-tty-input-coding-system', as well as getting set when a
|
|
125 conversion operation was started with coding system `undecided' and the
|
2297
|
126 correct coding system was then detected.) #### This suggests implementing
|
|
127 compound text extended segments by saving the state of the ctext stream,
|
|
128 and installing an appropriate for the duration of the segment.
|
428
|
129
|
771
|
130 IMPORTANT NOTE: There are at least two ancillary data structures
|
|
131 associated with a coding system type. (There may also be detection data;
|
|
132 see elsewhere.) It's important, when writing a coding system type, to
|
|
133 keep straight which type of data goes where. In particular, `struct
|
|
134 foo_coding_system' is attached to the coding system object itself. This
|
|
135 is a permanent object and there's only one per coding system. It's
|
|
136 created once, usually at init time, and never destroyed. So, `struct
|
|
137 foo_coding_system' should in general not contain dynamic data! (Just
|
|
138 data describing the properties of the coding system.) In particular,
|
|
139 *NO* data about any conversion in progress. There may be many
|
|
140 conversions going on simultaneously using a particular coding system,
|
|
141 and by storing conversion data in the coding system, these conversions
|
|
142 will overwrite each other's data.
|
|
143
|
|
144 Instead, use the lstream object, whose purpose is to encapsulate a
|
|
145 particular conversion and all associated data. From the lstream object,
|
|
146 you can get the struct coding_stream using something like
|
|
147
|
|
148 struct coding_stream *str = LSTREAM_TYPE_DATA (lstr, coding);
|
|
149
|
|
150 But usually this structure is already passed to you as one of the
|
|
151 parameters of the method being invoked.
|
|
152
|
|
153 From the struct coding_stream, you can retrieve the
|
|
154 coding-system-type-specific data using something like
|
|
155
|
|
156 struct foo_coding_stream *data = CODING_STREAM_TYPE_DATA (str, foo);
|
|
157
|
|
158 Then, use this structure to hold all data relevant to the particular
|
|
159 conversion being done.
|
|
160
|
|
161 Initialize this structure whenever init_coding_stream_method is called
|
|
162 (this may happen more than once), and finalize it (free resources, etc.)
|
|
163 when finalize_coding_stream_method is called.
|
|
164 */
|
|
165
|
|
166 struct coding_stream;
|
|
167 struct detection_state;
|
|
168
|
1204
|
169 extern const struct sized_memory_description coding_system_methods_description;
|
771
|
170
|
|
171 struct coding_system_methods;
|
|
172
|
|
173 enum source_sink_type
|
428
|
174 {
|
771
|
175 DECODES_CHARACTER_TO_BYTE,
|
|
176 DECODES_BYTE_TO_BYTE,
|
|
177 DECODES_BYTE_TO_CHARACTER,
|
|
178 DECODES_CHARACTER_TO_CHARACTER
|
428
|
179 };
|
|
180
|
|
181 enum eol_type
|
|
182 {
|
|
183 EOL_LF,
|
|
184 EOL_CRLF,
|
771
|
185 EOL_CR,
|
1429
|
186 EOL_AUTODETECT
|
428
|
187 };
|
|
188
|
|
189 struct Lisp_Coding_System
|
|
190 {
|
2720
|
191 #ifdef MC_ALLOC
|
|
192 struct lrecord_header header;
|
|
193 #else /* MC_ALLOC */
|
428
|
194 struct lcrecord_header header;
|
2720
|
195 #endif /* MC_ALLOC */
|
771
|
196 struct coding_system_methods *methods;
|
428
|
197
|
1204
|
198 #define CODING_SYSTEM_SLOT_DECLARATION
|
|
199 #define MARKED_SLOT(x) Lisp_Object x;
|
|
200 #include "coding-system-slots.h"
|
771
|
201
|
1204
|
202 /* Eol type requested by user. See comment about EOL junk in
|
|
203 coding-system-slots.h. */
|
771
|
204 enum eol_type eol_type;
|
428
|
205
|
2132
|
206 /* If true, this is an internal coding system, which will not show up in
|
|
207 coding-system-list unless a special parameter is given to it. */
|
|
208 int internal_p;
|
|
209
|
771
|
210 /* type-specific extra data attached to a coding_system */
|
|
211 char data[1];
|
428
|
212 };
|
|
213 typedef struct Lisp_Coding_System Lisp_Coding_System;
|
|
214
|
440
|
215 DECLARE_LRECORD (coding_system, Lisp_Coding_System);
|
|
216 #define XCODING_SYSTEM(x) XRECORD (x, coding_system, Lisp_Coding_System)
|
617
|
217 #define wrap_coding_system(p) wrap_record (p, coding_system)
|
428
|
218 #define CODING_SYSTEMP(x) RECORDP (x, coding_system)
|
|
219 #define CHECK_CODING_SYSTEM(x) CHECK_RECORD (x, coding_system)
|
|
220 #define CONCHECK_CODING_SYSTEM(x) CONCHECK_RECORD (x, coding_system)
|
|
221
|
1204
|
222 enum coding_system_variant
|
|
223 {
|
|
224 no_conversion_coding_system,
|
|
225 convert_eol_coding_system,
|
|
226 undecided_coding_system,
|
|
227 chain_coding_system,
|
|
228 text_file_wrapper_coding_system,
|
|
229 internal_coding_system,
|
|
230 gzip_coding_system,
|
|
231 mswindows_multibyte_to_unicode_coding_system,
|
|
232 mswindows_multibyte_coding_system,
|
|
233 iso2022_coding_system,
|
|
234 ccl_coding_system,
|
|
235 shift_jis_coding_system,
|
|
236 big5_coding_system,
|
1429
|
237 unicode_coding_system
|
1204
|
238 };
|
|
239
|
771
|
240 struct coding_system_methods
|
|
241 {
|
|
242 Lisp_Object type;
|
|
243 Lisp_Object predicate_symbol;
|
|
244
|
1204
|
245 /* Type expressed as an enum, needed for KKCC marking of the
|
|
246 type-specific lstream data; copied into the struct coding_stream. */
|
|
247
|
|
248 enum coding_system_variant enumtype;
|
|
249
|
771
|
250 /* Implementation specific methods: */
|
|
251
|
|
252 /* Init method: Initialize coding-system data. Optional. */
|
|
253 void (*init_method) (Lisp_Object coding_system);
|
|
254
|
|
255 /* Mark method: Mark any Lisp objects in the type-specific data
|
|
256 attached to the coding-system object. Optional. */
|
|
257 void (*mark_method) (Lisp_Object coding_system);
|
|
258
|
|
259 /* Print method: Print the type-specific properties of this coding
|
|
260 system, as part of `print'-ing the object. If this method is defined
|
|
261 and prints anything, it should print a space as the first thing it
|
|
262 does. Optional. */
|
|
263 void (*print_method) (Lisp_Object cs, Lisp_Object printcharfun,
|
|
264 int escapeflag);
|
|
265
|
|
266 /* Canonicalize method: Convert this coding system to another one; called
|
|
267 once, at creation time, after all properties have been parsed. The
|
|
268 returned value should be a coding system created with
|
|
269 make_internal_coding_system() (passing the existing coding system as the
|
|
270 first argument), and will become the coding system returned by
|
|
271 `make-coding-system'. Optional.
|
|
272
|
|
273 NOTE: There are *three* different uses of "canonical" or "canonicalize"
|
|
274 w.r.t. coding systems, and it's important to keep them straight.
|
|
275
|
|
276 1. The canonicalize method. Used to specify a different coding
|
|
277 system, used when doing conversions, in place of the actual coding
|
|
278 system itself. Stored in the CANONICAL field of a coding system.
|
|
279
|
|
280 2. The canonicalize-after-coding method. Used to return the encoding
|
|
281 that was "actually" used to decode some text, such that this
|
|
282 particular encoding can be used to encode the text again with the
|
|
283 expectation that the result will be the same as the original encoding.
|
|
284 Particularly important with auto-detecting coding systems.
|
|
285
|
|
286 3. From the perspective of aliases, a "canonical" coding system is one
|
|
287 that's not an alias to some other coding system, and "canonicalization"
|
|
288 is the process of traversing the alias pointers to find the canonical
|
|
289 coding system that's equivalent to the alias.
|
|
290 */
|
|
291 Lisp_Object (*canonicalize_method) (Lisp_Object coding_system);
|
|
292
|
|
293 /* Canonicalize after coding method: Convert this coding system to
|
|
294 another one, after coding (usually decoding) has finished. This is
|
|
295 meant to be used by auto-detecting coding systems, which should return
|
|
296 the actually detected coding system. Optional. */
|
|
297 Lisp_Object (*canonicalize_after_coding_method)
|
|
298 (struct coding_stream *str);
|
|
299
|
|
300 /* Convert method: Decode or encode the data in SRC of size N, writing
|
|
301 the results into the Dynarr DST. If the conversion_end_type method
|
|
302 indicates that the source is characters (as opposed to bytes), you are
|
|
303 guaranteed to get only whole characters in the data in SRC/N. STR, a
|
|
304 struct coding_stream, stores all necessary state and other info about
|
|
305 the conversion. Coding-specific state (struct TYPE_coding_stream) can
|
|
306 be retrieved from STR using CODING_STREAM_TYPE_DATA(). Return value
|
|
307 indicates the number of bytes of the *INPUT* that were converted (not
|
|
308 the number of bytes written to the Dynarr!). This can be less than
|
|
309 the total amount of input passed in; if so, the remainder is
|
|
310 considered "rejected" and will appear again at the beginning of the
|
|
311 data passed in the next time the convert method is called. When EOF
|
|
312 is returned on the other end and there's no more data, the convert
|
|
313 method will be called one last time, STR->eof set and the passed-in
|
|
314 data will consist only of any rejected data from the previous
|
|
315 call. (At this point, file handles and similar resources can be
|
|
316 closed, but do NOT arbitrarily free data structures in the
|
|
317 type-specific data, because there are operations that can be done on
|
|
318 closed streams to query the results of the processing -- specifically,
|
|
319 for coding streams, there's the canonicalize_after_coding() method.)
|
|
320 Required. */
|
|
321 Bytecount (*convert_method) (struct coding_stream *str,
|
|
322 const unsigned char *src,
|
|
323 unsigned_char_dynarr *dst, Bytecount n);
|
|
324
|
|
325 /* Coding mark method: Mark any Lisp objects in the type-specific data
|
|
326 attached to `struct coding_stream'. Optional. */
|
|
327 void (*mark_coding_stream_method) (struct coding_stream *str);
|
|
328
|
|
329 /* Init coding stream method: Initialize the type-specific data attached
|
|
330 to the coding stream (i.e. in struct TYPE_coding_stream), when the
|
|
331 coding stream is opened. The type-specific data will be zeroed out.
|
|
332 Optional. */
|
|
333 void (*init_coding_stream_method) (struct coding_stream *str);
|
|
334
|
|
335 /* Rewind coding stream method: Reset any necessary type-specific data as
|
|
336 a result of the stream being rewound. Optional. */
|
|
337 void (*rewind_coding_stream_method) (struct coding_stream *str);
|
|
338
|
|
339 /* Finalize coding stream method: Clean up the type-specific data
|
|
340 attached to the coding stream (i.e. in struct TYPE_coding_stream).
|
|
341 Happens when the Lstream is deleted using Lstream_delete() or is
|
|
342 garbage-collected. Most streams are deleted after they've been used,
|
|
343 so it's less likely (but still possible) that allocated data will
|
|
344 stick around until GC time. (File handles can also be closed when EOF
|
|
345 is signalled; but some data must stick around after this point, for
|
|
346 the benefit of canonicalize_after_coding. See the convert method.)
|
|
347 Called only once (NOT called at disksave time). Optional. */
|
|
348 void (*finalize_coding_stream_method) (struct coding_stream *str);
|
|
349
|
|
350 /* Finalize method: Clean up type-specific data (e.g. free allocated
|
|
351 data) attached to the coding system (i.e. in struct
|
|
352 TYPE_coding_system), when the coding system is about to be garbage
|
|
353 collected. (Currently not called.) Called only once (NOT called at
|
|
354 disksave time). Optional. */
|
|
355 void (*finalize_method) (Lisp_Object codesys);
|
|
356
|
|
357 /* Conversion end type method: Does this coding system encode bytes ->
|
|
358 characters, characters -> characters, bytes -> bytes, or
|
|
359 characters -> bytes?. Default is characters -> bytes. Optional. */
|
|
360 enum source_sink_type (*conversion_end_type_method) (Lisp_Object codesys);
|
|
361
|
|
362 /* Putprop method: Set the value of a type-specific property. If
|
|
363 the property name is unrecognized, return 0. If the value is disallowed
|
|
364 or erroneous, signal an error. Currently called only at creation time.
|
|
365 Optional. */
|
|
366 int (*putprop_method) (Lisp_Object codesys,
|
|
367 Lisp_Object key,
|
|
368 Lisp_Object value);
|
|
369
|
|
370 /* Getprop method: Return the value of a type-specific property. If
|
|
371 the property name is unrecognized, return Qunbound. Optional.
|
|
372 */
|
|
373 Lisp_Object (*getprop_method) (Lisp_Object coding_system,
|
|
374 Lisp_Object prop);
|
|
375
|
|
376 /* These next three are set as part of the call to
|
|
377 INITIALIZE_CODING_SYSTEM_TYPE_WITH_DATA. */
|
|
378
|
|
379 /* Description of the extra data (struct foo_coding_system) attached to a
|
1204
|
380 coding system, for pdump purposes. */
|
|
381 const struct sized_memory_description *extra_description;
|
771
|
382 /* size of struct foo_coding_system -- extra data associated with
|
|
383 the coding system */
|
|
384 int extra_data_size;
|
|
385 /* size of struct foo_coding_stream -- extra data associated with the
|
|
386 struct coding_stream, needed for each active coding process
|
|
387 using this coding system. note that we can have more than one
|
|
388 process active at once (simply by creating more than one coding
|
|
389 lstream using this coding system), so we can't store this data in
|
|
390 the coding system object. */
|
|
391 int coding_data_size;
|
|
392 };
|
|
393
|
|
394 /***** Calling a coding-system method *****/
|
|
395
|
|
396 #define RAW_CODESYSMETH(cs, m) ((cs)->methods->m##_method)
|
|
397 #define HAS_CODESYSMETH_P(cs, m) (!!RAW_CODESYSMETH (cs, m))
|
|
398 #define CODESYSMETH(cs, m, args) (((cs)->methods->m##_method) args)
|
|
399
|
|
400 /* Call a void-returning coding-system method, if it exists. */
|
|
401 #define MAYBE_CODESYSMETH(cs, m, args) do { \
|
|
402 Lisp_Coding_System *maybe_codesysmeth_cs = (cs); \
|
|
403 if (HAS_CODESYSMETH_P (maybe_codesysmeth_cs, m)) \
|
|
404 CODESYSMETH (maybe_codesysmeth_cs, m, args); \
|
|
405 } while (0)
|
|
406
|
|
407 /* Call a coding-system method, if it exists, or return GIVEN.
|
|
408 NOTE: Multiply-evaluates CS. */
|
|
409 #define CODESYSMETH_OR_GIVEN(cs, m, args, given) \
|
|
410 (HAS_CODESYSMETH_P (cs, m) ? \
|
|
411 CODESYSMETH (cs, m, args) : (given))
|
|
412
|
|
413 #define XCODESYSMETH(cs, m, args) \
|
|
414 CODESYSMETH (XCODING_SYSTEM (cs), m, args)
|
|
415 #define MAYBE_XCODESYSMETH(cs, m, args) \
|
|
416 MAYBE_CODESYSMETH (XCODING_SYSTEM (cs), m, args)
|
|
417 #define XCODESYSMETH_OR_GIVEN(cs, m, args, given) \
|
|
418 CODESYSMETH_OR_GIVEN (XCODING_SYSTEM (cs), m, args, given)
|
|
419
|
|
420
|
|
421 /***** Defining new coding-system types *****/
|
|
422
|
1204
|
423 extern const struct sized_memory_description coding_system_empty_extra_description;
|
771
|
424
|
800
|
425 #ifdef ERROR_CHECK_TYPES
|
771
|
426 #define DECLARE_CODING_SYSTEM_TYPE(type) \
|
|
427 \
|
|
428 extern struct coding_system_methods * type##_coding_system_methods; \
|
826
|
429 DECLARE_INLINE_HEADER ( \
|
|
430 struct type##_coding_system * \
|
771
|
431 error_check_##type##_coding_system_data (Lisp_Coding_System *cs) \
|
826
|
432 ) \
|
771
|
433 { \
|
|
434 assert (CODING_SYSTEM_TYPE_P (cs, type)); \
|
|
435 /* Catch accidental use of INITIALIZE_CODING_SYSTEM_TYPE in place \
|
|
436 of INITIALIZE_CODING_SYSTEM_TYPE_WITH_DATA. */ \
|
|
437 assert (cs->methods->extra_data_size > 0); \
|
|
438 return (struct type##_coding_system *) cs->data; \
|
|
439 } \
|
|
440 \
|
826
|
441 DECLARE_INLINE_HEADER ( \
|
|
442 struct type##_coding_stream * \
|
771
|
443 error_check_##type##_coding_stream_data (struct coding_stream *s) \
|
826
|
444 ) \
|
771
|
445 { \
|
|
446 assert (XCODING_SYSTEM_TYPE_P (s->codesys, type)); \
|
|
447 return (struct type##_coding_stream *) s->data; \
|
|
448 } \
|
|
449 \
|
826
|
450 DECLARE_INLINE_HEADER ( \
|
|
451 Lisp_Coding_System * \
|
771
|
452 error_check_##type##_coding_system_type (Lisp_Object obj) \
|
826
|
453 ) \
|
771
|
454 { \
|
|
455 Lisp_Coding_System *cs = XCODING_SYSTEM (obj); \
|
|
456 assert (CODING_SYSTEM_TYPE_P (cs, type)); \
|
|
457 return cs; \
|
|
458 } \
|
|
459 \
|
|
460 DECLARE_NOTHING
|
|
461 #else
|
|
462 #define DECLARE_CODING_SYSTEM_TYPE(type) \
|
|
463 extern struct coding_system_methods * type##_coding_system_methods
|
800
|
464 #endif /* ERROR_CHECK_TYPES */
|
771
|
465
|
|
466 #define DEFINE_CODING_SYSTEM_TYPE(type) \
|
|
467 struct coding_system_methods * type##_coding_system_methods
|
|
468
|
1204
|
469 #define DEFINE_CODING_SYSTEM_TYPE_WITH_DATA(type) \
|
|
470 struct coding_system_methods * type##_coding_system_methods; \
|
|
471 static const struct sized_memory_description \
|
|
472 type##_coding_system_description_0 = { \
|
|
473 sizeof (struct type##_coding_system), \
|
|
474 type##_coding_system_description \
|
|
475 }
|
|
476
|
771
|
477 #define INITIALIZE_CODING_SYSTEM_TYPE(ty, pred_sym) do { \
|
|
478 ty##_coding_system_methods = \
|
|
479 xnew_and_zero (struct coding_system_methods); \
|
|
480 ty##_coding_system_methods->type = Q##ty; \
|
|
481 ty##_coding_system_methods->extra_description = \
|
1204
|
482 &coding_system_empty_extra_description; \
|
|
483 ty##_coding_system_methods->enumtype = ty##_coding_system; \
|
771
|
484 defsymbol_nodump (&ty##_coding_system_methods->predicate_symbol, \
|
|
485 pred_sym); \
|
|
486 add_entry_to_coding_system_type_list (ty##_coding_system_methods); \
|
2367
|
487 dump_add_root_block_ptr (&ty##_coding_system_methods, \
|
771
|
488 &coding_system_methods_description); \
|
|
489 } while (0)
|
|
490
|
|
491 #define REINITIALIZE_CODING_SYSTEM_TYPE(type) do { \
|
|
492 staticpro_nodump (&type##_coding_system_methods->predicate_symbol); \
|
|
493 } while (0)
|
|
494
|
|
495 /* This assumes the existence of two structures:
|
|
496
|
|
497 struct foo_coding_system (attached to the coding system)
|
|
498 struct foo_coding_stream (per coding process, attached to the
|
|
499 struct coding_stream)
|
1204
|
500 const struct memory_description foo_coding_system_description[]
|
|
501 (data description of struct foo_coding_system)
|
771
|
502
|
1204
|
503 For an example of how to do the description, see
|
771
|
504 chain_coding_system_description.
|
|
505 */
|
|
506 #define INITIALIZE_CODING_SYSTEM_TYPE_WITH_DATA(type, pred_sym) \
|
|
507 do { \
|
|
508 INITIALIZE_CODING_SYSTEM_TYPE (type, pred_sym); \
|
|
509 type##_coding_system_methods->extra_data_size = \
|
|
510 sizeof (struct type##_coding_system); \
|
|
511 type##_coding_system_methods->extra_description = \
|
1204
|
512 &type##_coding_system_description_0; \
|
771
|
513 type##_coding_system_methods->coding_data_size = \
|
|
514 sizeof (struct type##_coding_stream); \
|
|
515 } while (0)
|
|
516
|
|
517 /* Declare that coding-system-type TYPE has method METH; used in
|
|
518 initialization routines */
|
|
519 #define CODING_SYSTEM_HAS_METHOD(type, meth) \
|
|
520 (type##_coding_system_methods->meth##_method = type##_##meth)
|
|
521
|
|
522 /***** Macros for accessing coding-system types *****/
|
|
523
|
|
524 #define CODING_SYSTEM_TYPE_P(cs, type) \
|
|
525 ((cs)->methods == type##_coding_system_methods)
|
|
526 #define XCODING_SYSTEM_TYPE_P(cs, type) \
|
|
527 CODING_SYSTEM_TYPE_P (XCODING_SYSTEM (cs), type)
|
|
528
|
800
|
529 #ifdef ERROR_CHECK_TYPES
|
771
|
530 # define CODING_SYSTEM_TYPE_DATA(cs, type) \
|
|
531 error_check_##type##_coding_system_data (cs)
|
|
532 #else
|
|
533 # define CODING_SYSTEM_TYPE_DATA(cs, type) \
|
|
534 ((struct type##_coding_system *) \
|
|
535 (cs)->data)
|
|
536 #endif
|
|
537
|
|
538 #define XCODING_SYSTEM_TYPE_DATA(cs, type) \
|
|
539 CODING_SYSTEM_TYPE_DATA (XCODING_SYSTEM_OF_TYPE (cs, type), type)
|
|
540
|
800
|
541 #ifdef ERROR_CHECK_TYPES
|
771
|
542 # define XCODING_SYSTEM_OF_TYPE(x, type) \
|
|
543 error_check_##type##_coding_system_type (x)
|
|
544 # define XSETCODING_SYSTEM_OF_TYPE(x, p, type) do \
|
|
545 { \
|
793
|
546 x = wrap_coding_system (p); \
|
|
547 assert (CODING_SYSTEM_TYPEP (XCODING_SYSTEM (x), type)); \
|
771
|
548 } while (0)
|
|
549 #else
|
|
550 # define XCODING_SYSTEM_OF_TYPE(x, type) XCODING_SYSTEM (x)
|
793
|
551 # define XSETCODING_SYSTEM_OF_TYPE(x, p, type) do \
|
|
552 { \
|
|
553 x = wrap_coding_system (p); \
|
|
554 } while (0)
|
771
|
555 #endif /* ERROR_CHECK_TYPE_CHECK */
|
|
556
|
|
557 #define CODING_SYSTEM_TYPEP(x, type) \
|
|
558 (CODING_SYSTEMP (x) && CODING_SYSTEM_TYPE_P (XCODING_SYSTEM (x), type))
|
|
559 #define CHECK_CODING_SYSTEM_OF_TYPE(x, type) do { \
|
|
560 CHECK_CODING_SYSTEM (x); \
|
|
561 if (!CODING_SYSTEM_TYPE_P (XCODING_SYSTEM (x), type)) \
|
|
562 dead_wrong_type_argument \
|
|
563 (type##_coding_system_methods->predicate_symbol, x); \
|
|
564 } while (0)
|
|
565 #define CONCHECK_CODING_SYSTEM_OF_TYPE(x, type) do { \
|
|
566 CONCHECK_CODING_SYSTEM (x); \
|
|
567 if (!(CODING_SYSTEM_TYPEP (x, type))) \
|
|
568 x = wrong_type_argument \
|
|
569 (type##_coding_system_methods->predicate_symbol, x); \
|
|
570 } while (0)
|
|
571
|
|
572 #define CODING_SYSTEM_METHODS(codesys) ((codesys)->methods)
|
428
|
573 #define CODING_SYSTEM_NAME(codesys) ((codesys)->name)
|
771
|
574 #define CODING_SYSTEM_DESCRIPTION(codesys) ((codesys)->description)
|
|
575 #define CODING_SYSTEM_TYPE(codesys) ((codesys)->methods->type)
|
428
|
576 #define CODING_SYSTEM_MNEMONIC(codesys) ((codesys)->mnemonic)
|
771
|
577 #define CODING_SYSTEM_DOCUMENTATION(codesys) ((codesys)->documentation)
|
428
|
578 #define CODING_SYSTEM_POST_READ_CONVERSION(codesys) \
|
|
579 ((codesys)->post_read_conversion)
|
|
580 #define CODING_SYSTEM_PRE_WRITE_CONVERSION(codesys) \
|
|
581 ((codesys)->pre_write_conversion)
|
|
582 #define CODING_SYSTEM_EOL_TYPE(codesys) ((codesys)->eol_type)
|
771
|
583 #define CODING_SYSTEM_EOL_LF(codesys) ((codesys)->eol[EOL_LF])
|
|
584 #define CODING_SYSTEM_EOL_CRLF(codesys) ((codesys)->eol[EOL_CRLF])
|
|
585 #define CODING_SYSTEM_EOL_CR(codesys) ((codesys)->eol[EOL_CR])
|
|
586 #define CODING_SYSTEM_TEXT_FILE_WRAPPER(codesys) ((codesys)->text_file_wrapper)
|
|
587 #define CODING_SYSTEM_AUTO_EOL_WRAPPER(codesys) ((codesys)->auto_eol_wrapper)
|
|
588 #define CODING_SYSTEM_SUBSIDIARY_PARENT(codesys) ((codesys)->subsidiary_parent)
|
|
589 #define CODING_SYSTEM_CANONICAL(codesys) ((codesys)->canonical)
|
428
|
590
|
771
|
591 #define CODING_SYSTEM_CHAIN_CHAIN(codesys) \
|
|
592 (CODING_SYSTEM_TYPE_DATA (codesys, chain)->chain)
|
|
593 #define CODING_SYSTEM_CHAIN_COUNT(codesys) \
|
|
594 (CODING_SYSTEM_TYPE_DATA (codesys, chain)->count)
|
|
595 #define CODING_SYSTEM_CHAIN_CANONICALIZE_AFTER_CODING(codesys) \
|
|
596 (CODING_SYSTEM_TYPE_DATA (codesys, chain)->canonicalize_after_coding)
|
428
|
597
|
771
|
598 #define XCODING_SYSTEM_METHODS(codesys) \
|
|
599 CODING_SYSTEM_METHODS (XCODING_SYSTEM (codesys))
|
428
|
600 #define XCODING_SYSTEM_NAME(codesys) \
|
|
601 CODING_SYSTEM_NAME (XCODING_SYSTEM (codesys))
|
771
|
602 #define XCODING_SYSTEM_DESCRIPTION(codesys) \
|
|
603 CODING_SYSTEM_DESCRIPTION (XCODING_SYSTEM (codesys))
|
428
|
604 #define XCODING_SYSTEM_TYPE(codesys) \
|
|
605 CODING_SYSTEM_TYPE (XCODING_SYSTEM (codesys))
|
|
606 #define XCODING_SYSTEM_MNEMONIC(codesys) \
|
|
607 CODING_SYSTEM_MNEMONIC (XCODING_SYSTEM (codesys))
|
771
|
608 #define XCODING_SYSTEM_DOCUMENTATION(codesys) \
|
|
609 CODING_SYSTEM_DOCUMENTATION (XCODING_SYSTEM (codesys))
|
428
|
610 #define XCODING_SYSTEM_POST_READ_CONVERSION(codesys) \
|
|
611 CODING_SYSTEM_POST_READ_CONVERSION (XCODING_SYSTEM (codesys))
|
|
612 #define XCODING_SYSTEM_PRE_WRITE_CONVERSION(codesys) \
|
|
613 CODING_SYSTEM_PRE_WRITE_CONVERSION (XCODING_SYSTEM (codesys))
|
|
614 #define XCODING_SYSTEM_EOL_TYPE(codesys) \
|
|
615 CODING_SYSTEM_EOL_TYPE (XCODING_SYSTEM (codesys))
|
|
616 #define XCODING_SYSTEM_EOL_LF(codesys) \
|
|
617 CODING_SYSTEM_EOL_LF (XCODING_SYSTEM (codesys))
|
|
618 #define XCODING_SYSTEM_EOL_CRLF(codesys) \
|
|
619 CODING_SYSTEM_EOL_CRLF (XCODING_SYSTEM (codesys))
|
|
620 #define XCODING_SYSTEM_EOL_CR(codesys) \
|
|
621 CODING_SYSTEM_EOL_CR (XCODING_SYSTEM (codesys))
|
771
|
622 #define XCODING_SYSTEM_TEXT_FILE_WRAPPER(codesys) \
|
|
623 CODING_SYSTEM_TEXT_FILE_WRAPPER (XCODING_SYSTEM (codesys))
|
|
624 #define XCODING_SYSTEM_AUTO_EOL_WRAPPER(codesys) \
|
|
625 CODING_SYSTEM_AUTO_EOL_WRAPPER (XCODING_SYSTEM (codesys))
|
|
626 #define XCODING_SYSTEM_SUBSIDIARY_PARENT(codesys) \
|
|
627 CODING_SYSTEM_SUBSIDIARY_PARENT (XCODING_SYSTEM (codesys))
|
|
628 #define XCODING_SYSTEM_CANONICAL(codesys) \
|
|
629 CODING_SYSTEM_CANONICAL (XCODING_SYSTEM (codesys))
|
428
|
630
|
771
|
631 #define XCODING_SYSTEM_CHAIN_CHAIN(codesys) \
|
|
632 CODING_SYSTEM_CHAIN_CHAIN (XCODING_SYSTEM (codesys))
|
|
633 #define XCODING_SYSTEM_CHAIN_COUNT(codesys) \
|
|
634 CODING_SYSTEM_CHAIN_COUNT (XCODING_SYSTEM (codesys))
|
|
635 #define XCODING_SYSTEM_CHAIN_CANONICALIZE_AFTER_CODING(codesys) \
|
|
636 CODING_SYSTEM_CHAIN_CANONICALIZE_AFTER_CODING (XCODING_SYSTEM (codesys))
|
428
|
637
|
771
|
638 /**************************************************/
|
|
639 /* Detection */
|
|
640 /**************************************************/
|
428
|
641
|
771
|
642 #define MAX_DETECTOR_CATEGORIES 256
|
|
643 #define MAX_DETECTORS 64
|
428
|
644
|
771
|
645 #define MAX_BYTES_PROCESSED_FOR_DETECTION 65536
|
428
|
646
|
771
|
647 struct detection_state
|
428
|
648 {
|
771
|
649 int seen_non_ascii;
|
|
650 Bytecount bytes_seen;
|
428
|
651
|
771
|
652 char categories[MAX_DETECTOR_CATEGORIES];
|
|
653 Bytecount data_offset[MAX_DETECTORS];
|
|
654 /* ... more data follows; data_offset[detector_##TYPE] points to
|
|
655 the data for that type */
|
428
|
656 };
|
|
657
|
771
|
658 #define DETECTION_STATE_DATA(st, type) \
|
|
659 ((struct type##_detector *) \
|
|
660 ((char *) (st) + (st)->data_offset[detector_##type]))
|
428
|
661
|
448
|
662 /* Distinguishable categories of encodings.
|
|
663
|
|
664 This list determines the initial priority of the categories.
|
|
665
|
|
666 For better or worse, currently Mule files are encoded in 7-bit ISO 2022.
|
|
667 For this reason, under Mule ISO_7 gets highest priority.
|
|
668
|
|
669 Putting NO_CONVERSION second prevents "binary corruption" in the
|
|
670 default case in all but the (presumably) extremely rare case of a
|
|
671 binary file which contains redundant escape sequences but no 8-bit
|
|
672 characters.
|
|
673
|
|
674 The remaining priorities are based on perceived "internationalization
|
|
675 political correctness." An exception is UCS-4 at the bottom, since
|
|
676 basically everything is compatible with UCS-4, but it is likely to
|
|
677 be very rare as an external encoding. */
|
|
678
|
771
|
679 /* Macros to define code of control characters for ISO2022's functions. */
|
|
680 /* Used by the detection routines of other coding system types as well. */
|
|
681 /* code */ /* function */
|
|
682 #define ISO_CODE_LF 0x0A /* line-feed */
|
|
683 #define ISO_CODE_CR 0x0D /* carriage-return */
|
|
684 #define ISO_CODE_SO 0x0E /* shift-out */
|
|
685 #define ISO_CODE_SI 0x0F /* shift-in */
|
|
686 #define ISO_CODE_ESC 0x1B /* escape */
|
|
687 #define ISO_CODE_DEL 0x7F /* delete */
|
|
688 #define ISO_CODE_SS2 0x8E /* single-shift-2 */
|
|
689 #define ISO_CODE_SS3 0x8F /* single-shift-3 */
|
|
690 #define ISO_CODE_CSI 0x9B /* control-sequence-introduce */
|
|
691
|
|
692 enum detection_result
|
|
693 {
|
|
694 /* Basically means a magic cookie was seen indicating this type, or
|
|
695 something similar. */
|
|
696 DET_NEAR_CERTAINTY = 4,
|
|
697 DET_HIGHEST = 4,
|
|
698 /* Characteristics seen that are unlikely to be other coding system types
|
|
699 -- e.g. ISO-2022 escape sequences, or perhaps a consistent pattern of
|
|
700 alternating zero bytes in UTF-16, along with Unicode LF or CRLF
|
|
701 sequences at regular intervals. (Zero bytes are unlikely or impossible
|
|
702 in most text encodings.) */
|
|
703 DET_QUITE_PROBABLE = 3,
|
|
704 /* Strong or medium statistical likelihood. At least some
|
|
705 characteristics seen that match what's normally found in this encoding
|
|
706 -- e.g. in Shift-JIS, a number of two-byte Japanese character
|
|
707 sequences in the right range, and nothing out of range; or in Unicode,
|
|
708 much higher statistical variance in the odd bytes than in the even
|
|
709 bytes, or vice-versa (perhaps the presence of regular EOL sequences
|
|
710 would bump this too to DET_QUITE_PROBABLE). This is quite often a
|
|
711 statistical test. */
|
|
712 DET_SOMEWHAT_LIKELY = 2,
|
|
713 /* Weak statistical likelihood. Pretty much any features at all that
|
|
714 characterize this encoding, and nothing that rules against it. */
|
|
715 DET_SLIGHTLY_LIKELY = 1,
|
|
716 /* Default state. Perhaps it indicates pure ASCII or something similarly
|
|
717 vague seen in Shift-JIS, or, exactly as the level says, it might mean
|
|
718 in a statistical-based detector that the pros and cons are balanced
|
|
719 out. This is also the lowest level that will be accepted by the
|
|
720 auto-detector without asking the user: If all available detectors
|
|
721 report lower levels for all categories with attached coding systems,
|
|
722 the user will be shown the results and explicitly prompted for action.
|
|
723 The user will also be prompted if this is the highest available level
|
|
724 and more than one detector reports the level. (See below about the
|
|
725 consequent necessity of an "ASCII" detector, which will return level 1
|
|
726 or higher for most plain text files.) */
|
|
727 DET_AS_LIKELY_AS_UNLIKELY = 0,
|
|
728 /* Some characteristics seen that are unusual for this encoding --
|
|
729 e.g. unusual control characters in a plain-text encoding, lots of
|
|
730 8-bit characters, or little statistical variance in the odd and even
|
|
731 bytes in UTF-16. */
|
|
732 DET_SOMEWHAT_UNLIKELY = -1,
|
|
733 /* This indicates that there is very little chance the data is in the
|
|
734 right format; this is probably the lowest level you can get when
|
|
735 presenting random binary data to a text file, because there are no
|
|
736 "specific sequences" you can see that would totally rule out
|
|
737 recognition. */
|
|
738 DET_QUITE_IMPROBABLE = -2,
|
|
739 /* An erroneous sequence was seen. */
|
|
740 DET_NEARLY_IMPOSSIBLE = -3,
|
1429
|
741 DET_LOWEST = -3
|
771
|
742 };
|
|
743
|
|
744 extern int coding_detector_count;
|
|
745 extern int coding_detector_category_count;
|
|
746
|
|
747 struct detector_category
|
428
|
748 {
|
771
|
749 int id;
|
|
750 Lisp_Object sym;
|
|
751 };
|
|
752
|
|
753 typedef struct
|
|
754 {
|
|
755 Dynarr_declare (struct detector_category);
|
|
756 } detector_category_dynarr;
|
|
757
|
|
758 struct detector
|
|
759 {
|
|
760 int id;
|
|
761 detector_category_dynarr *cats;
|
|
762 Bytecount data_size;
|
|
763 /* Detect method: Required. */
|
|
764 void (*detect_method) (struct detection_state *st,
|
|
765 const unsigned char *src, Bytecount n);
|
|
766 /* Finalize detection state method: Clean up any allocated data in the
|
|
767 detection state. Called only once (NOT called at disksave time).
|
|
768 Optional. */
|
|
769 void (*finalize_detection_state_method) (struct detection_state *st);
|
428
|
770 };
|
|
771
|
771
|
772 /* Lvalue for a particular detection result -- detection state ST,
|
|
773 category CAT */
|
|
774 #define DET_RESULT(st, cat) ((st)->categories[detector_category_##cat])
|
|
775 /* In state ST, set all detection results associated with detector DET to
|
|
776 RESULT. */
|
|
777 #define SET_DET_RESULTS(st, det, result) \
|
|
778 set_detection_results (st, detector_##det, result)
|
|
779
|
|
780 typedef struct
|
|
781 {
|
|
782 Dynarr_declare (struct detector);
|
|
783 } detector_dynarr;
|
|
784
|
|
785 extern detector_dynarr *all_coding_detectors;
|
|
786
|
|
787 #define DEFINE_DETECTOR_CATEGORY(detector, cat) \
|
|
788 int detector_category_##cat
|
|
789 #define DECLARE_DETECTOR_CATEGORY(detector, cat) \
|
|
790 extern int detector_category_##cat
|
|
791 #define INITIALIZE_DETECTOR_CATEGORY(detector, cat) \
|
|
792 do { \
|
|
793 struct detector_category dog; \
|
|
794 xzero (dog); \
|
|
795 detector_category_##cat = coding_detector_category_count++; \
|
|
796 dump_add_opaque_int (&detector_category_##cat); \
|
|
797 dog.id = detector_category_##cat; \
|
|
798 dog.sym = Q##cat; \
|
|
799 Dynarr_add (Dynarr_at (all_coding_detectors, detector_##detector).cats, \
|
|
800 dog); \
|
|
801 } while (0)
|
|
802
|
|
803 #define DEFINE_DETECTOR(Detector) \
|
|
804 int detector_##Detector
|
|
805 #define DECLARE_DETECTOR(Detector) \
|
|
806 extern int detector_##Detector
|
|
807 #define INITIALIZE_DETECTOR(Detector) \
|
|
808 do { \
|
|
809 struct detector det; \
|
|
810 xzero (det); \
|
|
811 detector_##Detector = coding_detector_count++; \
|
|
812 dump_add_opaque_int (&detector_##Detector); \
|
|
813 det.id = detector_##Detector; \
|
|
814 det.cats = Dynarr_new2 (detector_category_dynarr, \
|
|
815 struct detector_category); \
|
|
816 det.data_size = sizeof (struct Detector##_detector); \
|
|
817 Dynarr_add (all_coding_detectors, det); \
|
|
818 } while (0)
|
|
819 #define DETECTOR_HAS_METHOD(Detector, Meth) \
|
|
820 Dynarr_at (all_coding_detectors, detector_##Detector).Meth##_method = \
|
802
|
821 Detector##_##Meth
|
771
|
822
|
|
823
|
|
824 /**************************************************/
|
|
825 /* Decoding/Encoding */
|
|
826 /**************************************************/
|
|
827
|
|
828 /* Is the source (SOURCEP == 1) or sink (SOURCEP == 0) when encoding specified
|
|
829 in characters? */
|
|
830
|
|
831 enum source_or_sink
|
|
832 {
|
|
833 CODING_SOURCE,
|
|
834 CODING_SINK
|
|
835 };
|
|
836
|
|
837 enum encode_decode
|
|
838 {
|
|
839 CODING_ENCODE,
|
|
840 CODING_DECODE
|
|
841 };
|
|
842
|
|
843 /* Data structure attached to an lstream of type `coding',
|
|
844 containing values specific to the coding process. Additional
|
|
845 data is stored in the DATA field below; the exact form of that data
|
|
846 is controlled by the type of the coding system that governs the
|
|
847 conversion (field CODESYS). CODESYS may be set at any time
|
|
848 throughout the lifetime of the lstream and possibly more than once.
|
|
849 See long comment above for more info. */
|
|
850
|
|
851 struct coding_stream
|
|
852 {
|
1204
|
853 /* Enumerated constant listing which type of console this is (TTY, X,
|
|
854 MS-Windows, etc.). This duplicates the method structure in
|
|
855 XCODING_SYSTEM (str->codesys)->methods->type, which formerly was the
|
|
856 only way to determine the coding system type. We need this constant
|
|
857 now for KKCC, so that it can be used in an XD_UNION clause to
|
|
858 determine the Lisp objects in the type-specific data. */
|
|
859 enum coding_system_variant type;
|
|
860
|
771
|
861 /* Coding system that governs the conversion. */
|
|
862 Lisp_Object codesys;
|
|
863 /* Original coding system, pre-canonicalization. */
|
|
864 Lisp_Object orig_codesys;
|
|
865
|
|
866 /* Back pointer to current stream. */
|
|
867 Lstream *us;
|
|
868
|
|
869 /* Stream that we read the unprocessed data from or write the processed
|
|
870 data to. */
|
|
871 Lstream *other_end;
|
|
872
|
|
873 /* In order to handle both reading to and writing from a coding stream,
|
|
874 we phrase the conversion methods like write methods -- we can
|
|
875 implement reading in terms of a write method but not vice-versa,
|
|
876 because the write method is forced to take only what it's given but
|
|
877 the read method can read more data from the other end if necessary.
|
|
878 On the other hand, the write method is free to generate all the data
|
2297
|
879 it wants (and just write it to the other end), but the read method
|
771
|
880 can return only as much as was asked for, so we need to implement our
|
|
881 own buffering. */
|
|
882
|
|
883 /* If we are reading, then we can return only a fixed amount of data, but
|
|
884 the converter is free to return as much as it wants, so we direct it
|
|
885 to store the data here and lop off chunks as we need them. If we are
|
|
886 writing, we use this because the converter takes a Dynarr but we are
|
|
887 supposed to write into a fixed buffer. (NOTE: This introduces an extra
|
|
888 memory copy.) */
|
|
889 unsigned_char_dynarr *convert_to;
|
|
890
|
|
891 /* The conversion method might reject some of the data -- this typically
|
|
892 includes partial characters, partial escape sequences, etc. When
|
|
893 writing, we just pass the rejection up to the Lstream module, and it
|
|
894 will buffer the data. When reading, however, we need to do the
|
|
895 buffering ourselves, and we put it here, combined with newly read
|
|
896 data. */
|
|
897 unsigned_char_dynarr *convert_from;
|
|
898
|
|
899 /* If set, this is the last chunk of data being processed. When this is
|
|
900 finished, output any necessary terminating control characters, escape
|
|
901 sequences, etc. */
|
|
902 unsigned int eof:1;
|
|
903
|
|
904 /* CH holds a partially built-up character. This is really part of the
|
|
905 state-dependent data and should be moved there. */
|
|
906 unsigned int ch;
|
|
907
|
|
908 /* Coding-system-specific data holding extra state about the
|
|
909 conversion. Logically a struct TYPE_coding_stream; a pointer
|
800
|
910 to such a struct, with (when ERROR_CHECK_TYPES is defined)
|
771
|
911 error-checking that this is really a structure of that type
|
|
912 (checking the corresponding coding system type) can be retrieved using
|
|
913 CODING_STREAM_TYPE_DATA(). Allocated at the same time that
|
|
914 CODESYS is set (which may occur at any time, even multiple times,
|
|
915 during the lifetime of the stream). The size comes from
|
|
916 methods->coding_data_size. */
|
|
917 void *data;
|
|
918
|
|
919 enum encode_decode direction;
|
|
920
|
800
|
921 /* If set, don't close the stream at the other end when being closed. */
|
|
922 unsigned int no_close_other:1;
|
802
|
923 /* If set, read only one byte at a time from other end to avoid any
|
|
924 possible blocking. */
|
|
925 unsigned int one_byte_at_a_time:1;
|
814
|
926 /* If set, and we're a read stream, we init char mode on ourselves as
|
|
927 necessary to prevent the caller from getting partial characters. (the
|
|
928 default) */
|
|
929 unsigned int set_char_mode_on_us_when_reading:1;
|
800
|
930
|
771
|
931 /* #### Temporary test */
|
|
932 unsigned int finalized:1;
|
|
933 };
|
|
934
|
|
935 #define CODING_STREAM_DATA(stream) LSTREAM_TYPE_DATA (stream, coding)
|
|
936
|
800
|
937 #ifdef ERROR_CHECK_TYPES
|
771
|
938 # define CODING_STREAM_TYPE_DATA(s, type) \
|
|
939 error_check_##type##_coding_stream_data (s)
|
|
940 #else
|
|
941 # define CODING_STREAM_TYPE_DATA(s, type) \
|
|
942 ((struct type##_coding_stream *) (s)->data)
|
|
943 #endif
|
|
944
|
|
945 /* C should be a binary character in the range 0 - 255; convert
|
|
946 to internal format and add to Dynarr DST. */
|
|
947
|
428
|
948 #ifdef MULE
|
771
|
949
|
|
950 #define DECODE_ADD_BINARY_CHAR(c, dst) \
|
|
951 do { \
|
826
|
952 if (byte_ascii_p (c)) \
|
771
|
953 Dynarr_add (dst, c); \
|
826
|
954 else if (byte_c1_p (c)) \
|
771
|
955 { \
|
|
956 Dynarr_add (dst, LEADING_BYTE_CONTROL_1); \
|
|
957 Dynarr_add (dst, c + 0x20); \
|
|
958 } \
|
|
959 else \
|
|
960 { \
|
|
961 Dynarr_add (dst, LEADING_BYTE_LATIN_ISO8859_1); \
|
|
962 Dynarr_add (dst, c); \
|
|
963 } \
|
|
964 } while (0)
|
|
965
|
|
966 #else /* not MULE */
|
|
967
|
|
968 #define DECODE_ADD_BINARY_CHAR(c, dst) \
|
|
969 do { \
|
|
970 Dynarr_add (dst, c); \
|
|
971 } while (0)
|
|
972
|
|
973 #endif /* MULE */
|
|
974
|
|
975 #define DECODE_OUTPUT_PARTIAL_CHAR(ch, dst) \
|
|
976 do { \
|
|
977 if (ch) \
|
|
978 { \
|
|
979 DECODE_ADD_BINARY_CHAR (ch, dst); \
|
|
980 ch = 0; \
|
|
981 } \
|
|
982 } while (0)
|
428
|
983
|
|
984 #ifdef MULE
|
|
985 /* Convert shift-JIS code (sj1, sj2) into internal string
|
|
986 representation (c1, c2). (The leading byte is assumed.) */
|
|
987
|
771
|
988 #define DECODE_SHIFT_JIS(sj1, sj2, c1, c2) \
|
428
|
989 do { \
|
|
990 int I1 = sj1, I2 = sj2; \
|
|
991 if (I2 >= 0x9f) \
|
|
992 c1 = (I1 << 1) - ((I1 >= 0xe0) ? 0xe0 : 0x60), \
|
|
993 c2 = I2 + 2; \
|
|
994 else \
|
|
995 c1 = (I1 << 1) - ((I1 >= 0xe0) ? 0xe1 : 0x61), \
|
|
996 c2 = I2 + ((I2 >= 0x7f) ? 0x60 : 0x61); \
|
|
997 } while (0)
|
|
998
|
|
999 /* Convert the internal string representation of a Shift-JIS character
|
|
1000 (c1, c2) into Shift-JIS code (sj1, sj2). The leading byte is
|
|
1001 assumed. */
|
|
1002
|
771
|
1003 #define ENCODE_SHIFT_JIS(c1, c2, sj1, sj2) \
|
428
|
1004 do { \
|
|
1005 int I1 = c1, I2 = c2; \
|
|
1006 if (I1 & 1) \
|
|
1007 sj1 = (I1 >> 1) + ((I1 < 0xdf) ? 0x31 : 0x71), \
|
|
1008 sj2 = I2 - ((I2 >= 0xe0) ? 0x60 : 0x61); \
|
|
1009 else \
|
|
1010 sj1 = (I1 >> 1) + ((I1 < 0xdf) ? 0x30 : 0x70), \
|
|
1011 sj2 = I2 - 2; \
|
|
1012 } while (0)
|
|
1013 #endif /* MULE */
|
|
1014
|
771
|
1015 DECLARE_CODING_SYSTEM_TYPE (no_conversion);
|
|
1016 DECLARE_CODING_SYSTEM_TYPE (convert_eol);
|
|
1017 #if 0
|
|
1018 DECLARE_CODING_SYSTEM_TYPE (text_file_wrapper);
|
|
1019 #endif /* 0 */
|
|
1020 DECLARE_CODING_SYSTEM_TYPE (undecided);
|
|
1021 DECLARE_CODING_SYSTEM_TYPE (chain);
|
|
1022
|
|
1023 #ifdef DEBUG_XEMACS
|
|
1024 DECLARE_CODING_SYSTEM_TYPE (internal);
|
|
1025 #endif
|
|
1026
|
|
1027 #ifdef MULE
|
|
1028 DECLARE_CODING_SYSTEM_TYPE (iso2022);
|
|
1029 DECLARE_CODING_SYSTEM_TYPE (ccl);
|
|
1030 DECLARE_CODING_SYSTEM_TYPE (shift_jis);
|
|
1031 DECLARE_CODING_SYSTEM_TYPE (big5);
|
|
1032 #endif
|
|
1033
|
|
1034 #ifdef HAVE_ZLIB
|
|
1035 DECLARE_CODING_SYSTEM_TYPE (gzip);
|
|
1036 #endif
|
428
|
1037
|
771
|
1038 DECLARE_CODING_SYSTEM_TYPE (unicode);
|
428
|
1039
|
1315
|
1040 #ifdef WIN32_ANY
|
771
|
1041 DECLARE_CODING_SYSTEM_TYPE (mswindows_multibyte_to_unicode);
|
|
1042 DECLARE_CODING_SYSTEM_TYPE (mswindows_multibyte);
|
428
|
1043 #endif
|
771
|
1044
|
|
1045 Lisp_Object coding_stream_detected_coding_system (Lstream *stream);
|
|
1046 Lisp_Object coding_stream_coding_system (Lstream *stream);
|
|
1047 void set_coding_stream_coding_system (Lstream *stream,
|
|
1048 Lisp_Object codesys);
|
|
1049 Lisp_Object detect_coding_stream (Lisp_Object stream);
|
867
|
1050 Ichar decode_big5_char (int o1, int o2);
|
771
|
1051 void add_entry_to_coding_system_type_list (struct coding_system_methods *m);
|
|
1052 Lisp_Object make_internal_coding_system (Lisp_Object existing,
|
2367
|
1053 Ascbyte *prefix,
|
771
|
1054 Lisp_Object type,
|
|
1055 Lisp_Object description,
|
|
1056 Lisp_Object props);
|
802
|
1057
|
814
|
1058 #define LSTREAM_FL_NO_CLOSE_OTHER (1 << 16)
|
|
1059 #define LSTREAM_FL_READ_ONE_BYTE_AT_A_TIME (1 << 17)
|
|
1060 #define LSTREAM_FL_NO_INIT_CHAR_MODE_WHEN_READING (1 << 18)
|
|
1061
|
771
|
1062 Lisp_Object make_coding_input_stream (Lstream *stream, Lisp_Object codesys,
|
800
|
1063 enum encode_decode direction,
|
802
|
1064 int flags);
|
771
|
1065 Lisp_Object make_coding_output_stream (Lstream *stream, Lisp_Object codesys,
|
800
|
1066 enum encode_decode direction,
|
802
|
1067 int flags);
|
771
|
1068 void set_detection_results (struct detection_state *st, int detector,
|
|
1069 int given);
|
428
|
1070
|
440
|
1071 #endif /* INCLUDED_file_coding_h_ */
|
|
1072
|