mirror of
https://git.savannah.gnu.org/git/guile.git
synced 2025-05-01 04:10:18 +02:00
Ports are given two additional properties: a character encoding and a conversion failure strategy. These properties have getters and setters. The new properties are used to convert any locale text to/from the internal representation of strings. If unspecified, ports use a default value. The default value of these properties is held in a fluid. The default character encoding can be modified by calling setlocale. ISO-8859-1 is treated specially. Since it is a native encoding of strings, it can be processed more quickly. Source code is assumed to be ISO-8859-1 unless otherwise specified. The encoding of a source code file can be given as 'coding: XXXXX' in a magic comment at the top of a file. The C functions that deal with encoding often use a null pointer as shorthand for the native Latin-1 encoding, for efficiency's sake. * test-suite/tests/encoding-iso88591.test: new tests * test-suite/tests/encoding-iso88597.test: new tests * test-suite/tests/encoding-utf8.test: new tests * test-suite/tests/encoding-escapes.test: new tests * test-suite/tests/numbers.test: declare 'binary' encoding * test-suite/tests/ports.test: declare 'binary' encoding * test-suite/tests/r6rs-ports.test: declare 'binary' encoding * module/system/base/compile.scm (compile-file): use source-code file's self-declared encoding when compiling files * libguile/strports.c: store string ports in locale encoding (scm_strport_to_locale_u8vector, scm_call_with_output_locale_u8vector) (scm_open_input_locale_u8vector, scm_get_output_locale_u8vector): new functions * libguile/strings.h: new declaration for scm_i_string_contains_char * libguile/strings.c (scm_i_string_contains_char): new function (scm_from_stringn, scm_to_stringn): use NULL for Latin-1 (scm_from_locale_stringn, scm_to_locale_stringn): respect character encoding of input and output ports * libguile/read.h: declaration for scm_scan_for_encoding * libguile/read.c: (read_token): now takes scheme string instead of C string/length (read_complete_token): new function (scm_read_sexp, scm_read_number, scm_read_mixed_case_symbol) (scm_read_number_and_radix, scm_read_quote, scm_read_semicolon_comment) (scm_read_srfi4_vector, scm_read_bytevector, scm_read_guile_bit_vector) (scm_read_scsh_block_comment, scm_read_commented_expression) (scm_read_extended_symbol, scm_read_sharp_extension, scm_read_shart) (scm_read_expression): use scm_t_wchar for char type, use read_complete_token (scm_scan_for_encoding): new function to find a file's character encoding (scm_file_encoding): new function to find a port's character encoding * libguile/rdelim.c: don't unpack strings * libguile/print.h: declaration for modified function scm_i_charprint * libguile/print.c: use locale when printing characters and strings (scm_i_charprint): input parameter is now scm_t_wchar (scm_simple_format): don't unpack strings * libguile/posix.h: new declaration for scm_setbinary. * libguile/posix.c (scm_setlocale): set default and stdio port encodings based on the locale's character encoding (scm_setbinary): new function * libguile/ports.h (scm_t_port): add encoding and failed conversion handler to port type. Declarations for new or modified functions scm_getc, scm_unget_byte, scm_ungetc, scm_i_get_port_encoding, scm_i_set_port_encoding_x, scm_port_encoding, scm_set_port_encoding_x, scm_i_get_conversion_strategy, scm_i_set_conversion_strategy_x, scm_port_conversion_strategy, scm_set_port_conversion_strategy_x. * libguile/ports.c: assign the current ports to zero on startup so we can see if they've been set. (scm_current_input_port, scm_current_output_port, scm_current_error_port): return #f if the port is not yet initialized (scm_new_port_table_entry): set up a new port's encoding and illegal sequence handler based on the thread's current defaults (scm_i_remove_port): free port encoding name when port is removed (scm_i_mode_bits_n): now takes a scheme string instead of a c string and length. All callers changed. (SCM_MBCHAR_BUF_SIZE): new const (scm_getc): new function, since the scm_getc in inline.h is now scm_get_byte_or_eof. This pulls one codepoint from a port. (scm_lfwrite_substr, scm_lfwrite_str): now uses port's encoding (scm_unget_byte): new function, incorportaing the low-level functionality of scm_ungetc (scm_ungetc): uses scm_unget_byte * libguile/numbers.h (scm_t_wchar): compilation order problem with scm_t_wchar being use in functions in multiple headers. Forward declare scm_t_wchar. * libguile/load.c (scm_primitive_load): scan for file encoding at top of file and use it to set the load port's encoding * libguile/inline.h (scm_get_byte_or_eof): new function incorporating most of the functionality of scm_getc. * libguile/fports.c (fport_fill_input): now returns scm_t_wchar * libguile/chars.h (scm_t_wchar): avoid compilation order problem with declaration of scm_t_wchar
91 lines
3.3 KiB
C
91 lines
3.3 KiB
C
/* classes: h_files */
|
||
|
||
#ifndef SCM_CHARS_H
|
||
#define SCM_CHARS_H
|
||
|
||
/* Copyright (C) 1995,1996,2000,2001,2004, 2006, 2008, 2009 Free Software Foundation, Inc.
|
||
*
|
||
* This library is free software; you can redistribute it and/or
|
||
* modify it under the terms of the GNU Lesser General Public License
|
||
* as published by the Free Software Foundation; either version 3 of
|
||
* the License, or (at your option) any later version.
|
||
*
|
||
* This library is distributed in the hope that it will be useful, but
|
||
* WITHOUT ANY WARRANTY; without even the implied warranty of
|
||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||
* Lesser General Public License for more details.
|
||
*
|
||
* You should have received a copy of the GNU Lesser General Public
|
||
* License along with this library; if not, write to the Free Software
|
||
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA
|
||
* 02110-1301 USA
|
||
*/
|
||
|
||
|
||
|
||
#include "libguile/__scm.h"
|
||
|
||
#ifndef SCM_T_WCHAR_DEFINED
|
||
typedef scm_t_int32 scm_t_wchar;
|
||
#define SCM_T_WCHAR_DEFINED
|
||
#endif /* SCM_T_WCHAR_DEFINED */
|
||
|
||
|
||
/* Immediate Characters
|
||
*/
|
||
#define SCM_CHARP(x) (SCM_ITAG8(x) == scm_tc8_char)
|
||
#define SCM_CHAR(x) ((scm_t_wchar)SCM_ITAG8_DATA(x))
|
||
|
||
/* SCM_MAKE_CHAR maps signed chars (-128 to 127) and unsigned chars (0
|
||
to 255) to Latin-1 codepoints (0 to 255) while allowing higher
|
||
codepoints (256 to 1114111) to pass through unchanged.
|
||
|
||
This macro evaluates x twice, which may lead to side effects if not
|
||
used properly. */
|
||
#define SCM_MAKE_CHAR(x) \
|
||
((x) <= 1 \
|
||
? SCM_MAKE_ITAG8 ((scm_t_bits) (unsigned char) (x), scm_tc8_char) \
|
||
: SCM_MAKE_ITAG8 ((scm_t_bits) (x), scm_tc8_char))
|
||
|
||
#define SCM_CODEPOINT_MAX (0x10ffff)
|
||
#define SCM_IS_UNICODE_CHAR(c) \
|
||
((scm_t_wchar) (c) <= 0xd7ff \
|
||
|| ((scm_t_wchar) (c) >= 0xe000 && (scm_t_wchar) (c) <= SCM_CODEPOINT_MAX))
|
||
|
||
|
||
|
||
SCM_API SCM scm_char_p (SCM x);
|
||
SCM_API SCM scm_char_eq_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_less_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_leq_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_gr_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_geq_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_ci_eq_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_ci_less_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_ci_leq_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_ci_gr_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_ci_geq_p (SCM x, SCM y);
|
||
SCM_API SCM scm_char_alphabetic_p (SCM chr);
|
||
SCM_API SCM scm_char_numeric_p (SCM chr);
|
||
SCM_API SCM scm_char_whitespace_p (SCM chr);
|
||
SCM_API SCM scm_char_upper_case_p (SCM chr);
|
||
SCM_API SCM scm_char_lower_case_p (SCM chr);
|
||
SCM_API SCM scm_char_is_both_p (SCM chr);
|
||
SCM_API SCM scm_char_to_integer (SCM chr);
|
||
SCM_API SCM scm_integer_to_char (SCM n);
|
||
SCM_API SCM scm_char_upcase (SCM chr);
|
||
SCM_API SCM scm_char_downcase (SCM chr);
|
||
SCM_API scm_t_wchar scm_c_upcase (scm_t_wchar c);
|
||
SCM_API scm_t_wchar scm_c_downcase (scm_t_wchar c);
|
||
SCM_INTERNAL const char *scm_i_charname (SCM chr);
|
||
SCM_INTERNAL SCM scm_i_charname_to_char (const char *charname,
|
||
size_t charname_len);
|
||
SCM_INTERNAL void scm_init_chars (void);
|
||
|
||
#endif /* SCM_CHARS_H */
|
||
|
||
/*
|
||
Local Variables:
|
||
c-file-style: "gnu"
|
||
End:
|
||
*/
|