mirror of
https://git.savannah.gnu.org/git/guile.git
synced 2025-06-17 09:10:22 +02:00
Add Unicode strings and symbols
This adds full Unicode strings as a datatype, and it adds some minimal functionality. The terminal and port encoding is assumed to be ISO-8859-1. Non-ISO-8859-1 characters are written or input as string character escapes. The string character escapes now have 3 forms: \xXX \uXXXX and \UXXXXXX, for unprintable characters that have 2, 4 or 6 hex digits. The process for writing to strings has been modified. There is now a function scm_i_string_start_writing that does the copy-on-write conversion if necessary. To compile strings that may be wide, the VM storage of strings and string-likes has changed. Most string-using functions have not yet been updated and may break when used with wide strings. * module/language/assembly/compile-bytecode.scm (write-bytecode): use variable width string bytecode format * module/language/assembly.scm (byte-length): use variable width bytecode format * libguile/vm-i-loader.c (load-string, load-symbol): (load-keyword, define): use variable-width bytecode format * libguile/vm-engine.h (FETCH_WIDTH): new macro * libguile/strings.h: new declarations * libguile/strings.c (make_wide_stringbuf): new function (widen_stringbuf): new function (scm_i_make_wide_string): new function (scm_i_is_narrow_string): new function (scm_i_string_wide_chars): new function (scm_i_string_start_writing): new function (scm_i_string_ref): new function (scm_i_string_set_x): new function (scm_i_is_narrow_symbol): new function (scm_i_symbol_wide_chars, scm_i_symbol_ref): new function (scm_string_width): new function (unistring_escapes_to_guile_escapes): new function (scm_to_stringn): new function (scm_i_stringbuf_free): modify for wide strings (scm_i_substring_copy): modify for wide strings (scm_i_string_chars, scm_string_append): modify for wide strings (scm_i_make_symbol, scm_to_locale_stringn): modify for wide strings (scm_string_dump, scm_symbol_dump, scm_to_locale_stringbuf): (scm_string, scm_i_deprecated_string_chars): modify for wide strings (scm_from_locale_string, scm_from_locale_stringn): add null test * libguile/srfi-13.c: add calls for scm_i_string_start_writing for each call of scm_i_string_stop_writing (scm_string_for_each): modify for wide strings * libguile/socket.c: add calls for scm_i_string_start_writing for each call of scm_i_string_stop_writing * libguile/rw.c: add calls for scm_i_string_start_writing for each call of scm_i_string_stop_writing * libguile/read.c (scm_read_string): allow reading of wide strings * libguile/print.h: add declaration for scm_charprint * libguile/print.c (iprin1): print wide strings and add new string escapes (scm_charprint): new function * libguile/ports.h: new declarations for scm_lfwrite_substr and scm_lfwrite_str * libguile/ports.c (update_port_lf): new function (scm_lfwrite): use update_port_lf (scm_lfwrite_substr): new function (scm_lfwrite_str): new function * test-suite/tests/asm-to-bytecode.test ("compiler"): add string width byte to sting-like asm tests
This commit is contained in:
parent
a876e7dcea
commit
9c44cd4559
15 changed files with 1046 additions and 306 deletions
|
@ -23,6 +23,7 @@
|
|||
|
||||
|
||||
|
||||
#include <uniconv.h>
|
||||
#include "libguile/__scm.h"
|
||||
|
||||
|
||||
|
@ -46,26 +47,37 @@
|
|||
|
||||
Internal, low level interface to the character arrays
|
||||
|
||||
- Use scm_i_string_chars to get a pointer to the byte array of a
|
||||
string for reading. Use scm_i_string_length to get the number of
|
||||
bytes in that array. The array is not null-terminated.
|
||||
- Use scm_is_narrow_string to determine is the string is narrow or
|
||||
wide.
|
||||
|
||||
- Use scm_i_string_chars or scm_i_string_wide_chars to get a
|
||||
pointer to the byte or scm_t_wchar array of a string for reading.
|
||||
Use scm_i_string_length to get the number of characters in that
|
||||
array. The array is not null-terminated.
|
||||
|
||||
- The array is valid as long as the corresponding SCM object is
|
||||
protected but only until the next SCM_TICK. During such a 'safe
|
||||
point', strings might change their representation.
|
||||
|
||||
- Use scm_i_string_writable_chars to get the same pointer as with
|
||||
scm_i_string_chars, but for reading and writing. This is a
|
||||
potentially costly operation since it implements the
|
||||
copy-on-write behavior. When done with the writing, call
|
||||
scm_i_string_stop_writing. You must do this before the next
|
||||
SCM_TICK. (This means, before calling almost any other scm_
|
||||
function and you can't allow throws, of course.)
|
||||
- Use scm_i_string_start_writing to get a version of the string
|
||||
ready for reading and writing. This is a potentially costly
|
||||
operation since it implements the copy-on-write behavior. When
|
||||
done with the writing, call scm_i_string_stop_writing. You must
|
||||
do this before the next SCM_TICK. (This means, before calling
|
||||
almost any other scm_ function and you can't allow throws, of
|
||||
course.)
|
||||
|
||||
- New strings can be created with scm_i_make_string. This gives
|
||||
access to a writable pointer that remains valid as long as nobody
|
||||
else makes a copy-on-write substring of the string. Do not call
|
||||
scm_i_string_stop_writing for this pointer.
|
||||
- New strings can be created with scm_i_make_string or
|
||||
scm_i_make_wide_string. This gives access to a writable pointer
|
||||
that remains valid as long as nobody else makes a copy-on-write
|
||||
substring of the string. Do not call scm_i_string_stop_writing
|
||||
for this pointer.
|
||||
|
||||
- Alternately, scm_i_string_ref and scm_i_string_set_x can be used
|
||||
to read and write strings without worrying about whether the
|
||||
string is narrow or wide. scm_i_string_set_x still needs to be
|
||||
bracketed by scm_i_string_start_writing and
|
||||
scm_i_string_stop_writing.
|
||||
|
||||
Legacy interface
|
||||
|
||||
|
@ -74,13 +86,15 @@
|
|||
- SCM_STRING_CHARS uses scm_i_string_writable_chars and immediately
|
||||
calls scm_i_stop_writing, hoping for the best. SCM_STRING_LENGTH
|
||||
is the same as scm_i_string_length. SCM_STRING_CHARS will throw
|
||||
an error for for strings that are not null-terminated.
|
||||
an error for for strings that are not null-terminated. There is
|
||||
no wide version of this interface.
|
||||
*/
|
||||
|
||||
SCM_API SCM scm_string_p (SCM x);
|
||||
SCM_API SCM scm_string (SCM chrs);
|
||||
SCM_API SCM scm_make_string (SCM k, SCM chr);
|
||||
SCM_API SCM scm_string_length (SCM str);
|
||||
SCM_API SCM scm_string_width (SCM str);
|
||||
SCM_API SCM scm_string_ref (SCM str, SCM k);
|
||||
SCM_API SCM scm_string_set_x (SCM str, SCM k, SCM chr);
|
||||
SCM_API SCM scm_substring (SCM str, SCM start, SCM end);
|
||||
|
@ -106,6 +120,9 @@ SCM_API SCM scm_take_locale_string (char *str);
|
|||
SCM_API SCM scm_take_locale_stringn (char *str, size_t len);
|
||||
SCM_API char *scm_to_locale_string (SCM str);
|
||||
SCM_API char *scm_to_locale_stringn (SCM str, size_t *lenp);
|
||||
SCM_INTERNAL char *scm_to_stringn (SCM str, size_t *lenp,
|
||||
const char *encoding,
|
||||
enum iconv_ilseq_handler handler);
|
||||
SCM_API size_t scm_to_locale_stringbuf (SCM str, char *buf, size_t max_len);
|
||||
|
||||
SCM_API SCM scm_makfromstrs (int argc, char **argv);
|
||||
|
@ -113,15 +130,20 @@ SCM_API SCM scm_makfromstrs (int argc, char **argv);
|
|||
/* internal accessor functions. Arguments must be valid. */
|
||||
|
||||
SCM_INTERNAL SCM scm_i_make_string (size_t len, char **datap);
|
||||
SCM_INTERNAL SCM scm_i_make_wide_string (size_t len, scm_t_wchar **datap);
|
||||
SCM_INTERNAL SCM scm_i_substring (SCM str, size_t start, size_t end);
|
||||
SCM_INTERNAL SCM scm_i_substring_read_only (SCM str, size_t start, size_t end);
|
||||
SCM_INTERNAL SCM scm_i_substring_shared (SCM str, size_t start, size_t end);
|
||||
SCM_INTERNAL SCM scm_i_substring_copy (SCM str, size_t start, size_t end);
|
||||
SCM_INTERNAL size_t scm_i_string_length (SCM str);
|
||||
SCM_API /* FIXME: not internal */ const char *scm_i_string_chars (SCM str);
|
||||
SCM_API const scm_t_wchar *scm_i_string_wide_chars (SCM str);
|
||||
SCM_API /* FIXME: not internal */ char *scm_i_string_writable_chars (SCM str);
|
||||
SCM_INTERNAL SCM scm_i_string_start_writing (SCM str);
|
||||
SCM_INTERNAL void scm_i_string_stop_writing (void);
|
||||
|
||||
SCM_INTERNAL int scm_i_is_narrow_string (SCM str);
|
||||
SCM_INTERNAL scm_t_wchar scm_i_string_ref (SCM str, size_t x);
|
||||
SCM_INTERNAL void scm_i_string_set_x (SCM str, size_t p, scm_t_wchar chr);
|
||||
/* internal functions related to symbols. */
|
||||
|
||||
SCM_INTERNAL SCM scm_i_make_symbol (SCM name, scm_t_bits flags,
|
||||
|
@ -133,8 +155,11 @@ SCM_INTERNAL SCM
|
|||
scm_i_c_take_symbol (char *name, size_t len,
|
||||
scm_t_bits flags, unsigned long hash, SCM props);
|
||||
SCM_INTERNAL const char *scm_i_symbol_chars (SCM sym);
|
||||
SCM_INTERNAL const scm_t_wchar *scm_i_symbol_wide_chars (SCM sym);
|
||||
SCM_INTERNAL size_t scm_i_symbol_length (SCM sym);
|
||||
SCM_INTERNAL int scm_i_is_narrow_symbol (SCM str);
|
||||
SCM_INTERNAL SCM scm_i_symbol_substring (SCM sym, size_t start, size_t end);
|
||||
SCM_INTERNAL scm_t_wchar scm_i_symbol_ref (SCM sym, size_t x);
|
||||
|
||||
/* internal GC functions. */
|
||||
|
||||
|
|
Loading…
Add table
Add a link
Reference in a new issue