strings.h 8.9 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223
  1. /* classes: h_files */
  2. #ifndef SCM_STRINGS_H
  3. #define SCM_STRINGS_H
  4. /* Copyright (C) 1995,1996,1997,1998,2000,2001, 2004, 2005, 2006, 2008, 2009 Free Software Foundation, Inc.
  5. *
  6. * This library is free software; you can redistribute it and/or
  7. * modify it under the terms of the GNU Lesser General Public License
  8. * as published by the Free Software Foundation; either version 3 of
  9. * the License, or (at your option) any later version.
  10. *
  11. * This library is distributed in the hope that it will be useful, but
  12. * WITHOUT ANY WARRANTY; without even the implied warranty of
  13. * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
  14. * Lesser General Public License for more details.
  15. *
  16. * You should have received a copy of the GNU Lesser General Public
  17. * License along with this library; if not, write to the Free Software
  18. * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA
  19. * 02110-1301 USA
  20. */
  21. #include "libguile/__scm.h"
  22. /* String representation.
  23. A string is a piece of a stringbuf. A stringbuf can be used by
  24. more than one string. When a string is written to and the
  25. stringbuf of that string is used by more than one string, a new
  26. stringbuf is created. That is, strings are copy-on-write. This
  27. behavior can be used to make the substring operation quite
  28. efficient.
  29. The implementation is tuned so that mutating a string is costly,
  30. but just reading it is cheap and lock-free.
  31. There are also mutation-sharing strings. They refer to a part of
  32. an ordinary string. Writing to a mutation-sharing string just
  33. writes to the ordinary string.
  34. Internal, low level interface to the character arrays
  35. - Use scm_is_narrow_string to determine is the string is narrow or
  36. wide.
  37. - Use scm_i_string_chars or scm_i_string_wide_chars to get a
  38. pointer to the byte or scm_t_wchar array of a string for reading.
  39. Use scm_i_string_length to get the number of characters in that
  40. array. The array is not null-terminated.
  41. - The array is valid as long as the corresponding SCM object is
  42. protected but only until the next SCM_TICK. During such a 'safe
  43. point', strings might change their representation.
  44. - Use scm_i_string_start_writing to get a version of the string
  45. ready for reading and writing. This is a potentially costly
  46. operation since it implements the copy-on-write behavior. When
  47. done with the writing, call scm_i_string_stop_writing. You must
  48. do this before the next SCM_TICK. (This means, before calling
  49. almost any other scm_ function and you can't allow throws, of
  50. course.)
  51. - New strings can be created with scm_i_make_string or
  52. scm_i_make_wide_string. This gives access to a writable pointer
  53. that remains valid as long as nobody else makes a copy-on-write
  54. substring of the string. Do not call scm_i_string_stop_writing
  55. for this pointer.
  56. - Alternately, scm_i_string_ref and scm_i_string_set_x can be used
  57. to read and write strings without worrying about whether the
  58. string is narrow or wide. scm_i_string_set_x still needs to be
  59. bracketed by scm_i_string_start_writing and
  60. scm_i_string_stop_writing.
  61. Legacy interface
  62. - SCM_STRINGP is just scm_is_string.
  63. - SCM_STRING_CHARS uses scm_i_string_writable_chars and immediately
  64. calls scm_i_stop_writing, hoping for the best. SCM_STRING_LENGTH
  65. is the same as scm_i_string_length. SCM_STRING_CHARS will throw
  66. an error for for strings that are not null-terminated. There is
  67. no wide version of this interface.
  68. */
  69. /* A type indicating what strategy to take when string locale
  70. conversion is unsuccessful. */
  71. typedef enum
  72. {
  73. SCM_FAILED_CONVERSION_ERROR = SCM_ICONVEH_ERROR,
  74. SCM_FAILED_CONVERSION_QUESTION_MARK = SCM_ICONVEH_QUESTION_MARK,
  75. SCM_FAILED_CONVERSION_ESCAPE_SEQUENCE = SCM_ICONVEH_ESCAPE_SEQUENCE
  76. } scm_t_string_failed_conversion_handler;
  77. SCM_API SCM scm_string_p (SCM x);
  78. SCM_API SCM scm_string (SCM chrs);
  79. SCM_API SCM scm_make_string (SCM k, SCM chr);
  80. SCM_API SCM scm_string_length (SCM str);
  81. SCM_API SCM scm_string_width (SCM str);
  82. SCM_API SCM scm_string_ref (SCM str, SCM k);
  83. SCM_API SCM scm_string_set_x (SCM str, SCM k, SCM chr);
  84. SCM_API SCM scm_substring (SCM str, SCM start, SCM end);
  85. SCM_API SCM scm_substring_read_only (SCM str, SCM start, SCM end);
  86. SCM_API SCM scm_substring_shared (SCM str, SCM start, SCM end);
  87. SCM_API SCM scm_substring_copy (SCM str, SCM start, SCM end);
  88. SCM_API SCM scm_string_append (SCM args);
  89. SCM_API SCM scm_c_make_string (size_t len, SCM chr);
  90. SCM_API size_t scm_c_string_length (SCM str);
  91. SCM_API size_t scm_c_symbol_length (SCM sym);
  92. SCM_API SCM scm_c_string_ref (SCM str, size_t pos);
  93. SCM_API void scm_c_string_set_x (SCM str, size_t pos, SCM chr);
  94. SCM_API SCM scm_c_substring (SCM str, size_t start, size_t end);
  95. SCM_API SCM scm_c_substring_read_only (SCM str, size_t start, size_t end);
  96. SCM_API SCM scm_c_substring_shared (SCM str, size_t start, size_t end);
  97. SCM_API SCM scm_c_substring_copy (SCM str, size_t start, size_t end);
  98. SCM_API int scm_is_string (SCM x);
  99. SCM_API SCM scm_from_locale_string (const char *str);
  100. SCM_API SCM scm_from_locale_stringn (const char *str, size_t len);
  101. SCM_API SCM scm_take_locale_string (char *str);
  102. SCM_API SCM scm_take_locale_stringn (char *str, size_t len);
  103. SCM_API char *scm_to_locale_string (SCM str);
  104. SCM_API char *scm_to_locale_stringn (SCM str, size_t *lenp);
  105. SCM_INTERNAL char *scm_to_stringn (SCM str, size_t *lenp,
  106. const char *encoding,
  107. scm_t_string_failed_conversion_handler
  108. handler);
  109. SCM_API size_t scm_to_locale_stringbuf (SCM str, char *buf, size_t max_len);
  110. SCM_API SCM scm_makfromstrs (int argc, char **argv);
  111. /* internal accessor functions. Arguments must be valid. */
  112. SCM_INTERNAL SCM scm_i_make_string (size_t len, char **datap);
  113. SCM_INTERNAL SCM scm_i_make_wide_string (size_t len, scm_t_wchar **datap);
  114. SCM_INTERNAL SCM scm_i_substring (SCM str, size_t start, size_t end);
  115. SCM_INTERNAL SCM scm_i_substring_read_only (SCM str, size_t start, size_t end);
  116. SCM_INTERNAL SCM scm_i_substring_shared (SCM str, size_t start, size_t end);
  117. SCM_INTERNAL SCM scm_i_substring_copy (SCM str, size_t start, size_t end);
  118. SCM_INTERNAL size_t scm_i_string_length (SCM str);
  119. SCM_API /* FIXME: not internal */ const char *scm_i_string_chars (SCM str);
  120. SCM_API /* FIXME: not internal */ char *scm_i_string_writable_chars (SCM str);
  121. SCM_INTERNAL const scm_t_wchar *scm_i_string_wide_chars (SCM str);
  122. SCM_INTERNAL SCM scm_i_string_start_writing (SCM str);
  123. SCM_INTERNAL void scm_i_string_stop_writing (void);
  124. SCM_INTERNAL int scm_i_is_narrow_string (SCM str);
  125. SCM_INTERNAL scm_t_wchar scm_i_string_ref (SCM str, size_t x);
  126. SCM_INTERNAL void scm_i_string_set_x (SCM str, size_t p, scm_t_wchar chr);
  127. /* internal functions related to symbols. */
  128. SCM_INTERNAL SCM scm_i_make_symbol (SCM name, scm_t_bits flags,
  129. unsigned long hash, SCM props);
  130. SCM_INTERNAL SCM
  131. scm_i_c_make_symbol (const char *name, size_t len,
  132. scm_t_bits flags, unsigned long hash, SCM props);
  133. SCM_INTERNAL SCM
  134. scm_i_c_take_symbol (char *name, size_t len,
  135. scm_t_bits flags, unsigned long hash, SCM props);
  136. SCM_INTERNAL const char *scm_i_symbol_chars (SCM sym);
  137. SCM_INTERNAL const scm_t_wchar *scm_i_symbol_wide_chars (SCM sym);
  138. SCM_INTERNAL size_t scm_i_symbol_length (SCM sym);
  139. SCM_INTERNAL int scm_i_is_narrow_symbol (SCM str);
  140. SCM_INTERNAL SCM scm_i_symbol_substring (SCM sym, size_t start, size_t end);
  141. SCM_INTERNAL scm_t_wchar scm_i_symbol_ref (SCM sym, size_t x);
  142. /* internal GC functions. */
  143. SCM_INTERNAL SCM scm_i_string_mark (SCM str);
  144. SCM_INTERNAL SCM scm_i_stringbuf_mark (SCM buf);
  145. SCM_INTERNAL SCM scm_i_symbol_mark (SCM buf);
  146. SCM_INTERNAL void scm_i_string_free (SCM str);
  147. SCM_INTERNAL void scm_i_stringbuf_free (SCM buf);
  148. SCM_INTERNAL void scm_i_symbol_free (SCM sym);
  149. /* internal utility functions. */
  150. SCM_INTERNAL char **scm_i_allocate_string_pointers (SCM list);
  151. SCM_INTERNAL void scm_i_free_string_pointers (char **pointers);
  152. SCM_INTERNAL void scm_i_get_substring_spec (size_t len,
  153. SCM start, size_t *cstart,
  154. SCM end, size_t *cend);
  155. SCM_INTERNAL SCM scm_i_take_stringbufn (char *str, size_t len);
  156. /* Debugging functions */
  157. SCM_API SCM scm_sys_string_dump (SCM);
  158. SCM_API SCM scm_sys_symbol_dump (SCM);
  159. #if SCM_STRING_LENGTH_HISTOGRAM
  160. SCM_API SCM scm_sys_stringbuf_hist (void);
  161. #endif
  162. /* deprecated stuff */
  163. #if SCM_ENABLE_DEPRECATED
  164. SCM_API int scm_i_deprecated_stringp (SCM obj);
  165. SCM_API char *scm_i_deprecated_string_chars (SCM str);
  166. SCM_API size_t scm_i_deprecated_string_length (SCM str);
  167. #define SCM_STRINGP(x) scm_i_deprecated_stringp(x)
  168. #define SCM_STRING_CHARS(x) scm_i_deprecated_string_chars(x)
  169. #define SCM_STRING_LENGTH(x) scm_i_deprecated_string_length(x)
  170. #define SCM_STRING_UCHARS(str) ((unsigned char *)SCM_STRING_CHARS (str))
  171. #endif
  172. SCM_INTERNAL void scm_init_strings (void);
  173. #endif /* SCM_STRINGS_H */
  174. /*
  175. Local Variables:
  176. c-file-style: "gnu"
  177. End:
  178. */