diff options
| author | nsfisis <nsfisis@gmail.com> | 2026-08-09 11:13:06 +0900 |
|---|---|---|
| committer | nsfisis <nsfisis@gmail.com> | 2026-08-09 11:13:06 +0900 |
| commit | 880ba0fd1d05faf98588cc326b3d9fbe625ebf2b (patch) | |
| tree | 055341e80b6297c59cfb267863e0a21419aa6ac3 /crates/shirabe-symfony-string/src/unicode_string.rs | |
| parent | 66c3eba15ba6302d43de057a9063f7feee8c6fb3 (diff) | |
| download | php-shirabe-880ba0fd1d05faf98588cc326b3d9fbe625ebf2b.tar.gz php-shirabe-880ba0fd1d05faf98588cc326b3d9fbe625ebf2b.tar.zst php-shirabe-880ba0fd1d05faf98588cc326b3d9fbe625ebf2b.zip | |
refactor(symfony-string): extract symfony/string into the shirabe-symfony-string crate
Move `Symfony\Component\String` out of shirabe-external-packages and into
its own crate, so the path is `shirabe_symfony_string::ByteString`
instead of `shirabe_external_packages::symfony::string::ByteString`.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Diffstat (limited to 'crates/shirabe-symfony-string/src/unicode_string.rs')
| -rw-r--r-- | crates/shirabe-symfony-string/src/unicode_string.rs | 56 |
1 files changed, 56 insertions, 0 deletions
diff --git a/crates/shirabe-symfony-string/src/unicode_string.rs b/crates/shirabe-symfony-string/src/unicode_string.rs new file mode 100644 index 00000000..9048f77b --- /dev/null +++ b/crates/shirabe-symfony-string/src/unicode_string.rs @@ -0,0 +1,56 @@ +//! ref: composer/vendor/symfony/string/UnicodeString.php + +#[derive(Debug, Clone)] +pub struct UnicodeString { + pub(crate) string: String, +} + +impl UnicodeString { + // TODO(phase-c): the real constructor runs `normalizer_normalize` (Unicode NFC normalization), + // which has no Rust std equivalent and would need a dedicated normalization implementation. + // Normalization is skipped here, which is only correct for already-NFC input such as ASCII. + pub fn new(string: &str) -> Self { + Self { + string: string.to_string(), + } + } + + // TODO(phase-c): ASCII-only provisional implementation. The faithful `width()` uses `wcswidth` + // with the Unicode width tables to treat wide characters as width 2 and skip zero-width / + // combining characters; here every character counts as width 1, correct only for ASCII. The + // ANSI/control-character stripping driven by `ignore_ansi_decoration` is likewise not handled. + pub fn width(&self, _ignore_ansi_decoration: bool) -> i64 { + let s = self.string.replace(['\x00', '\x05', '\x07'], ""); + let s = s.replace("\r\n", "\n").replace('\r', "\n"); + + let mut width: i64 = 0; + for line in s.split('\n') { + let line_width = line.chars().count() as i64; + if line_width > width { + width = line_width; + } + } + + width + } + + // TODO(phase-c): the faithful `length()` uses `grapheme_strlen` (extended grapheme clusters), + // which needs Unicode segmentation tables with no Rust std equivalent and no permitted crate. + // Approximated with the code-point count, exact only when no combining/multi-code-point + // clusters are present (e.g. ASCII). + pub fn length(&self) -> i64 { + shirabe_php_shim::mb_strlen(&self.string, "UTF-8") + } + + // TODO(phase-c): the faithful `slice()` uses `grapheme_substr` (grapheme-cluster offsets), which + // needs Unicode segmentation tables with no std equivalent and no permitted crate. Approximated + // with code-point offsets via `mb_substr`, exact only without combining/multi-code-point clusters. + pub fn slice(&self, start: i64, length: Option<i64>) -> Self { + Self::new(&shirabe_php_shim::mb_substr( + &self.string, + start, + length, + Some("UTF-8"), + )) + } +} |
