diff options
| author | nsfisis <nsfisis@gmail.com> | 2026-06-25 14:39:00 +0900 |
|---|---|---|
| committer | nsfisis <nsfisis@gmail.com> | 2026-06-25 23:47:47 +0900 |
| commit | d0336078c5b63b174e7313d54d973a1832228928 (patch) | |
| tree | b0ed23c382e7ffd854fd3afb266a6932002e3335 /crates/shirabe-external-packages/src/symfony/string | |
| parent | eebba7ebad103a2f7afe885a25ba2e96efddbd89 (diff) | |
| download | php-shirabe-d0336078c5b63b174e7313d54d973a1832228928.tar.gz php-shirabe-d0336078c5b63b174e7313d54d973a1832228928.tar.zst php-shirabe-d0336078c5b63b174e7313d54d973a1832228928.zip | |
feat(external-packages,shim): implement impl todos across components
Port seld/jsonlint JsonParser (+hand-written Lexer), unblocking 10 json_file
parse-error tests verified byte-for-byte against PHP. Implement Symfony Finder
SplFileInfo, executable finders, String classes (byte/code-point/unicode),
ZipArchive shim (via the zip crate), SPDX license validation, and shim
date/stream functions. Genuinely-blocked sites (reflection, PHP runtime
constants, non-UTF-8 transcoding, recursive PCRE) stay todo!() with reasons.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Diffstat (limited to 'crates/shirabe-external-packages/src/symfony/string')
4 files changed, 196 insertions, 14 deletions
diff --git a/crates/shirabe-external-packages/src/symfony/string/byte_string.rs b/crates/shirabe-external-packages/src/symfony/string/byte_string.rs index ab17c1e..cd8eae8 100644 --- a/crates/shirabe-external-packages/src/symfony/string/byte_string.rs +++ b/crates/shirabe-external-packages/src/symfony/string/byte_string.rs @@ -8,11 +8,25 @@ pub struct ByteString { } impl ByteString { - pub fn new(_string: &str) -> Self { - todo!() + pub fn new(string: &str) -> Self { + Self { + string: string.to_string(), + } } - pub fn to_code_point_string(&self, _encoding: &str) -> CodePointString { + pub fn to_code_point_string(&self, from_encoding: &str) -> CodePointString { + // The source `string` is always valid UTF-8 (Rust `String`/`&str` guarantee). PHP takes the + // early-return branch whenever `preg_match('//u', ...)` holds for a UTF-8/null encoding, so + // the result mirrors the input bytes verbatim. The `mb_detect_encoding`/`iconv` conversion + // path only applies to genuinely non-UTF-8 byte strings, which cannot occur here. + if matches!(from_encoding, "" | "utf8" | "utf-8" | "UTF8" | "UTF-8") { + return CodePointString { + string: self.string.clone(), + }; + } + + // TODO(phase-d): non-UTF-8 source encodings would require mb_convert_encoding/iconv-style + // decoding, which is unreachable for the UTF-8-only inputs Shirabe currently produces. todo!() } } diff --git a/crates/shirabe-external-packages/src/symfony/string/code_point_string.rs b/crates/shirabe-external-packages/src/symfony/string/code_point_string.rs index b76f157..ce67428 100644 --- a/crates/shirabe-external-packages/src/symfony/string/code_point_string.rs +++ b/crates/shirabe-external-packages/src/symfony/string/code_point_string.rs @@ -6,11 +6,166 @@ pub struct CodePointString { } impl CodePointString { - pub fn wordwrap(&self, _width: i64, _break: &str, _cut: bool) -> Self { - todo!() + /// Port of `AbstractString::wordwrap()`, specialised to the non-`ignoreCase` code-point case. + pub fn wordwrap(&self, width: i64, r#break: &str, cut: bool) -> Self { + // `split($break)` with no flags reduces to `explode($break, $string)` here, then `chunk()` + // yields one entry per code point. `ignoreCase` is always false for freshly built instances. + let lines: Vec<&str> = if !r#break.is_empty() { + self.string.split(r#break).collect() + } else { + vec![&self.string] + }; + + let mut chars: Vec<String> = Vec::new(); + let mut mask = String::new(); + + if lines.len() == 1 && lines[0].is_empty() { + return Self { + string: String::new(), + }; + } + + for (i, line) in lines.iter().enumerate() { + if i != 0 { + chars.push(r#break.to_string()); + mask.push('#'); + } + + for ch in line.chars() { + let s = ch.to_string(); + mask.push(if s == " " { ' ' } else { '?' }); + chars.push(s); + } + } + + let mut string = String::new(); + let mut j: usize = 0; + // PHP seeds both `$b` and `$i` at -1; mirror with signed indices. + let mut i: i64 = -1; + let mask = php_wordwrap(&mask, width, "#", cut); + let mask_bytes = mask.as_bytes(); + + let mut b: i64 = -1; + loop { + // strpos($mask, '#', $b + 1) + let from = (b + 1) as usize; + let Some(rel) = mask_bytes[from..].iter().position(|&c| c == b'#') else { + break; + }; + b = (from + rel) as i64; + + i += 1; + while i < b { + string.push_str(&chars[j]); + j += 1; + i += 1; + } + + if chars[j] == r#break || chars[j] == " " { + j += 1; + } + + string.push_str(r#break); + } + + for c in &chars[j..] { + string.push_str(c); + } + + Self { string } } - pub fn to_byte_string(&self, _encoding: &str) -> String { + pub fn to_byte_string(&self, to_encoding: &str) -> String { + // The source is always valid UTF-8, so PHP's `toByteString` returns the string verbatim + // whenever the target is null/UTF-8 (the only encodings reached here). The + // mb_convert_encoding/iconv path applies only to non-UTF-8 targets, which do not occur. + if matches!(to_encoding, "" | "utf8" | "utf-8" | "UTF8" | "UTF-8") { + return self.string.clone(); + } + + // TODO(phase-d): converting to a non-UTF-8 target encoding needs mb_convert_encoding/iconv, + // unreachable for Shirabe's UTF-8-only output. todo!() } } + +/// Port of PHP's built-in `wordwrap()` (`PHP_FUNCTION(wordwrap)` in ext/standard/string.c). +/// Byte-based, matching PHP's single-byte/multi-byte break and cut handling. +fn php_wordwrap(text: &str, linelength: i64, breakchar: &str, docut: bool) -> String { + let text = text.as_bytes(); + let breakchar = breakchar.as_bytes(); + let textlen = text.len() as i64; + let breaklen = breakchar.len() as i64; + + if textlen == 0 { + return String::new(); + } + + let mut laststart: i64 = 0; + let mut lastspace: i64 = 0; + + // Special case for a single-character break that needs no extra storage. + if breaklen == 1 && !docut { + let mut out = text.to_vec(); + let mut current = 0i64; + while current < textlen { + let c = out[current as usize]; + if c == breakchar[0] { + laststart = current + 1; + lastspace = current + 1; + } else if c == b' ' { + if current - laststart >= linelength { + out[current as usize] = breakchar[0]; + laststart = current + 1; + } + lastspace = current; + } else if current - laststart >= linelength && laststart != lastspace { + out[lastspace as usize] = breakchar[0]; + laststart = lastspace + 1; + } + current += 1; + } + return String::from_utf8_lossy(&out).into_owned(); + } + + // Multiple character line break or forced cut. + let mut out: Vec<u8> = Vec::new(); + let mut current = 0i64; + while current < textlen { + // When we hit an existing break, copy to the new buffer and fix up laststart/lastspace. + if text[current as usize] == breakchar[0] + && current + breaklen < textlen + && &text[current as usize..(current + breaklen) as usize] == breakchar + { + out.extend_from_slice(&text[laststart as usize..(current + breaklen) as usize]); + current += breaklen - 1; + laststart = current + 1; + lastspace = current + 1; + } else if text[current as usize] == b' ' { + if current - laststart >= linelength { + out.extend_from_slice(&text[laststart as usize..current as usize]); + out.extend_from_slice(breakchar); + laststart = current + 1; + } + lastspace = current; + } else if current - laststart >= linelength && docut && laststart >= lastspace { + out.extend_from_slice(&text[laststart as usize..current as usize]); + out.extend_from_slice(breakchar); + laststart = current; + lastspace = current; + } else if current - laststart >= linelength && laststart < lastspace { + out.extend_from_slice(&text[laststart as usize..lastspace as usize]); + out.extend_from_slice(breakchar); + laststart = lastspace + 1; + lastspace = lastspace + 1; + } + current += 1; + } + + // Copy over any stragglers. + if laststart != current { + out.extend_from_slice(&text[laststart as usize..current as usize]); + } + + String::from_utf8_lossy(&out).into_owned() +} diff --git a/crates/shirabe-external-packages/src/symfony/string/mod.rs b/crates/shirabe-external-packages/src/symfony/string/mod.rs index 411a3f3..139d726 100644 --- a/crates/shirabe-external-packages/src/symfony/string/mod.rs +++ b/crates/shirabe-external-packages/src/symfony/string/mod.rs @@ -6,11 +6,12 @@ pub use byte_string::*; pub use code_point_string::*; pub use unicode_string::*; -/// Mirror of Symfony's `u()` / `b()` helper functions. -pub fn b(_string: &str) -> ByteString { - todo!() +/// Mirror of Symfony's `b()` helper function. +pub fn b(string: &str) -> ByteString { + ByteString::new(string) } -pub fn s(_string: &str) -> UnicodeString { - todo!() +/// Mirror of Symfony's `u()` helper function. +pub fn s(string: &str) -> UnicodeString { + UnicodeString::new(string) } diff --git a/crates/shirabe-external-packages/src/symfony/string/unicode_string.rs b/crates/shirabe-external-packages/src/symfony/string/unicode_string.rs index 8917f20..e415846 100644 --- a/crates/shirabe-external-packages/src/symfony/string/unicode_string.rs +++ b/crates/shirabe-external-packages/src/symfony/string/unicode_string.rs @@ -34,11 +34,23 @@ impl UnicodeString { width } + // TODO(phase-d): the faithful `length()` uses `grapheme_strlen` (extended grapheme clusters), + // which needs Unicode segmentation tables with no Rust std equivalent and no permitted crate. + // Approximated with the code-point count, exact only when no combining/multi-code-point + // clusters are present (e.g. ASCII). pub fn length(&self) -> i64 { - todo!() + shirabe_php_shim::mb_strlen(&self.string, "UTF-8") } - pub fn slice(&self, _start: i64, _length: Option<i64>) -> Self { - todo!() + // TODO(phase-d): the faithful `slice()` uses `grapheme_substr` (grapheme-cluster offsets), which + // needs Unicode segmentation tables with no std equivalent and no permitted crate. Approximated + // with code-point offsets via `mb_substr`, exact only without combining/multi-code-point clusters. + pub fn slice(&self, start: i64, length: Option<i64>) -> Self { + Self::new(&shirabe_php_shim::mb_substr( + &self.string, + start, + length, + Some("UTF-8"), + )) } } |
