diff options
| author | nsfisis <nsfisis@gmail.com> | 2026-08-01 04:52:50 +0900 |
|---|---|---|
| committer | nsfisis <nsfisis@gmail.com> | 2026-08-01 04:52:50 +0900 |
| commit | b8d46b0495d00815a699932ced0b43955c949ab9 (patch) | |
| tree | 17c340f51cbf259ea631ca7da7549714c91b916f /crates/shirabe-php-shim/src/string.rs | |
| parent | 8e8a3c147aa388c4b0fc021a08b30113eb590727 (diff) | |
| download | php-shirabe-b8d46b0495d00815a699932ced0b43955c949ab9.tar.gz php-shirabe-b8d46b0495d00815a699932ced0b43955c949ab9.tar.zst php-shirabe-b8d46b0495d00815a699932ced0b43955c949ab9.zip | |
feat(php-src): add a BSD-licensed crate for php-src derived code
The audit in .ken/php-shim-copying.md judged 14 functions in
shirabe-php-shim (plus php_wordwrap in shirabe-external-packages) to be
line-by-line transcriptions or structural imitations of php-src. PHP's
relicensing to 3-clause BSD makes keeping them legal, but the boundary
between BSD-derived and MIT code was invisible in the source tree.
Moving them into their own crate puts the license into the build
metadata (so NOTICE generation follows the binary), makes a reverse
dependency a compile error, and encodes the origin in the module path,
which mirrors php-src's ext tree. Each function records its origin in a
fixed-format doc comment, and a new php_src_derivation_boundary linter
fails if `php-src` appears in any Rust source outside the crate.
Public paths under shirabe_php_shim:: are unchanged: functions that are
themselves derived are re-exported with `pub use`, and the wrappers that
only validate arguments stay on the MIT side.
This also resolves the duplicate wordwrap implementation.
shirabe_php_shim::wordwrap was todo!(), so SymfonyStyle::block panicked,
while shirabe-external-packages carried its own copy. Both now go
through the single port, verified against real PHP on 13 cases covering
multi-character breaks and cut.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Diffstat (limited to 'crates/shirabe-php-shim/src/string.rs')
| -rw-r--r-- | crates/shirabe-php-shim/src/string.rs | 373 |
1 files changed, 13 insertions, 360 deletions
diff --git a/crates/shirabe-php-shim/src/string.rs b/crates/shirabe-php-shim/src/string.rs index dc566bec..8a96a8c3 100644 --- a/crates/shirabe-php-shim/src/string.rs +++ b/crates/shirabe-php-shim/src/string.rs @@ -1,5 +1,8 @@ use crate::PhpMixed; use indexmap::IndexMap; +pub use shirabe_php_src::standard::string::{addcslashes, strip_tags, stripcslashes}; +use shirabe_php_src::standard::string::{php_trim_mask, php_wordwrap}; +use shirabe_php_src::standard::strnatcmp::strnatcmp_ex; pub fn str_replace(search: &str, replace: &str, subject: &str) -> String { // PHP returns the subject unchanged when the search string is empty, whereas Rust's @@ -223,147 +226,6 @@ pub fn strnatcmp(s1: &str, s2: &str) -> i64 { strnatcmp_ex(s1.as_bytes(), s2.as_bytes(), false) } -// Port of PHP's strnatcmp_ex (ext/standard/strnatcmp.c). Operating on byte -// slices, an out-of-range index reads as 0, reproducing the NUL terminator that -// the C implementation relies on. -fn strnatcmp_ex(a: &[u8], b: &[u8], fold_case: bool) -> i64 { - let a_len = a.len(); - let b_len = b.len(); - if a_len == 0 || b_len == 0 { - return match a_len.cmp(&b_len) { - std::cmp::Ordering::Less => -1, - std::cmp::Ordering::Greater => 1, - std::cmp::Ordering::Equal => 0, - }; - } - - let mut ap = 0usize; - let mut bp = 0usize; - let mut leading = true; - loop { - let mut ca = natcmp_at(a, ap); - let mut cb = natcmp_at(b, bp); - - // Skip over leading zeros. - while leading && ca == b'0' && natcmp_at(a, ap + 1).is_ascii_digit() { - ap += 1; - ca = natcmp_at(a, ap); - } - while leading && cb == b'0' && natcmp_at(b, bp + 1).is_ascii_digit() { - bp += 1; - cb = natcmp_at(b, bp); - } - leading = false; - - // Skip consecutive whitespace. - while natcmp_is_space(ca) { - ap += 1; - ca = natcmp_at(a, ap); - } - while natcmp_is_space(cb) { - bp += 1; - cb = natcmp_at(b, bp); - } - - // Process a run of digits. - if ca.is_ascii_digit() && cb.is_ascii_digit() { - let fractional = ca == b'0' || cb == b'0'; - let result = if fractional { - natcmp_compare_left(a, &mut ap, b, &mut bp) - } else { - natcmp_compare_right(a, &mut ap, b, &mut bp) - }; - if result != 0 { - return result; - } - } - - if ap == a_len && bp == b_len { - return 0; - } else if ap == a_len { - return -1; - } else if bp == b_len { - return 1; - } - - if fold_case { - ca = natcmp_at(a, ap).to_ascii_uppercase(); - cb = natcmp_at(b, bp).to_ascii_uppercase(); - } else { - ca = natcmp_at(a, ap); - cb = natcmp_at(b, bp); - } - - if ca < cb { - return -1; - } else if ca > cb { - return 1; - } - - ap += 1; - bp += 1; - } -} - -fn natcmp_at(s: &[u8], i: usize) -> u8 { - if i < s.len() { s[i] } else { 0 } -} - -fn natcmp_is_space(c: u8) -> bool { - matches!(c, b' ' | b'\t' | b'\n' | 0x0b | 0x0c | b'\r') -} - -// Compare two right-aligned numbers: the longest run of digits wins; failing -// that, the first differing digit decides, but only once magnitudes are known -// equal (tracked in `bias`). -fn natcmp_compare_right(a: &[u8], ap: &mut usize, b: &[u8], bp: &mut usize) -> i64 { - let mut bias = 0i64; - loop { - let ca = natcmp_at(a, *ap); - let cb = natcmp_at(b, *bp); - let a_digit = ca.is_ascii_digit(); - let b_digit = cb.is_ascii_digit(); - if !a_digit && !b_digit { - return bias; - } else if !a_digit { - return -1; - } else if !b_digit { - return 1; - } else if ca < cb { - if bias == 0 { - bias = -1; - } - } else if ca > cb && bias == 0 { - bias = 1; - } - *ap += 1; - *bp += 1; - } -} - -// Compare two left-aligned numbers: the first differing digit decides. -fn natcmp_compare_left(a: &[u8], ap: &mut usize, b: &[u8], bp: &mut usize) -> i64 { - loop { - let ca = natcmp_at(a, *ap); - let cb = natcmp_at(b, *bp); - let a_digit = ca.is_ascii_digit(); - let b_digit = cb.is_ascii_digit(); - if !a_digit && !b_digit { - return 0; - } else if !a_digit { - return -1; - } else if !b_digit { - return 1; - } else if ca < cb { - return -1; - } else if ca > cb { - return 1; - } - *ap += 1; - *bp += 1; - } -} - pub fn strcspn(string: &str, characters: &str) -> usize { let set = characters.as_bytes(); let mut count = 0; @@ -393,29 +255,6 @@ pub fn strstr3(haystack: &str, needle: &str, before_needle: bool) -> Option<Stri /// PHP's default trim character mask: " \t\n\r\0\x0B". const PHP_TRIM_DEFAULT_CHARS: &[u8] = b" \t\n\r\0\x0B"; -/// Build the set of bytes to strip from a PHP trim `$characters` argument, -/// expanding `a..b` range syntax as PHP does. -fn php_trim_mask(chars: &[u8]) -> [bool; 256] { - let mut mask = [false; 256]; - let mut i = 0; - while i < chars.len() { - if i + 3 < chars.len() && chars[i + 1] == b'.' && chars[i + 2] == b'.' { - let start = chars[i]; - let end = chars[i + 3]; - if start <= end { - for b in start..=end { - mask[b as usize] = true; - } - i += 4; - continue; - } - } - mask[chars[i] as usize] = true; - i += 1; - } - mask -} - pub fn rtrim(s: &str, chars: Option<&str>) -> String { let mask = php_trim_mask( chars @@ -1159,87 +998,6 @@ fn php_to_float(v: &PhpMixed) -> f64 { } } -// Port of PHP's php_strip_tags without the allowed-tags parameter (which this signature omits). -// State: 0 = text, 1 = inside a tag, 2 = inside an HTML comment, 3 = inside `<? ... ?>` / `<! ...`. -// TODO(phase-d): this omits allowed-tags handling and the tag-depth counter, so it can diverge from -// PHP on malformed markup (unterminated comments/quotes, nested `<`). -pub fn strip_tags(_str: &str) -> String { - let bytes = _str.as_bytes(); - let n = bytes.len(); - let mut out: Vec<u8> = Vec::with_capacity(n); - let mut state: u8 = 0; - // Quote char while inside a quoted attribute value, or 0. - let mut in_q: u8 = 0; - let mut i = 0; - while i < n { - let c = bytes[i]; - match c { - b'<' => { - if in_q == 0 { - if state == 0 && i + 1 < n && bytes[i + 1].is_ascii_whitespace() { - // PHP keeps "< " (a `<` followed by whitespace) as literal text. - out.push(c); - } else if state == 0 { - state = 1; - } - } - } - b'>' => { - if in_q == 0 { - match state { - 1 | 3 => state = 0, - 2 => { - if i >= 2 && bytes[i - 1] == b'-' && bytes[i - 2] == b'-' { - state = 0; - } - } - _ => out.push(c), - } - } - } - b'"' | b'\'' => { - if state == 1 { - if in_q == 0 { - in_q = c; - } else if in_q == c && !(i > 0 && bytes[i - 1] == b'\\') { - in_q = 0; - } - } else if state == 0 { - out.push(c); - } - } - b'!' => { - if state == 1 && i > 0 && bytes[i - 1] == b'<' { - state = 3; - } else if state == 0 { - out.push(c); - } - } - b'?' => { - if state == 1 && i > 0 && bytes[i - 1] == b'<' { - state = 3; - } else if state == 0 { - out.push(c); - } - } - b'-' => { - if state == 3 && i >= 2 && bytes[i - 1] == b'-' && bytes[i - 2] == b'!' { - state = 2; - } else if state == 0 { - out.push(c); - } - } - _ => { - if state == 0 { - out.push(c); - } - } - } - i += 1; - } - String::from_utf8_lossy(&out).into_owned() -} - pub fn html_entity_decode(_s: &str) -> String { // TODO(phase-d): only numeric entities and the most common named entities (the HTML 4.01 markup // set PHP enables by default) are decoded; the full named-entity table is not ported. @@ -1299,36 +1057,6 @@ pub fn ucfirst(s: &str) -> String { } } -// Port of PHP's addcslashes: every byte that falls in the (range-expanded) charlist is -// backslash-escaped, with non-printable bytes rendered as the C escape or a three-digit octal. -pub fn addcslashes(_string: &str, _charlist: &str) -> String { - let mask = php_trim_mask(_charlist.as_bytes()); - let mut out: Vec<u8> = Vec::with_capacity(_string.len()); - for &c in _string.as_bytes() { - if mask[c as usize] { - if !(32..=126).contains(&c) { - out.push(b'\\'); - match c { - b'\n' => out.push(b'n'), - b'\t' => out.push(b't'), - b'\r' => out.push(b'r'), - 0x07 => out.push(b'a'), - 0x0B => out.push(b'v'), - 0x08 => out.push(b'b'), - 0x0C => out.push(b'f'), - _ => out.extend_from_slice(format!("{:03o}", c).as_bytes()), - } - } else { - out.push(b'\\'); - out.push(c); - } - } else { - out.push(c); - } - } - String::from_utf8_lossy(&out).into_owned() -} - pub fn php_strip_whitespace(path: &str) -> String { // PHP `php_strip_whitespace()` tokenizes the source and re-emits it with comments removed and // each run of whitespace collapsed to a single space. There is no PHP tokenizer in the shim, so @@ -1487,92 +1215,17 @@ pub fn byte_at(s: &str, i: usize) -> u8 { s.as_bytes().get(i).copied().unwrap_or(0) } -// Port of PHP's stripcslashes: the inverse of addcslashes, decoding C escape sequences including -// octal (\ooo) and hex (\xHH). -pub fn stripcslashes(_s: &str) -> String { - let bytes = _s.as_bytes(); - let n = bytes.len(); - let mut out: Vec<u8> = Vec::with_capacity(n); - let mut i = 0; - while i < n { - if bytes[i] == b'\\' && i + 1 < n { - i += 1; - match bytes[i] { - b'n' => { - out.push(b'\n'); - i += 1; - } - b'r' => { - out.push(b'\r'); - i += 1; - } - b'a' => { - out.push(0x07); - i += 1; - } - b't' => { - out.push(b'\t'); - i += 1; - } - b'v' => { - out.push(0x0B); - i += 1; - } - b'b' => { - out.push(0x08); - i += 1; - } - b'f' => { - out.push(0x0C); - i += 1; - } - b'\\' => { - out.push(b'\\'); - i += 1; - } - b'x' => { - if i + 1 < n && bytes[i + 1].is_ascii_hexdigit() { - let mut val: u8 = 0; - let mut count = 0; - i += 1; - while i < n && count < 2 && bytes[i].is_ascii_hexdigit() { - val = val.wrapping_mul(16) + hex_digit_value(bytes[i]).unwrap(); - i += 1; - count += 1; - } - out.push(val); - } else { - out.push(b'x'); - i += 1; - } - } - b'0'..=b'7' => { - let mut val: u8 = 0; - let mut count = 0; - while i < n && count < 3 && (b'0'..=b'7').contains(&bytes[i]) { - val = val.wrapping_mul(8).wrapping_add(bytes[i] - b'0'); - i += 1; - count += 1; - } - out.push(val); - } - other => { - out.push(other); - i += 1; - } - } - } else { - out.push(bytes[i]); - i += 1; - } - } - String::from_utf8_lossy(&out).into_owned() -} - pub fn wordwrap(_s: &str, _width: i64, _break_str: &str, _cut: bool) -> String { - // TODO(phase-d): an exact byte-for-byte port of php_string_wordwrap (with its lastspace/cut - // bookkeeping) is intricate; left unported as it has no current callers. - todo!() + // PHP throws a ValueError for either argument combination before reaching the wrapping loop. + assert!( + !_break_str.is_empty(), + "wordwrap(): Argument #3 ($break) must not be empty" + ); + assert!( + !(_width == 0 && _cut), + "wordwrap(): Argument #4 ($cut) cannot be true when argument #2 ($width) is 0" + ); + php_wordwrap(_s, _width, _break_str, _cut) } pub fn levenshtein(string1: &str, string2: &str) -> i64 { |
