From b8d46b0495d00815a699932ced0b43955c949ab9 Mon Sep 17 00:00:00 2001 From: nsfisis Date: Sat, 1 Aug 2026 04:52:50 +0900 Subject: feat(php-src): add a BSD-licensed crate for php-src derived code The audit in .ken/php-shim-copying.md judged 14 functions in shirabe-php-shim (plus php_wordwrap in shirabe-external-packages) to be line-by-line transcriptions or structural imitations of php-src. PHP's relicensing to 3-clause BSD makes keeping them legal, but the boundary between BSD-derived and MIT code was invisible in the source tree. Moving them into their own crate puts the license into the build metadata (so NOTICE generation follows the binary), makes a reverse dependency a compile error, and encodes the origin in the module path, which mirrors php-src's ext tree. Each function records its origin in a fixed-format doc comment, and a new php_src_derivation_boundary linter fails if `php-src` appears in any Rust source outside the crate. Public paths under shirabe_php_shim:: are unchanged: functions that are themselves derived are re-exported with `pub use`, and the wrappers that only validate arguments stay on the MIT side. This also resolves the duplicate wordwrap implementation. shirabe_php_shim::wordwrap was todo!(), so SymfonyStyle::block panicked, while shirabe-external-packages carried its own copy. Both now go through the single port, verified against real PHP on 13 cases covering multi-character breaks and cut. Co-Authored-By: Claude Opus 5 (1M context) --- crates/shirabe-php-shim/Cargo.toml | 1 + crates/shirabe-php-shim/src/process.rs | 39 +--- crates/shirabe-php-shim/src/runtime.rs | 111 +--------- crates/shirabe-php-shim/src/string.rs | 373 ++------------------------------- 4 files changed, 16 insertions(+), 508 deletions(-) (limited to 'crates/shirabe-php-shim') diff --git a/crates/shirabe-php-shim/Cargo.toml b/crates/shirabe-php-shim/Cargo.toml index d3b64a2c..a0e0ab3b 100644 --- a/crates/shirabe-php-shim/Cargo.toml +++ b/crates/shirabe-php-shim/Cargo.toml @@ -4,6 +4,7 @@ version.workspace = true edition.workspace = true [dependencies] +shirabe-php-src.workspace = true anyhow.workspace = true chrono.workspace = true fastrand.workspace = true diff --git a/crates/shirabe-php-shim/src/process.rs b/crates/shirabe-php-shim/src/process.rs index 4615ff91..57d9b35a 100644 --- a/crates/shirabe-php-shim/src/process.rs +++ b/crates/shirabe-php-shim/src/process.rs @@ -1,5 +1,6 @@ use crate::{ChildPipe, PhpMixed, PhpResource, StreamBacking, StreamState}; use indexmap::IndexMap; +pub use shirabe_php_src::standard::exec::escapeshellcmd; pub const SIGINT: i64 = 2; pub const SIGTERM: i64 = 15; @@ -77,44 +78,6 @@ pub fn system(command: &str, result_code: Option<&mut i64>) -> Option { Some(lines.last().cloned().unwrap_or_default()) } -// Port of PHP's escapeshellcmd (Unix branch). Shell metacharacters are backslash-escaped; quote -// characters are escaped only when unpaired, paired quotes being left intact. -pub fn escapeshellcmd(command: &str) -> String { - let bytes = command.as_bytes(); - let len = bytes.len(); - let mut out: Vec = Vec::with_capacity(len); - // Byte index of the matching closing quote while inside a paired quote run. - let mut paired: Option = None; - let mut x = 0; - while x < len { - let c = bytes[x]; - match c { - b'"' | b'\'' => { - if paired.is_none() { - if let Some(rel) = bytes[x + 1..].iter().position(|&b| b == c) { - paired = Some(x + 1 + rel); - } else { - out.push(b'\\'); - } - } else if paired == Some(x) { - paired = None; - } else { - out.push(b'\\'); - } - out.push(c); - } - b'#' | b'&' | b';' | b'`' | b'|' | b'*' | b'?' | b'~' | b'<' | b'>' | b'^' | b'(' - | b')' | b'[' | b']' | b'{' | b'}' | b'$' | b'\\' | 0x0A | 0xFF => { - out.push(b'\\'); - out.push(c); - } - _ => out.push(c), - } - x += 1; - } - String::from_utf8_lossy(&out).into_owned() -} - // Unix branch of PHP's escapeshellarg: wrap in single quotes, escaping embedded single quotes. pub fn escapeshellarg(arg: &str) -> String { let mut out = String::with_capacity(arg.len() + 2); diff --git a/crates/shirabe-php-shim/src/runtime.rs b/crates/shirabe-php-shim/src/runtime.rs index f65ce871..00104051 100644 --- a/crates/shirabe-php-shim/src/runtime.rs +++ b/crates/shirabe-php-shim/src/runtime.rs @@ -1,5 +1,6 @@ use crate::PhpMixed; use indexmap::IndexMap; +use shirabe_php_src::standard::versioning::php_version_compare; pub const PHP_VERSION_ID: i64 = 80100; pub const PHP_VERSION: &str = "8.1.0"; @@ -276,116 +277,6 @@ pub fn version_compare_2(_v1: &str, _v2: &str) -> i64 { php_version_compare(_v1, _v2) as i64 } -// Port of PHP's php_version_compare (ext/standard/versioning.c). Returns -1, 0 or 1. -fn php_version_compare(v1: &str, v2: &str) -> i32 { - if v1.is_empty() || v2.is_empty() { - return match (v1.is_empty(), v2.is_empty()) { - (true, true) => 0, - (false, _) => 1, - (_, false) => -1, - }; - } - let c1 = canonicalize_version(v1); - let c2 = canonicalize_version(v2); - let t1: Vec<&str> = c1.split('.').filter(|s| !s.is_empty()).collect(); - let t2: Vec<&str> = c2.split('.').filter(|s| !s.is_empty()).collect(); - - let mut compare = 0; - let mut i = 0; - while i < t1.len() && i < t2.len() && compare == 0 { - compare = version_token_compare(t1[i], t2[i]); - i += 1; - } - if compare == 0 { - // A leftover numeric token wins; a leftover special form is compared against the implicit - // release baseline ("#", order 4). - if i < t1.len() { - let p = t1[i]; - compare = if p.as_bytes()[0].is_ascii_digit() { - 1 - } else { - special_form_order(p).cmp(&4) as i32 - }; - } else if i < t2.len() { - let p = t2[i]; - compare = if p.as_bytes()[0].is_ascii_digit() { - -1 - } else { - 4.cmp(&special_form_order(p)) as i32 - }; - } - } - compare -} - -// PHP's php_canonicalize_version: separators (-, _, +, .) collapse to a single '.', and a '.' is -// inserted at every digit <-> non-digit boundary. -fn canonicalize_version(version: &str) -> String { - let bytes = version.as_bytes(); - if bytes.is_empty() { - return String::new(); - } - let mut q: Vec = Vec::with_capacity(bytes.len() * 2); - q.push(bytes[0]); - for &raw in &bytes[1..] { - let ch = if matches!(raw, b'-' | b'_' | b'+') { - b'.' - } else { - raw - }; - let last = *q.last().unwrap(); - if ch == b'.' { - if last != b'.' { - q.push(b'.'); - } - } else if last.is_ascii_digit() != ch.is_ascii_digit() { - q.push(b'.'); - q.push(ch); - } else { - q.push(ch); - } - } - String::from_utf8_lossy(&q).into_owned() -} - -fn version_token_compare(t1: &str, t2: &str) -> i32 { - let d1 = t1.as_bytes()[0].is_ascii_digit(); - let d2 = t2.as_bytes()[0].is_ascii_digit(); - if d1 && d2 { - let l1 = t1.parse::().unwrap_or(0); - let l2 = t2.parse::().unwrap_or(0); - l1.cmp(&l2) as i32 - } else if !d1 && !d2 { - special_form_order(t1).cmp(&special_form_order(t2)) as i32 - } else if d1 { - // A numeric token is treated as the "#" form (order 4). - 4.cmp(&special_form_order(t2)) as i32 - } else { - special_form_order(t1).cmp(&4) as i32 - } -} - -fn special_form_order(form: &str) -> i32 { - const FORMS: &[(&str, i32)] = &[ - ("dev", 0), - ("alpha", 1), - ("a", 1), - ("beta", 2), - ("b", 2), - ("RC", 3), - ("rc", 3), - ("#", 4), - ("pl", 5), - ("p", 5), - ]; - for (name, order) in FORMS { - if form.starts_with(name) { - return *order; - } - } - -1 -} - // TODO(php-runtime): the previous handler should be restored in the PHP runtime. // Paired with set_error_handler, which is a no-op in this shim. pub fn restore_error_handler() {} diff --git a/crates/shirabe-php-shim/src/string.rs b/crates/shirabe-php-shim/src/string.rs index dc566bec..8a96a8c3 100644 --- a/crates/shirabe-php-shim/src/string.rs +++ b/crates/shirabe-php-shim/src/string.rs @@ -1,5 +1,8 @@ use crate::PhpMixed; use indexmap::IndexMap; +pub use shirabe_php_src::standard::string::{addcslashes, strip_tags, stripcslashes}; +use shirabe_php_src::standard::string::{php_trim_mask, php_wordwrap}; +use shirabe_php_src::standard::strnatcmp::strnatcmp_ex; pub fn str_replace(search: &str, replace: &str, subject: &str) -> String { // PHP returns the subject unchanged when the search string is empty, whereas Rust's @@ -223,147 +226,6 @@ pub fn strnatcmp(s1: &str, s2: &str) -> i64 { strnatcmp_ex(s1.as_bytes(), s2.as_bytes(), false) } -// Port of PHP's strnatcmp_ex (ext/standard/strnatcmp.c). Operating on byte -// slices, an out-of-range index reads as 0, reproducing the NUL terminator that -// the C implementation relies on. -fn strnatcmp_ex(a: &[u8], b: &[u8], fold_case: bool) -> i64 { - let a_len = a.len(); - let b_len = b.len(); - if a_len == 0 || b_len == 0 { - return match a_len.cmp(&b_len) { - std::cmp::Ordering::Less => -1, - std::cmp::Ordering::Greater => 1, - std::cmp::Ordering::Equal => 0, - }; - } - - let mut ap = 0usize; - let mut bp = 0usize; - let mut leading = true; - loop { - let mut ca = natcmp_at(a, ap); - let mut cb = natcmp_at(b, bp); - - // Skip over leading zeros. - while leading && ca == b'0' && natcmp_at(a, ap + 1).is_ascii_digit() { - ap += 1; - ca = natcmp_at(a, ap); - } - while leading && cb == b'0' && natcmp_at(b, bp + 1).is_ascii_digit() { - bp += 1; - cb = natcmp_at(b, bp); - } - leading = false; - - // Skip consecutive whitespace. - while natcmp_is_space(ca) { - ap += 1; - ca = natcmp_at(a, ap); - } - while natcmp_is_space(cb) { - bp += 1; - cb = natcmp_at(b, bp); - } - - // Process a run of digits. - if ca.is_ascii_digit() && cb.is_ascii_digit() { - let fractional = ca == b'0' || cb == b'0'; - let result = if fractional { - natcmp_compare_left(a, &mut ap, b, &mut bp) - } else { - natcmp_compare_right(a, &mut ap, b, &mut bp) - }; - if result != 0 { - return result; - } - } - - if ap == a_len && bp == b_len { - return 0; - } else if ap == a_len { - return -1; - } else if bp == b_len { - return 1; - } - - if fold_case { - ca = natcmp_at(a, ap).to_ascii_uppercase(); - cb = natcmp_at(b, bp).to_ascii_uppercase(); - } else { - ca = natcmp_at(a, ap); - cb = natcmp_at(b, bp); - } - - if ca < cb { - return -1; - } else if ca > cb { - return 1; - } - - ap += 1; - bp += 1; - } -} - -fn natcmp_at(s: &[u8], i: usize) -> u8 { - if i < s.len() { s[i] } else { 0 } -} - -fn natcmp_is_space(c: u8) -> bool { - matches!(c, b' ' | b'\t' | b'\n' | 0x0b | 0x0c | b'\r') -} - -// Compare two right-aligned numbers: the longest run of digits wins; failing -// that, the first differing digit decides, but only once magnitudes are known -// equal (tracked in `bias`). -fn natcmp_compare_right(a: &[u8], ap: &mut usize, b: &[u8], bp: &mut usize) -> i64 { - let mut bias = 0i64; - loop { - let ca = natcmp_at(a, *ap); - let cb = natcmp_at(b, *bp); - let a_digit = ca.is_ascii_digit(); - let b_digit = cb.is_ascii_digit(); - if !a_digit && !b_digit { - return bias; - } else if !a_digit { - return -1; - } else if !b_digit { - return 1; - } else if ca < cb { - if bias == 0 { - bias = -1; - } - } else if ca > cb && bias == 0 { - bias = 1; - } - *ap += 1; - *bp += 1; - } -} - -// Compare two left-aligned numbers: the first differing digit decides. -fn natcmp_compare_left(a: &[u8], ap: &mut usize, b: &[u8], bp: &mut usize) -> i64 { - loop { - let ca = natcmp_at(a, *ap); - let cb = natcmp_at(b, *bp); - let a_digit = ca.is_ascii_digit(); - let b_digit = cb.is_ascii_digit(); - if !a_digit && !b_digit { - return 0; - } else if !a_digit { - return -1; - } else if !b_digit { - return 1; - } else if ca < cb { - return -1; - } else if ca > cb { - return 1; - } - *ap += 1; - *bp += 1; - } -} - pub fn strcspn(string: &str, characters: &str) -> usize { let set = characters.as_bytes(); let mut count = 0; @@ -393,29 +255,6 @@ pub fn strstr3(haystack: &str, needle: &str, before_needle: bool) -> Option [bool; 256] { - let mut mask = [false; 256]; - let mut i = 0; - while i < chars.len() { - if i + 3 < chars.len() && chars[i + 1] == b'.' && chars[i + 2] == b'.' { - let start = chars[i]; - let end = chars[i + 3]; - if start <= end { - for b in start..=end { - mask[b as usize] = true; - } - i += 4; - continue; - } - } - mask[chars[i] as usize] = true; - i += 1; - } - mask -} - pub fn rtrim(s: &str, chars: Option<&str>) -> String { let mask = php_trim_mask( chars @@ -1159,87 +998,6 @@ fn php_to_float(v: &PhpMixed) -> f64 { } } -// Port of PHP's php_strip_tags without the allowed-tags parameter (which this signature omits). -// State: 0 = text, 1 = inside a tag, 2 = inside an HTML comment, 3 = inside `` / ` String { - let bytes = _str.as_bytes(); - let n = bytes.len(); - let mut out: Vec = Vec::with_capacity(n); - let mut state: u8 = 0; - // Quote char while inside a quoted attribute value, or 0. - let mut in_q: u8 = 0; - let mut i = 0; - while i < n { - let c = bytes[i]; - match c { - b'<' => { - if in_q == 0 { - if state == 0 && i + 1 < n && bytes[i + 1].is_ascii_whitespace() { - // PHP keeps "< " (a `<` followed by whitespace) as literal text. - out.push(c); - } else if state == 0 { - state = 1; - } - } - } - b'>' => { - if in_q == 0 { - match state { - 1 | 3 => state = 0, - 2 => { - if i >= 2 && bytes[i - 1] == b'-' && bytes[i - 2] == b'-' { - state = 0; - } - } - _ => out.push(c), - } - } - } - b'"' | b'\'' => { - if state == 1 { - if in_q == 0 { - in_q = c; - } else if in_q == c && !(i > 0 && bytes[i - 1] == b'\\') { - in_q = 0; - } - } else if state == 0 { - out.push(c); - } - } - b'!' => { - if state == 1 && i > 0 && bytes[i - 1] == b'<' { - state = 3; - } else if state == 0 { - out.push(c); - } - } - b'?' => { - if state == 1 && i > 0 && bytes[i - 1] == b'<' { - state = 3; - } else if state == 0 { - out.push(c); - } - } - b'-' => { - if state == 3 && i >= 2 && bytes[i - 1] == b'-' && bytes[i - 2] == b'!' { - state = 2; - } else if state == 0 { - out.push(c); - } - } - _ => { - if state == 0 { - out.push(c); - } - } - } - i += 1; - } - String::from_utf8_lossy(&out).into_owned() -} - pub fn html_entity_decode(_s: &str) -> String { // TODO(phase-d): only numeric entities and the most common named entities (the HTML 4.01 markup // set PHP enables by default) are decoded; the full named-entity table is not ported. @@ -1299,36 +1057,6 @@ pub fn ucfirst(s: &str) -> String { } } -// Port of PHP's addcslashes: every byte that falls in the (range-expanded) charlist is -// backslash-escaped, with non-printable bytes rendered as the C escape or a three-digit octal. -pub fn addcslashes(_string: &str, _charlist: &str) -> String { - let mask = php_trim_mask(_charlist.as_bytes()); - let mut out: Vec = Vec::with_capacity(_string.len()); - for &c in _string.as_bytes() { - if mask[c as usize] { - if !(32..=126).contains(&c) { - out.push(b'\\'); - match c { - b'\n' => out.push(b'n'), - b'\t' => out.push(b't'), - b'\r' => out.push(b'r'), - 0x07 => out.push(b'a'), - 0x0B => out.push(b'v'), - 0x08 => out.push(b'b'), - 0x0C => out.push(b'f'), - _ => out.extend_from_slice(format!("{:03o}", c).as_bytes()), - } - } else { - out.push(b'\\'); - out.push(c); - } - } else { - out.push(c); - } - } - String::from_utf8_lossy(&out).into_owned() -} - pub fn php_strip_whitespace(path: &str) -> String { // PHP `php_strip_whitespace()` tokenizes the source and re-emits it with comments removed and // each run of whitespace collapsed to a single space. There is no PHP tokenizer in the shim, so @@ -1487,92 +1215,17 @@ pub fn byte_at(s: &str, i: usize) -> u8 { s.as_bytes().get(i).copied().unwrap_or(0) } -// Port of PHP's stripcslashes: the inverse of addcslashes, decoding C escape sequences including -// octal (\ooo) and hex (\xHH). -pub fn stripcslashes(_s: &str) -> String { - let bytes = _s.as_bytes(); - let n = bytes.len(); - let mut out: Vec = Vec::with_capacity(n); - let mut i = 0; - while i < n { - if bytes[i] == b'\\' && i + 1 < n { - i += 1; - match bytes[i] { - b'n' => { - out.push(b'\n'); - i += 1; - } - b'r' => { - out.push(b'\r'); - i += 1; - } - b'a' => { - out.push(0x07); - i += 1; - } - b't' => { - out.push(b'\t'); - i += 1; - } - b'v' => { - out.push(0x0B); - i += 1; - } - b'b' => { - out.push(0x08); - i += 1; - } - b'f' => { - out.push(0x0C); - i += 1; - } - b'\\' => { - out.push(b'\\'); - i += 1; - } - b'x' => { - if i + 1 < n && bytes[i + 1].is_ascii_hexdigit() { - let mut val: u8 = 0; - let mut count = 0; - i += 1; - while i < n && count < 2 && bytes[i].is_ascii_hexdigit() { - val = val.wrapping_mul(16) + hex_digit_value(bytes[i]).unwrap(); - i += 1; - count += 1; - } - out.push(val); - } else { - out.push(b'x'); - i += 1; - } - } - b'0'..=b'7' => { - let mut val: u8 = 0; - let mut count = 0; - while i < n && count < 3 && (b'0'..=b'7').contains(&bytes[i]) { - val = val.wrapping_mul(8).wrapping_add(bytes[i] - b'0'); - i += 1; - count += 1; - } - out.push(val); - } - other => { - out.push(other); - i += 1; - } - } - } else { - out.push(bytes[i]); - i += 1; - } - } - String::from_utf8_lossy(&out).into_owned() -} - pub fn wordwrap(_s: &str, _width: i64, _break_str: &str, _cut: bool) -> String { - // TODO(phase-d): an exact byte-for-byte port of php_string_wordwrap (with its lastspace/cut - // bookkeeping) is intricate; left unported as it has no current callers. - todo!() + // PHP throws a ValueError for either argument combination before reaching the wrapping loop. + assert!( + !_break_str.is_empty(), + "wordwrap(): Argument #3 ($break) must not be empty" + ); + assert!( + !(_width == 0 && _cut), + "wordwrap(): Argument #4 ($cut) cannot be true when argument #2 ($width) is 0" + ); + php_wordwrap(_s, _width, _break_str, _cut) } pub fn levenshtein(string1: &str, string2: &str) -> i64 { -- cgit v1.3.1