From b8d46b0495d00815a699932ced0b43955c949ab9 Mon Sep 17 00:00:00 2001 From: nsfisis Date: Sat, 1 Aug 2026 04:52:50 +0900 Subject: feat(php-src): add a BSD-licensed crate for php-src derived code The audit in .ken/php-shim-copying.md judged 14 functions in shirabe-php-shim (plus php_wordwrap in shirabe-external-packages) to be line-by-line transcriptions or structural imitations of php-src. PHP's relicensing to 3-clause BSD makes keeping them legal, but the boundary between BSD-derived and MIT code was invisible in the source tree. Moving them into their own crate puts the license into the build metadata (so NOTICE generation follows the binary), makes a reverse dependency a compile error, and encodes the origin in the module path, which mirrors php-src's ext tree. Each function records its origin in a fixed-format doc comment, and a new php_src_derivation_boundary linter fails if `php-src` appears in any Rust source outside the crate. Public paths under shirabe_php_shim:: are unchanged: functions that are themselves derived are re-exported with `pub use`, and the wrappers that only validate arguments stay on the MIT side. This also resolves the duplicate wordwrap implementation. shirabe_php_shim::wordwrap was todo!(), so SymfonyStyle::block panicked, while shirabe-external-packages carried its own copy. Both now go through the single port, verified against real PHP on 13 cases covering multi-character breaks and cut. Co-Authored-By: Claude Opus 5 (1M context) --- crates/shirabe-php-src/src/standard/exec.rs | 40 +++ crates/shirabe-php-src/src/standard/string.rs | 317 ++++++++++++++++++++++ crates/shirabe-php-src/src/standard/strnatcmp.rs | 146 ++++++++++ crates/shirabe-php-src/src/standard/versioning.rs | 116 ++++++++ 4 files changed, 619 insertions(+) create mode 100644 crates/shirabe-php-src/src/standard/exec.rs create mode 100644 crates/shirabe-php-src/src/standard/string.rs create mode 100644 crates/shirabe-php-src/src/standard/strnatcmp.rs create mode 100644 crates/shirabe-php-src/src/standard/versioning.rs (limited to 'crates/shirabe-php-src/src/standard') diff --git a/crates/shirabe-php-src/src/standard/exec.rs b/crates/shirabe-php-src/src/standard/exec.rs new file mode 100644 index 00000000..80ab26ae --- /dev/null +++ b/crates/shirabe-php-src/src/standard/exec.rs @@ -0,0 +1,40 @@ +/// php-src: ext/standard/exec.c `php_escape_shell_cmd` (PHP 8.5.2) +/// +/// Unix branch only. Shell metacharacters are backslash-escaped; quote characters are escaped +/// only when unpaired, paired quotes being left intact. The multibyte skip (`php_mblen`), the +/// command length check and the Windows branch are not ported. +pub fn escapeshellcmd(command: &str) -> String { + let bytes = command.as_bytes(); + let len = bytes.len(); + let mut out: Vec = Vec::with_capacity(len); + // Byte index of the matching closing quote while inside a paired quote run. + let mut paired: Option = None; + let mut x = 0; + while x < len { + let c = bytes[x]; + match c { + b'"' | b'\'' => { + if paired.is_none() { + if let Some(rel) = bytes[x + 1..].iter().position(|&b| b == c) { + paired = Some(x + 1 + rel); + } else { + out.push(b'\\'); + } + } else if paired == Some(x) { + paired = None; + } else { + out.push(b'\\'); + } + out.push(c); + } + b'#' | b'&' | b';' | b'`' | b'|' | b'*' | b'?' | b'~' | b'<' | b'>' | b'^' | b'(' + | b')' | b'[' | b']' | b'{' | b'}' | b'$' | b'\\' | 0x0A | 0xFF => { + out.push(b'\\'); + out.push(c); + } + _ => out.push(c), + } + x += 1; + } + String::from_utf8_lossy(&out).into_owned() +} diff --git a/crates/shirabe-php-src/src/standard/string.rs b/crates/shirabe-php-src/src/standard/string.rs new file mode 100644 index 00000000..5af60f5f --- /dev/null +++ b/crates/shirabe-php-src/src/standard/string.rs @@ -0,0 +1,317 @@ +/// php-src: ext/standard/string.c `php_charmask` (PHP 8.5.2) +/// +/// Build the set of bytes to strip from a PHP trim `$characters` argument, expanding `a..b` +/// range syntax as PHP does. The original's four warning branches for malformed `..` ranges are +/// not ported. +pub fn php_trim_mask(chars: &[u8]) -> [bool; 256] { + let mut mask = [false; 256]; + let mut i = 0; + while i < chars.len() { + if i + 3 < chars.len() && chars[i + 1] == b'.' && chars[i + 2] == b'.' { + let start = chars[i]; + let end = chars[i + 3]; + if start <= end { + for b in start..=end { + mask[b as usize] = true; + } + i += 4; + continue; + } + } + mask[chars[i] as usize] = true; + i += 1; + } + mask +} + +/// php-src: ext/standard/string.c `php_addcslashes_str` (PHP 8.5.2) +/// +/// Every byte that falls in the (range-expanded) charlist is backslash-escaped, with +/// non-printable bytes rendered as the C escape or a three-digit octal. +pub fn addcslashes(_string: &str, _charlist: &str) -> String { + let mask = php_trim_mask(_charlist.as_bytes()); + let mut out: Vec = Vec::with_capacity(_string.len()); + for &c in _string.as_bytes() { + if mask[c as usize] { + if !(32..=126).contains(&c) { + out.push(b'\\'); + match c { + b'\n' => out.push(b'n'), + b'\t' => out.push(b't'), + b'\r' => out.push(b'r'), + 0x07 => out.push(b'a'), + 0x0B => out.push(b'v'), + 0x08 => out.push(b'b'), + 0x0C => out.push(b'f'), + _ => out.extend_from_slice(format!("{:03o}", c).as_bytes()), + } + } else { + out.push(b'\\'); + out.push(c); + } + } else { + out.push(c); + } + } + String::from_utf8_lossy(&out).into_owned() +} + +/// php-src: ext/standard/string.c `php_stripcslashes` (PHP 8.5.2) +/// +/// The inverse of `addcslashes`, decoding C escape sequences including octal (\ooo) and hex +/// (\xHH). +pub fn stripcslashes(_s: &str) -> String { + let bytes = _s.as_bytes(); + let n = bytes.len(); + let mut out: Vec = Vec::with_capacity(n); + let mut i = 0; + while i < n { + if bytes[i] == b'\\' && i + 1 < n { + i += 1; + match bytes[i] { + b'n' => { + out.push(b'\n'); + i += 1; + } + b'r' => { + out.push(b'\r'); + i += 1; + } + b'a' => { + out.push(0x07); + i += 1; + } + b't' => { + out.push(b'\t'); + i += 1; + } + b'v' => { + out.push(0x0B); + i += 1; + } + b'b' => { + out.push(0x08); + i += 1; + } + b'f' => { + out.push(0x0C); + i += 1; + } + b'\\' => { + out.push(b'\\'); + i += 1; + } + b'x' => { + if i + 1 < n && bytes[i + 1].is_ascii_hexdigit() { + let mut val: u8 = 0; + let mut count = 0; + i += 1; + while i < n && count < 2 && bytes[i].is_ascii_hexdigit() { + val = val.wrapping_mul(16) + hex_digit_value(bytes[i]).unwrap(); + i += 1; + count += 1; + } + out.push(val); + } else { + out.push(b'x'); + i += 1; + } + } + b'0'..=b'7' => { + let mut val: u8 = 0; + let mut count = 0; + while i < n && count < 3 && (b'0'..=b'7').contains(&bytes[i]) { + val = val.wrapping_mul(8).wrapping_add(bytes[i] - b'0'); + i += 1; + count += 1; + } + out.push(val); + } + other => { + out.push(other); + i += 1; + } + } + } else { + out.push(bytes[i]); + i += 1; + } + } + String::from_utf8_lossy(&out).into_owned() +} + +fn hex_digit_value(b: u8) -> Option { + match b { + b'0'..=b'9' => Some(b - b'0'), + b'a'..=b'f' => Some(b - b'a' + 10), + b'A'..=b'F' => Some(b - b'A' + 10), + _ => None, + } +} + +/// php-src: ext/standard/string.c `php_strip_tags_ex` (PHP 8.5.2) +/// +/// The allowed-tags parameter is omitted from this signature. +/// State: 0 = text, 1 = inside a tag, 2 = inside an HTML comment, 3 = inside `` / ` String { + let bytes = _str.as_bytes(); + let n = bytes.len(); + let mut out: Vec = Vec::with_capacity(n); + let mut state: u8 = 0; + // Quote char while inside a quoted attribute value, or 0. + let mut in_q: u8 = 0; + let mut i = 0; + while i < n { + let c = bytes[i]; + match c { + b'<' => { + if in_q == 0 { + if state == 0 && i + 1 < n && bytes[i + 1].is_ascii_whitespace() { + // PHP keeps "< " (a `<` followed by whitespace) as literal text. + out.push(c); + } else if state == 0 { + state = 1; + } + } + } + b'>' => { + if in_q == 0 { + match state { + 1 | 3 => state = 0, + 2 => { + if i >= 2 && bytes[i - 1] == b'-' && bytes[i - 2] == b'-' { + state = 0; + } + } + _ => out.push(c), + } + } + } + b'"' | b'\'' => { + if state == 1 { + if in_q == 0 { + in_q = c; + } else if in_q == c && !(i > 0 && bytes[i - 1] == b'\\') { + in_q = 0; + } + } else if state == 0 { + out.push(c); + } + } + b'!' => { + if state == 1 && i > 0 && bytes[i - 1] == b'<' { + state = 3; + } else if state == 0 { + out.push(c); + } + } + b'?' => { + if state == 1 && i > 0 && bytes[i - 1] == b'<' { + state = 3; + } else if state == 0 { + out.push(c); + } + } + b'-' => { + if state == 3 && i >= 2 && bytes[i - 1] == b'-' && bytes[i - 2] == b'!' { + state = 2; + } else if state == 0 { + out.push(c); + } + } + _ => { + if state == 0 { + out.push(c); + } + } + } + i += 1; + } + String::from_utf8_lossy(&out).into_owned() +} + +/// php-src: ext/standard/string.c `PHP_FUNCTION(wordwrap)` (PHP 8.5.2) +/// +/// Byte-based, matching PHP's single-byte/multi-byte break and cut handling. The original's +/// output buffer pre-allocation and growth (`chk` / `alloced` / `newtextlen`) is replaced by +/// pushing onto a `Vec`. Argument validation stays with the caller. +pub fn php_wordwrap(text: &str, linelength: i64, breakchar: &str, docut: bool) -> String { + let text = text.as_bytes(); + let breakchar = breakchar.as_bytes(); + let textlen = text.len() as i64; + let breaklen = breakchar.len() as i64; + + if textlen == 0 { + return String::new(); + } + + let mut laststart: i64 = 0; + let mut lastspace: i64 = 0; + + // Special case for a single-character break that needs no extra storage. + if breaklen == 1 && !docut { + let mut out = text.to_vec(); + let mut current = 0i64; + while current < textlen { + let c = out[current as usize]; + if c == breakchar[0] { + laststart = current + 1; + lastspace = current + 1; + } else if c == b' ' { + if current - laststart >= linelength { + out[current as usize] = breakchar[0]; + laststart = current + 1; + } + lastspace = current; + } else if current - laststart >= linelength && laststart != lastspace { + out[lastspace as usize] = breakchar[0]; + laststart = lastspace + 1; + } + current += 1; + } + return String::from_utf8_lossy(&out).into_owned(); + } + + // Multiple character line break or forced cut. + let mut out: Vec = Vec::new(); + let mut current = 0i64; + while current < textlen { + // When we hit an existing break, copy to the new buffer and fix up laststart/lastspace. + if text[current as usize] == breakchar[0] + && current + breaklen < textlen + && &text[current as usize..(current + breaklen) as usize] == breakchar + { + out.extend_from_slice(&text[laststart as usize..(current + breaklen) as usize]); + current += breaklen - 1; + laststart = current + 1; + lastspace = current + 1; + } else if text[current as usize] == b' ' { + if current - laststart >= linelength { + out.extend_from_slice(&text[laststart as usize..current as usize]); + out.extend_from_slice(breakchar); + laststart = current + 1; + } + lastspace = current; + } else if current - laststart >= linelength && docut && laststart >= lastspace { + out.extend_from_slice(&text[laststart as usize..current as usize]); + out.extend_from_slice(breakchar); + laststart = current; + lastspace = current; + } else if current - laststart >= linelength && laststart < lastspace { + out.extend_from_slice(&text[laststart as usize..lastspace as usize]); + out.extend_from_slice(breakchar); + laststart = lastspace + 1; + lastspace += 1; + } + current += 1; + } + + // Copy over any stragglers. + if laststart != current { + out.extend_from_slice(&text[laststart as usize..current as usize]); + } + + String::from_utf8_lossy(&out).into_owned() +} diff --git a/crates/shirabe-php-src/src/standard/strnatcmp.rs b/crates/shirabe-php-src/src/standard/strnatcmp.rs new file mode 100644 index 00000000..30f4176a --- /dev/null +++ b/crates/shirabe-php-src/src/standard/strnatcmp.rs @@ -0,0 +1,146 @@ +/// php-src: ext/standard/strnatcmp.c `strnatcmp_ex` (PHP 8.5.2) +/// +/// Operating on byte slices, an out-of-range index reads as 0, reproducing the NUL terminator +/// that the C implementation relies on. +pub fn strnatcmp_ex(a: &[u8], b: &[u8], fold_case: bool) -> i64 { + let a_len = a.len(); + let b_len = b.len(); + if a_len == 0 || b_len == 0 { + return match a_len.cmp(&b_len) { + std::cmp::Ordering::Less => -1, + std::cmp::Ordering::Greater => 1, + std::cmp::Ordering::Equal => 0, + }; + } + + let mut ap = 0usize; + let mut bp = 0usize; + let mut leading = true; + loop { + let mut ca = natcmp_at(a, ap); + let mut cb = natcmp_at(b, bp); + + // Skip over leading zeros. + while leading && ca == b'0' && natcmp_at(a, ap + 1).is_ascii_digit() { + ap += 1; + ca = natcmp_at(a, ap); + } + while leading && cb == b'0' && natcmp_at(b, bp + 1).is_ascii_digit() { + bp += 1; + cb = natcmp_at(b, bp); + } + leading = false; + + // Skip consecutive whitespace. + while natcmp_is_space(ca) { + ap += 1; + ca = natcmp_at(a, ap); + } + while natcmp_is_space(cb) { + bp += 1; + cb = natcmp_at(b, bp); + } + + // Process a run of digits. + if ca.is_ascii_digit() && cb.is_ascii_digit() { + let fractional = ca == b'0' || cb == b'0'; + let result = if fractional { + natcmp_compare_left(a, &mut ap, b, &mut bp) + } else { + natcmp_compare_right(a, &mut ap, b, &mut bp) + }; + if result != 0 { + return result; + } + } + + if ap == a_len && bp == b_len { + return 0; + } else if ap == a_len { + return -1; + } else if bp == b_len { + return 1; + } + + if fold_case { + ca = natcmp_at(a, ap).to_ascii_uppercase(); + cb = natcmp_at(b, bp).to_ascii_uppercase(); + } else { + ca = natcmp_at(a, ap); + cb = natcmp_at(b, bp); + } + + if ca < cb { + return -1; + } else if ca > cb { + return 1; + } + + ap += 1; + bp += 1; + } +} + +/// php-src: ext/standard/strnatcmp.c (no direct counterpart) (PHP 8.5.2) +fn natcmp_at(s: &[u8], i: usize) -> u8 { + if i < s.len() { s[i] } else { 0 } +} + +/// php-src: ext/standard/strnatcmp.c `isspace(3)` usage (PHP 8.5.2) +fn natcmp_is_space(c: u8) -> bool { + matches!(c, b' ' | b'\t' | b'\n' | 0x0b | 0x0c | b'\r') +} + +/// php-src: ext/standard/strnatcmp.c `compare_right` (PHP 8.5.2) +/// +/// Compare two right-aligned numbers: the longest run of digits wins; failing that, the first +/// differing digit decides, but only once magnitudes are known equal (tracked in `bias`). +fn natcmp_compare_right(a: &[u8], ap: &mut usize, b: &[u8], bp: &mut usize) -> i64 { + let mut bias = 0i64; + loop { + let ca = natcmp_at(a, *ap); + let cb = natcmp_at(b, *bp); + let a_digit = ca.is_ascii_digit(); + let b_digit = cb.is_ascii_digit(); + if !a_digit && !b_digit { + return bias; + } else if !a_digit { + return -1; + } else if !b_digit { + return 1; + } else if ca < cb { + if bias == 0 { + bias = -1; + } + } else if ca > cb && bias == 0 { + bias = 1; + } + *ap += 1; + *bp += 1; + } +} + +/// php-src: ext/standard/strnatcmp.c `compare_left` (PHP 8.5.2) +/// +/// Compare two left-aligned numbers: the first differing digit decides. +fn natcmp_compare_left(a: &[u8], ap: &mut usize, b: &[u8], bp: &mut usize) -> i64 { + loop { + let ca = natcmp_at(a, *ap); + let cb = natcmp_at(b, *bp); + let a_digit = ca.is_ascii_digit(); + let b_digit = cb.is_ascii_digit(); + if !a_digit && !b_digit { + return 0; + } else if !a_digit { + return -1; + } else if !b_digit { + return 1; + } else if ca < cb { + return -1; + } else if ca > cb { + return 1; + } + *ap += 1; + *bp += 1; + } +} diff --git a/crates/shirabe-php-src/src/standard/versioning.rs b/crates/shirabe-php-src/src/standard/versioning.rs new file mode 100644 index 00000000..27f06309 --- /dev/null +++ b/crates/shirabe-php-src/src/standard/versioning.rs @@ -0,0 +1,116 @@ +/// php-src: ext/standard/versioning.c `php_version_compare` (PHP 8.5.2) +/// +/// Returns -1, 0 or 1. The original walks the canonicalized strings with destructive `.` splits +/// and moving pointers; this splits into a `Vec<&str>` and indexes instead. +pub fn php_version_compare(v1: &str, v2: &str) -> i32 { + if v1.is_empty() || v2.is_empty() { + return match (v1.is_empty(), v2.is_empty()) { + (true, true) => 0, + (false, _) => 1, + (_, false) => -1, + }; + } + let c1 = canonicalize_version(v1); + let c2 = canonicalize_version(v2); + let t1: Vec<&str> = c1.split('.').filter(|s| !s.is_empty()).collect(); + let t2: Vec<&str> = c2.split('.').filter(|s| !s.is_empty()).collect(); + + let mut compare = 0; + let mut i = 0; + while i < t1.len() && i < t2.len() && compare == 0 { + compare = version_token_compare(t1[i], t2[i]); + i += 1; + } + if compare == 0 { + // A leftover numeric token wins; a leftover special form is compared against the implicit + // release baseline ("#", order 4). + if i < t1.len() { + let p = t1[i]; + compare = if p.as_bytes()[0].is_ascii_digit() { + 1 + } else { + special_form_order(p).cmp(&4) as i32 + }; + } else if i < t2.len() { + let p = t2[i]; + compare = if p.as_bytes()[0].is_ascii_digit() { + -1 + } else { + 4.cmp(&special_form_order(p)) as i32 + }; + } + } + compare +} + +/// php-src: ext/standard/versioning.c `php_canonicalize_version` (PHP 8.5.2) +/// +/// Separators (-, _, +, .) collapse to a single '.', and a '.' is inserted at every digit <-> +/// non-digit boundary. The original's `!isalnum(*p)` branch is not ported. +fn canonicalize_version(version: &str) -> String { + let bytes = version.as_bytes(); + if bytes.is_empty() { + return String::new(); + } + let mut q: Vec = Vec::with_capacity(bytes.len() * 2); + q.push(bytes[0]); + for &raw in &bytes[1..] { + let ch = if matches!(raw, b'-' | b'_' | b'+') { + b'.' + } else { + raw + }; + let last = *q.last().unwrap(); + if ch == b'.' { + if last != b'.' { + q.push(b'.'); + } + } else if last.is_ascii_digit() != ch.is_ascii_digit() { + q.push(b'.'); + q.push(ch); + } else { + q.push(ch); + } + } + String::from_utf8_lossy(&q).into_owned() +} + +/// php-src: ext/standard/versioning.c `php_version_compare` loop body (PHP 8.5.2) +fn version_token_compare(t1: &str, t2: &str) -> i32 { + let d1 = t1.as_bytes()[0].is_ascii_digit(); + let d2 = t2.as_bytes()[0].is_ascii_digit(); + if d1 && d2 { + let l1 = t1.parse::().unwrap_or(0); + let l2 = t2.parse::().unwrap_or(0); + l1.cmp(&l2) as i32 + } else if !d1 && !d2 { + special_form_order(t1).cmp(&special_form_order(t2)) as i32 + } else if d1 { + // A numeric token is treated as the "#" form (order 4). + 4.cmp(&special_form_order(t2)) as i32 + } else { + special_form_order(t1).cmp(&4) as i32 + } +} + +/// php-src: ext/standard/versioning.c `compare_special_version_forms` (PHP 8.5.2) +fn special_form_order(form: &str) -> i32 { + const FORMS: &[(&str, i32)] = &[ + ("dev", 0), + ("alpha", 1), + ("a", 1), + ("beta", 2), + ("b", 2), + ("RC", 3), + ("rc", 3), + ("#", 4), + ("pl", 5), + ("p", 5), + ]; + for (name, order) in FORMS { + if form.starts_with(name) { + return *order; + } + } + -1 +} -- cgit v1.3.1