aboutsummaryrefslogtreecommitdiffhomepage
path: root/crates/shirabe-class-map-generator
diff options
context:
space:
mode:
authornsfisis <nsfisis@gmail.com>2026-08-18 01:57:02 +0900
committernsfisis <nsfisis@gmail.com>2026-08-18 01:57:02 +0900
commit6aeda8b237fcbf7a56ca0e8c0fff415477d31d22 (patch)
tree13942695ae0e7c749cfcdb12d5824daf7134c008 /crates/shirabe-class-map-generator
parentfed0a6e7ac361af9b963c1f62411b1a85478230c (diff)
downloadphp-shirabe-6aeda8b237fcbf7a56ca0e8c0fff415477d31d22.tar.gz
php-shirabe-6aeda8b237fcbf7a56ca0e8c0fff415477d31d22.tar.zst
php-shirabe-6aeda8b237fcbf7a56ca0e8c0fff415477d31d22.zip
refactor(preg): make preg_match_all yield matches per occurrence
PHP's PREG_PATTERN_ORDER is column-oriented, but 7 of the 10 call sites read it row-wise, rebuilding each occurrence by indexing every column at the same offset. Return an iterator of PregMatches instead, which is also what the set-order and offset-capture variants were carrying, so the three functions collapse into one and PregMatchesAll, PregMatchesAllWithOffsets, CaptureKey and preg_match_map! all go away. The offset-capture call sites are served by the new PregMatches get_offset/name_offset accessors. The search stays eager: regex::Captures borrows only the subject, so the matches outlive the pattern resolved for the call, and PHP's preg_match_all is eager too. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Diffstat (limited to 'crates/shirabe-class-map-generator')
-rw-r--r--crates/shirabe-class-map-generator/src/php_file_parser.rs40
1 files changed, 10 insertions, 30 deletions
diff --git a/crates/shirabe-class-map-generator/src/php_file_parser.rs b/crates/shirabe-class-map-generator/src/php_file_parser.rs
index 7cc96228..b7fb595c 100644
--- a/crates/shirabe-class-map-generator/src/php_file_parser.rs
+++ b/crates/shirabe-class-map-generator/src/php_file_parser.rs
@@ -2,9 +2,9 @@
use crate::php_file_cleaner::PhpFileCleaner;
use shirabe_php_shim::{
- CaptureKey, PHP_EOL, RuntimeException, file_exists, file_get_contents, function_exists,
- is_file, is_readable, ltrim, php_strip_whitespace, preg_match_all, str_replace_array, strrpos,
- substr, trim,
+ PHP_EOL, RuntimeException, file_exists, file_get_contents, function_exists, is_file,
+ is_readable, ltrim, php_strip_whitespace, preg_match_all, str_replace_array, strrpos, substr,
+ trim,
};
use std::sync::OnceLock;
@@ -58,7 +58,7 @@ impl PhpFileParser {
// return early if there is no chance of matching anything in this file
let pattern = format!("{{\\b(?:class|interface|trait{})\\s}}i", extra_types);
- let max_matches = preg_match_all(&pattern, &contents).occurrence_count();
+ let max_matches = preg_match_all(&pattern, &contents).count();
if max_matches == 0 {
return Ok(vec![]);
}
@@ -89,21 +89,10 @@ impl PhpFileParser {
let mut classes = vec![];
let mut namespace = String::new();
- let len = matches
- .get(&CaptureKey::ByName("type".to_owned()))
- .map(|v| v.len())
- .unwrap_or(0);
- for i in 0..len {
- let ns = matches
- .get(&CaptureKey::ByName("ns".to_owned()))
- .and_then(|v| v.get(i))
- .and_then(|s| s.as_deref());
+ for r#match in matches {
+ let ns = r#match.name("ns");
if ns.is_some_and(|ns| !ns.is_empty()) {
- let nsname = matches
- .get(&CaptureKey::ByName("nsname".to_owned()))
- .and_then(|v| v.get(i))
- .and_then(|s| s.as_deref())
- .unwrap_or("");
+ let nsname = r#match.name("nsname").unwrap_or("");
namespace = str_replace_array(
&[
" ".to_string(),
@@ -115,10 +104,8 @@ impl PhpFileParser {
nsname,
) + "\\";
} else {
- let name = matches
- .get(&CaptureKey::ByName("name".to_owned()))
- .and_then(|v| v.get(i))
- .and_then(|s| s.as_deref())
+ let name = r#match
+ .name("name")
.expect("the `name` group participates whenever `ns` does not");
// skip anon classes extending/implementing
if name == "extends" {
@@ -136,14 +123,7 @@ impl PhpFileParser {
&["_".to_string(), "__".to_string()],
stripped,
)
- } else if matches
- .get(&CaptureKey::ByName("type".to_owned()))
- .and_then(|v| v.get(i))
- .and_then(|s| s.as_deref())
- .unwrap_or("")
- .to_lowercase()
- == "enum"
- {
+ } else if r#match.name("type").unwrap_or("").to_lowercase() == "enum" {
// something like:
// enum Foo: int { HERP = '123'; }
// The regex above captures the colon, which isn't part of