Merge branch 'method_alias' of https://github.com/Asmod4n/mruby into method_alias

This commit is contained in:
Asmod4n
2026-06-04 17:37:12 +02:00
40 changed files with 822 additions and 229 deletions
+3 -3
View File
@@ -22,12 +22,12 @@ jobs:
with:
persist-credentials: false
- name: Initialize CodeQL
uses: github/codeql-action/init@68bde559dea0fdcac2102bfdf6230c5f70eb485e # v4.35.4
uses: github/codeql-action/init@7211b7c8077ea37d8641b6271f6a365a22a5fbfa # v4.36.0
with:
languages: ${{ matrix.language }}
- name: Autobuild
uses: github/codeql-action/autobuild@68bde559dea0fdcac2102bfdf6230c5f70eb485e # v4.35.4
uses: github/codeql-action/autobuild@7211b7c8077ea37d8641b6271f6a365a22a5fbfa # v4.36.0
- name: Perform CodeQL Analysis
uses: github/codeql-action/analyze@68bde559dea0fdcac2102bfdf6230c5f70eb485e # v4.35.4
uses: github/codeql-action/analyze@7211b7c8077ea37d8641b6271f6a365a22a5fbfa # v4.36.0
with:
category: "Security"
+1 -1
View File
@@ -15,7 +15,7 @@ jobs:
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- uses: j178/prek-action@6ad80277337ad479fe43bd70701c3f7f8aa74db3 # v2.0.3
- uses: j178/prek-action@bdca6f102f98e2b4c7029491a53dfd366469e33d # v2.0.4
with:
install-only: true
- name: Run manual pre-commit hooks
+1 -1
View File
@@ -15,6 +15,6 @@ jobs:
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- uses: j178/prek-action@6ad80277337ad479fe43bd70701c3f7f8aa74db3 # v2.0.3
- uses: j178/prek-action@bdca6f102f98e2b4c7029491a53dfd366469e33d # v2.0.4
with:
extra-args: --all-files
+1 -1
View File
@@ -118,7 +118,7 @@ repos:
types: [markdown]
files: \.md$
- repo: https://github.com/rubocop/rubocop
rev: v1.86.1
rev: v1.86.2
hooks:
- id: rubocop
name: run rubocop
+16 -12
View File
@@ -1,25 +1,25 @@
# Authors of mruby (mruby developers)
## The List of Contributors sorted by number of commits (as of 2026-03-02 02877f0)
## The List of Contributors sorted by number of commits (as of 2026-05-31 9d084b0)
7532 Yukihiro "Matz" Matsumoto (@matz)*
712 dearblue (@dearblue)*
7747 Yukihiro "Matz" Matsumoto (@matz)*
749 dearblue (@dearblue)*
587 KOBAYASHI Shuji (@shuujii)
353 Daniel Bovensiepen (@bovi)*
345 Takeshi Watanabe (@take-cheeze)*
333 Masaki Muranaka (@monaka)
255 John Bampton (@jbampton)
266 John Bampton (@jbampton)
234 Jun Hiroe (@suzukaze)
228 Tomoyuki Sahara (@tsahara)*
220 Cremno (@cremno)*
209 Yuki Kurihara (@ksss)+
144 Yasuhiro Matsumoto (@mattn)*
146 Yasuhiro Matsumoto (@mattn)*
113 Carson McDonald (@carsonmcdonald)
104 Tomasz Pędraszewski (@dabroz)*
83 Akira Yumiyama (@akiray03)*
83 skandhas (@skandhas)
80 Masamitsu MURASE (@masamitsu-murase)
73 Hiroshi Mimaki (@mimaki)*
79 Hiroshi Mimaki (@mimaki)*
71 Tatsuhiko Kubo (@cubicdaiya)*
71 Yuichiro MASUI (@masuidrive)
62 Yuichiro Kaneko (@yui-knk)+
@@ -36,8 +36,10 @@
32 Masayoshi Takahashi (@takahashim)+
31 MATSUMOTO Ryosuke (@matsumotory)*
30 Nobuyoshi Nakada (@nobu)
29 HASUMI Hitoshi (@hasumikin)
26 Hoshiumi Arata (@hoshiumiarata)*
25 Julian Aron Prenner (@furunkel)*
23 leviongit (@leviongit)
22 Clayton Smith (@clayton-shopify)
22 Uchio Kondo (@udzura)*
22 Zachary Scott (@zzak)*
@@ -50,10 +52,9 @@
18 Corey Powell (@IceDragon200)
18 Hidetaka Takano (@TJ-Hidetaka-Takano)
18 Jon Maken (@jonforums)+
18 leviongit (@leviongit)
18 mirichi (@mirichi)
17 Mitchell Blank Jr (@mitchblank)*
16 HASUMI Hitoshi (@hasumikin)
16 Hendrik (@Asmod4n)
16 bggd (@bggd)
16 kano4 (@kano4)
15 Felix Jones (@felixjones)*
@@ -76,7 +77,7 @@
11 RIZAL Reckordp (@Reckordp)+
11 Seeker (@SeekingMeaning)
11 takkaw (@takkaw)
10 Hendrik (@Asmod4n)
10 Chris Hasiński (@khasinski)
10 Miura Hideki (@miura1729)
10 Narihiro Nakamura (@authorNari)
10 YAMAMOTO Masaya (pandax381)
@@ -88,6 +89,7 @@
8 Wataru Ashihara (@wataash)*
7 Bhargava Shastry (@bshastry)*
7 Kouichi Nakanishi (@keizo042)
7 Paweł Świątkowski (@katafrakt)
7 Rubyist (@expeditiousRubyist)
7 Simon Génier (@simon-shopify)
7 Terence Lee (@hone)
@@ -101,7 +103,6 @@
6 INOUE Yasuyuki (@yasuyuki)
6 Junji Sawada (@junjis0203)
6 Kenji Okimoto (@okkez)+
6 Paweł Świątkowski (@katafrakt)
6 Selman ULUG (@selman)
6 Yusuke Endoh (@mame)*
6 buty4649 (@buty4649)
@@ -120,7 +121,6 @@
5 dreamedge (@dreamedge)
5 nkshigeru (@nkshigeru)
5 xuejianqing (@joans321)
4 Chris Hasiński (@khasinski)
4 Dante Catalfamo (@dantecatalfamo)
4 Goro Kikuchi (@gorogit)
4 Herwin Weststrate (@herwinw)
@@ -140,6 +140,7 @@
4 Yuji Yamano (@yyamano)
4 kurodash (@kurodash)*
4 wanabe (@wanabe)*
2 0x1eef (@0x1eef)
3 Anton Davydov (@davydovanton)
3 Aurora Nockert (@auroranockert)
3 Carlo Prelz (@asfluido)*
@@ -192,6 +193,7 @@
2 Masahiro Wakame (@vvkame)+
2 Minao Yamamoto (@tarosay)+
2 Nihad Abbasov (@NARKOZ)
2 Pete Kinnecom (@petekinnecom)
2 Robert Mosolgo (@rmosolgo)
2 Russel Hunter Yukawa (@rhykw)+
2 Ryunosuke SATO (@tricknotes)
@@ -220,6 +222,7 @@
1 Colin MacKenzie IV (@sinisterchipmunk)
1 Daehyub Kim (@lateau)
1 Daniel Varga (@vargad)
1 David Korczynski (@DavidKorczynski)
1 Diamond Rivero (@diamant3)
1 Edgar Boda-Majer (@eboda)
1 Fangrui Song (@MaskRay)
@@ -274,7 +277,6 @@
1 Patrick Ellis (@pje)
1 Patrick Pokatilo (@SHyx0rmZ)
1 Pavel Evstigneev (@Paxa)+
1 Pete Kinnecom (@petekinnecom)
1 Piotr Usewicz (@pusewicz)
1 Prayag Verma (@pra85)
1 Ranmocy (@ranmocy)
@@ -282,6 +284,7 @@
1 Ryan Scott Lewis (@RyanScottLewis)
1 Ryo Okubo (@syucream)
1 SAkira a.k.a. Akira Suzuki (@sakisakira)
1 SaekiMototsune (@saeki-mototsune)
1 Santiago Rodriguez (@sanrodari)
1 Satoh, Hiroh (@cho45)+
1 Satoru Naba (@snaba)+
@@ -325,6 +328,7 @@
1 sbsoftware (@sbsoftware)
1 ssmallkirby (@smallkirby)
1 taku toyama (@tsuichu)
1 vobloeb (@vobloeb)
`*` - Entries unified according to names and addresses
`+` - Entries with names different from commits
+1 -1
View File
@@ -3,7 +3,7 @@ GEM
specs:
coderay (1.1.3)
rake (13.4.2)
yard (0.9.43)
yard (0.9.44)
yard-coderay (0.1.0)
coderay
yard
+17 -6
View File
@@ -15,6 +15,23 @@ This document is collecting these limitations.
This document does not contain a complete list of limitations.
Please help to improve it by submitting your findings.
## Features provided by mrbgems
Many Ruby features that CRuby builds into its core are provided by
mrbgems in mruby. Which features are actually available depends on
which mrbgems are linked into the build. The `default.gembox` and
`stdlib.gembox` cover the common cases, but a minimal build can omit
familiar features such as `Kernel#binding` (provided by
`mruby-binding`), `Kernel#catch`/`throw` (by `mruby-catch`),
`Enumerable` extensions, `Comparable`, IO, regular expressions, and
many more.
This is by design rather than a limitation per se. When porting Ruby
code to mruby, a `NoMethodError` or `NameError` often means "the gem
providing this feature is not linked in" rather than "mruby does not
support it." Adding the relevant gem to the build configuration is
usually enough.
## `Kernel.raise` in rescue clause
`Kernel.raise` without arguments does not raise the current exception within
@@ -133,12 +150,6 @@ The re-defined `+` operator does not accept any arguments.
`'ab'`
Behavior of the operator wasn't changed.
## `Kernel#binding` is not supported without mruby-binding gem
`Kernel#binding` method requires the `mruby-binding` gem (included
in the `metaprog` gembox). Without this gem, `binding` is not
available.
## `nil?` redefinition in conditional expressions
Redefinition of `nil?` is ignored in conditional expressions.
+2
View File
@@ -274,6 +274,8 @@ typedef struct mrb_task_state {
volatile mrb_bool switching; /* Context switch pending flag */
struct mrb_task *main_task; /* Main task wrapper for root context */
uint8_t scheduler_lock; /* Lock counter for synchronous execution */
mrb_bool loop_running; /* Active mrb_task_run loop flag */
mrb_bool exception_as_result; /* Return unhandled task exceptions as values */
} mrb_task_state;
#endif
+2 -2
View File
@@ -50,7 +50,7 @@ spi = SPI.new(
\*SPI3 availability depends on ESP32 variant.
### SPI#write(*data)
### `SPI#write(*data)`
Write data to the SPI bus. Data can be Integer, Array, or String.
@@ -68,7 +68,7 @@ data = spi.read(4) # transmits 0x00 while reading
data = spi.read(4, 0xFF) # transmits 0xFF while reading
```
### SPI#transfer(*data, additional_read_bytes: 0)
### `SPI#transfer(*data, additional_read_bytes: 0)`
Full-duplex transfer. Sends data and returns received bytes.
Use `additional_read_bytes:` to append zero-filled read bytes.
+4
View File
@@ -1636,6 +1636,8 @@ ary_combination_next(mrb_state *mrb, mrb_value self)
for (mrb_int i = 0; i < state->k; i++) {
if (state->indices[i] >= state->n) {
state->mode = comb_finished;
mrb_free(mrb, state->indices);
state->indices = NULL;
return mrb_nil_value();
}
}
@@ -1698,6 +1700,8 @@ ary_combination_next(mrb_state *mrb, mrb_value self)
}
state->mode = comb_finished;
mrb_free(mrb, state->indices);
state->indices = NULL;
return result;
}
+18 -5
View File
@@ -373,6 +373,8 @@ trim(mpz_t *x)
while (x->sz && x->p[x->sz-1] == 0) {
x->sz--;
}
/* Maintain invariant: sz == 0 implies sn == 0 (zero is canonical). */
if (x->sz == 0) x->sn = 0;
}
/* z = x + y, without regard for sign */
@@ -3147,8 +3149,11 @@ mpz_mod(mpz_ctx_t *ctx, mpz_t *r, mpz_t *x, mpz_t *y)
return;
}
/* Barrett reduction for moderate-sized moduli (>= 4 limbs where setup is worthwhile) */
if (y->sz >= 4 && y->sz <= 16 && x->sz >= y->sz + 2) {
/* Barrett reduction for moderate-sized moduli (>= 4 limbs where setup is worthwhile).
* Barrett's precondition is x < 2^(2*bits(m)); inputs beyond ~2*m.sz limbs
* violate it and the algorithm silently truncates high limbs. Fall through
* to general division for those. */
if (y->sz >= 4 && y->sz <= 16 && x->sz >= y->sz + 2 && x->sz <= 2 * y->sz) {
mpz_t mu;
mpz_init_temp(ctx, &mu, y->sz + 1);
mpz_barrett_mu(ctx, &mu, y);
@@ -5248,12 +5253,19 @@ mpz_powm_montgomery(mpz_ctx_t *ctx, mpz_t *result,
mpz_init(ctx, &one_mont);
mpz_montgomery_reduce(ctx, &one_mont, &R2, n, rho);
/* Convert base to Montgomery form: base_mont = base * R mod n = REDC(base * R^2) */
mpz_t base_mont, temp;
/* Convert base to Montgomery form: base_mont = base * R mod n = REDC(base * R^2).
* REDC requires its input T to satisfy T < R*N. If `base` is not already
* reduced (e.g. base >= n), `base * R^2` can exceed R*N and REDC produces
* a wrong result. Pre-reduce base modulo n via mpz_mmod (the general
* division path) -- both operands are non-negative here so this is
* semantically equivalent to mpz_mod. */
mpz_t base_mont, base_reduced, temp;
mpz_init(ctx, &base_mont);
mpz_init(ctx, &base_reduced);
mpz_init_temp(ctx, &temp, n->sz * 4);
mpz_mul(ctx, &temp, (mpz_t*)base, &R2);
mpz_mmod(ctx, &base_reduced, (mpz_t*)base, (mpz_t*)n);
mpz_mul(ctx, &temp, &base_reduced, &R2);
mpz_montgomery_reduce(ctx, &base_mont, &temp, n, rho);
/* Initialize accumulator to 1 in Montgomery form */
@@ -5285,6 +5297,7 @@ mpz_powm_montgomery(mpz_ctx_t *ctx, mpz_t *result,
mpz_clear(ctx, &R2);
mpz_clear(ctx, &one_mont);
mpz_clear(ctx, &base_mont);
mpz_clear(ctx, &base_reduced);
mpz_clear(ctx, &temp);
mpz_clear(ctx, &acc);
pool_restore(ctx, pool_state);
+26
View File
@@ -150,6 +150,32 @@ assert 'Bigint pow' do
# assert_equal(-1041439304, n.pow(n, -1234567890))
end
assert 'Bigint Integer#pow(e, m) - Montgomery path' do
# Regression: mpz_powm_montgomery() failed to pre-reduce base mod n,
# producing wrong results when base >= n. Also trim() must restore
# the canonical sn=0 when sz becomes 0, otherwise an inconsistent
# zero bignum (sn!=0, sz=0) propagates through the squaring loop.
m = (2**40) + 1
assert_equal 1, (2**160).pow(2, m)
assert_equal 1, (2**320).pow(2, m)
assert_equal 8, ((2**160) + 1).pow(3, m)
m2 = (2**100) + 3
assert_equal (3**500) % m2, (3**500).pow(1, m2)
assert_equal ((5**300) ** 7) % m2, (5**300).pow(7, m2)
end
assert 'Bigint Integer#remainder large operand' do
# Regression: mpz_mod's Barrett path didn't enforce its precondition
# x < 2^(2*bits(m)), so it silently truncated high limbs when x was
# much larger than m^2, producing the wrong remainder. Integer#%
# took the udiv path and worked, but Integer#remainder went through
# mpz_mod and was broken.
m = (2**100) + 3
assert_equal (3**500) % m, (3**500).remainder(m)
assert_equal (5**500) % ((2**150) + 1), (5**500).remainder((2**150) + 1)
assert_equal (2**400) % ((2**130) + 1), (2**400).remainder((2**130) + 1)
end
assert 'Bigint abs' do
n = 1<<65
assert_equal 36893488147419103232, n.abs
@@ -239,6 +239,7 @@ mrb_debug_set_break_method(mrb_state *mrb, mrb_debug_context *dbg, const char *c
set_method = mrdb_strdup(mrb, method_name);
if (set_method == NULL) {
mrb_free(mrb, set_class);
return MRB_DEBUG_NOBUF;
}
index = alloc_breakpoint(dbg, MRB_DEBUG_BPTYPE_METHOD);
+29 -19
View File
@@ -3974,7 +3974,13 @@ codegen_if(codegen_scope *s, node *varnode, int val)
mrb_sym sym_nil_p = MRB_SYM_Q(nil);
if (call_n->method_name == sym_nil_p && callargs_empty(call_n->args)) {
nil_p = TRUE;
codegen(s, call_n->receiver, VAL);
if (call_n->receiver) {
codegen(s, call_n->receiver, VAL);
}
else {
/* implicit receiver: bare `nil?` means `self.nil?` */
gen_load_op1(s, OP_LOADSELF, VAL);
}
}
}
@@ -4285,6 +4291,22 @@ codegen_case(codegen_scope *s, node *varnode, int val)
*/
static void codegen_pattern(codegen_scope *s, node *pattern, int target, uint32_t *fail_pos, int known_array_len);
/* Return the static element count of an array literal AST node, or -1 if any
* element is a splat (whose runtime length is unknown).
*/
static int
array_literal_known_len(node *value)
{
if (node_type(value) != NODE_ARRAY) return -1;
struct mrb_ast_array_node *arr = array_node(value);
int len = 0;
for (node *elem = arr->elements; elem; elem = elem->cdr) {
if (is_splat_node(elem->car)) return -1;
len++;
}
return len;
}
/* Pattern matching case/in expression */
static void
codegen_case_match(codegen_scope *s, node *varnode, int val)
@@ -4298,13 +4320,7 @@ codegen_case_match(codegen_scope *s, node *varnode, int val)
uint32_t tmp;
/* Check if value is an array literal - allows optimizations in pattern matching */
int known_array_len = -1;
if (node_type(value) == NODE_ARRAY) {
struct mrb_ast_array_node *arr = array_node(value);
node *elem;
known_array_len = 0;
for (elem = arr->elements; elem; elem = elem->cdr) known_array_len++;
}
int known_array_len = array_literal_known_len(value);
/* Generate code for the case value */
codegen(s, value, VAL);
@@ -6552,16 +6568,15 @@ codegen(codegen_scope *s, node *tree, int val)
/* Optimize: array literal => array pattern with matching sizes */
if (node_type(mp->value) == NODE_ARRAY &&
node_type(mp->pattern) == NODE_PAT_ARRAY) {
struct mrb_ast_array_node *arr = array_node(mp->value);
struct mrb_ast_pat_array_node *pat = pat_array_node(mp->pattern);
/* Only optimize for exact match (no rest, no post) */
if (pat->rest == 0 && pat->post == NULL) {
/* Count array elements and pattern pre elements */
int arr_len = 0, pat_len = 0;
/* Count array elements (bail if splat present) and pattern pre elements */
int arr_len = array_literal_known_len(mp->value);
int pat_len = 0;
node *e;
for (e = arr->elements; e; e = e->cdr) arr_len++;
for (e = pat->pre; e; e = e->cdr) pat_len++;
if (arr_len == pat_len) {
if (arr_len >= 0 && arr_len == pat_len) {
/* Sizes match - skip deconstruct and size check */
int arr_reg = cursp();
int i = 0;
@@ -6600,12 +6615,7 @@ codegen(codegen_scope *s, node *tree, int val)
head = cursp();
/* Check if value is array literal for optimization */
if (node_type(mp->value) == NODE_ARRAY) {
struct mrb_ast_array_node *arr = array_node(mp->value);
node *elem;
known_array_len = 0;
for (elem = arr->elements; elem; elem = elem->cdr) known_array_len++;
}
known_array_len = array_literal_known_len(mp->value);
/* Evaluate the value */
codegen(s, mp->value, VAL);
+16 -17
View File
@@ -911,18 +911,12 @@ io_syswrite(mrb_state *mrb, mrb_value io)
/* end */
static mrb_int
fd_write(mrb_state *mrb, int fd, mrb_value str)
fd_write_buf(mrb_state *mrb, int fd, const char *ptr, mrb_int len)
{
fssize_t n;
str = mrb_obj_as_string(mrb, str);
fssize_t len = (fssize_t)RSTRING_LEN(str);
if (len == 0) return 0;
const char *ptr = RSTRING_PTR(str);
fssize_t sum = 0;
while (sum < len) {
n = write(fd, ptr + sum, len - sum);
while (sum < (fssize_t)len) {
fssize_t n = write(fd, ptr + sum, (size_t)(len - sum));
if (n == -1) {
if (errno == EINTR) continue;
mrb_sys_fail(mrb, "syswrite");
@@ -932,6 +926,15 @@ fd_write(mrb_state *mrb, int fd, mrb_value str)
return len;
}
static mrb_int
fd_write(mrb_state *mrb, int fd, mrb_value str)
{
str = mrb_obj_as_string(mrb, str);
return fd_write_buf(mrb, fd, RSTRING_PTR(str), RSTRING_LEN(str));
}
#define FD_WRITE_LIT(mrb, fd, s) fd_write_buf(mrb, fd, "" s "", sizeof(s) - 1)
/* Helper function to prepare IO object for writing by adjusting buffer state */
static void
io_prepare_write(mrb_state *mrb, struct mrb_io *fptr)
@@ -987,8 +990,7 @@ io_puts_str(mrb_state *mrb, int fd, mrb_value str)
/* Add newline if string doesn't end with one */
if (len == 0 || ptr[len-1] != '\n') {
mrb_value newline = mrb_str_new_lit(mrb, "\n");
fd_write(mrb, fd, newline);
FD_WRITE_LIT(mrb, fd, "\n");
}
}
@@ -1001,8 +1003,7 @@ static void
io_puts_ary(mrb_state *mrb, int fd, mrb_value ary, int depth)
{
if (depth >= IO_PUTS_MAX_DEPTH) {
mrb_value mark = mrb_str_new_lit(mrb, "[...]\n");
fd_write(mrb, fd, mark);
FD_WRITE_LIT(mrb, fd, "[...]\n");
return;
}
@@ -1010,8 +1011,7 @@ io_puts_ary(mrb_state *mrb, int fd, mrb_value ary, int depth)
if (len == 0) {
/* Empty array - write a single newline */
mrb_value newline = mrb_str_new_lit(mrb, "\n");
fd_write(mrb, fd, newline);
FD_WRITE_LIT(mrb, fd, "\n");
return;
}
@@ -1041,8 +1041,7 @@ io_puts(mrb_state *mrb, mrb_value io)
if (argc == 0) {
/* No arguments - just write a newline */
mrb_value newline = mrb_str_new_lit(mrb, "\n");
fd_write(mrb, fd, newline);
FD_WRITE_LIT(mrb, fd, "\n");
return mrb_nil_value();
}
+20 -8
View File
@@ -84,16 +84,20 @@ mrb_value mrb_int_pow(mrb_state *mrb, mrb_value x, mrb_value y);
static mrb_int
mrb_int_gcd(mrb_int x, mrb_int y)
{
if (x < 0) x = -x;
if (y < 0) y = -y;
/* Negate via unsigned so MRB_INT_MIN doesn't overflow.
The cast back at the end produces MRB_INT_MIN only when the
true result is 2^63 (gcd of MRB_INT_MIN with itself or 0);
callers detect that case from the negative return value. */
mrb_uint ux = (x < 0) ? -(mrb_uint)x : (mrb_uint)x;
mrb_uint uy = (y < 0) ? -(mrb_uint)y : (mrb_uint)y;
while (y != 0) {
mrb_int temp = y;
y = x % y;
x = temp;
while (uy != 0) {
mrb_uint temp = uy;
uy = ux % uy;
ux = temp;
}
return x;
return (mrb_int)ux;
}
/*
@@ -122,7 +126,11 @@ int_gcd(mrb_state *mrb, mrb_value x)
if (!mrb_integer_p(y)) {
mrb_raisef(mrb, E_TYPE_ERROR, "can't convert %Y into Integer", y);
}
return mrb_int_value(mrb, mrb_int_gcd(mrb_integer(x), mrb_integer(y)));
mrb_int g = mrb_int_gcd(mrb_integer(x), mrb_integer(y));
/* g < 0 only when the mathematical result is 2^63 (= |MRB_INT_MIN|),
which does not fit in mrb_int. */
if (g < 0) mrb_int_overflow(mrb, "gcd");
return mrb_int_value(mrb, g);
}
/*
@@ -158,6 +166,10 @@ int_lcm(mrb_state *mrb, mrb_value x)
if (a == 0 || b == 0) return mrb_int_value(mrb, 0);
/* Negation of MRB_INT_MIN is UB and the lcm with any non-zero
operand would not fit in mrb_int anyway. */
if (a == MRB_INT_MIN || b == MRB_INT_MIN) mrb_int_overflow(mrb, "lcm");
gcd_val = mrb_int_gcd(a, b);
if (a < 0) a = -a;
if (b < 0) b = -b;
+13 -7
View File
@@ -76,7 +76,7 @@ typedef struct mrb_regexp_pattern {
uint8_t first_bytes[16]; /* bitmap of possible first bytes (128-bit, ASCII) */
mrb_bool has_first_bytes; /* true if first_bytes is usable for skipping */
mrb_bool is_literal; /* true if pattern is pure literal (no metacharacters) */
/* Cached VM state for pike_vm (avoids malloc per re_exec call) */
/* Cached VM state for pike_vm (avoids malloc per mrb_re_exec call) */
uint32_t *cached_visited; /* generation-based visited array */
void *cached_threads[2]; /* curr/next thread lists */
int cached_list_capa; /* capacity of cached thread lists */
@@ -97,6 +97,12 @@ typedef struct mrb_regexp_pattern {
#define MRB_REGEXP_STEP_LIMIT 1000000
#endif
/* Recursion-depth limit for bt_match: bounds C stack growth on
patterns like `(?=)+` that recurse without consuming input. */
#ifndef MRB_REGEXP_RECURSION_LIMIT
#define MRB_REGEXP_RECURSION_LIMIT 1000
#endif
/* Maximum captures */
#define RE_MAX_CAPTURES 32
@@ -107,21 +113,21 @@ typedef struct {
} re_thread_cache;
/* Compile a pattern string into bytecode */
mrb_regexp_pattern* re_compile(mrb_state *mrb, const char *pattern, mrb_int len, uint32_t flags);
mrb_regexp_pattern* mrb_re_compile(mrb_state *mrb, const char *pattern, mrb_int len, uint32_t flags);
/* Free a compiled pattern */
void re_free(mrb_state *mrb, mrb_regexp_pattern *pat);
void mrb_re_free(mrb_state *mrb, mrb_regexp_pattern *pat);
/* Execute a match.
Returns number of captures filled (0 = no match).
captures[2*n] = start, captures[2*n+1] = end for group n. */
int re_exec(mrb_state *mrb, const mrb_regexp_pattern *pat,
int mrb_re_exec(mrb_state *mrb, const mrb_regexp_pattern *pat,
const char *str, mrb_int len, mrb_int start,
int *captures, int captures_size);
/* UTF-8 helpers */
int re_utf8_charlen(const char *s, const char *end);
uint32_t re_utf8_decode(const char *s, int *len);
mrb_bool re_is_word_char(uint32_t c);
int mrb_re_utf8_charlen(const char *s, const char *end);
uint32_t mrb_re_utf8_decode(const char *s, int *len);
mrb_bool mrb_re_is_word_char(uint32_t c);
#endif /* MRB_RE_INTERNAL_H */
+19 -7
View File
@@ -91,12 +91,16 @@ insert_inst(re_compiler *c, uint32_t pos, uint8_t op, uint8_t a, uint16_t offset
c->code[pos].a = a;
c->code[pos].offset = offset;
/* fix all jump targets that point at or past the insertion point */
/* Fix jump targets that point past the insertion point. An offset equal
to `pos` already points to the inserted instruction's new location and
must not be bumped -- bumping it would shift the target to whatever
code got displaced by the insertion (e.g. the body of the quantified
atom), corrupting "skip past this atom" jumps emitted earlier. */
for (uint32_t i = 0; i < c->code_len; i++) {
if (i == pos) continue;
switch (c->code[i].op) {
case RE_JMP: case RE_SPLIT: case RE_SPLITNG:
if (c->code[i].offset >= pos && c->code[i].offset < 0xffff) {
if (c->code[i].offset > pos && c->code[i].offset < 0xffff) {
c->code[i].offset++;
}
break;
@@ -179,7 +183,7 @@ class_add_shorthand(re_charclass *cc, int ch)
break;
case 'W':
for (int i = 0; i < 128; i++) {
if (!re_is_word_char(i)) class_set_bit(cc, (uint8_t)i);
if (!mrb_re_is_word_char(i)) class_set_bit(cc, (uint8_t)i);
}
cc->utf8_any = TRUE;
break;
@@ -214,6 +218,8 @@ parse_escape(re_compiler *c)
case 'v': return '\v';
case 'a': return '\a';
case 'e': return 0x1b;
case 'b': return '\b'; /* backspace; only reachable inside [...] since the
top-level dispatcher emits RE_WBOUND for `\b` */
default: return ch; /* literal: \., \\, \/, \(, etc. */
}
}
@@ -817,12 +823,18 @@ first_set_walk(const re_inst *code, uint32_t code_len,
case RE_ANY: case RE_ANY_NL:
return FALSE; /* any byte possible */
case RE_MATCH:
return TRUE; /* empty match; first_bytes still valid for other branches */
/* Reaching MATCH via epsilon transitions means the regex can match
zero characters at any position. Skipping bytes that aren't in the
first-byte set would skip past valid empty-match positions, so the
optimization isn't safe -- bail out and accept any starting byte. */
return FALSE;
default:
return FALSE;
}
}
return TRUE;
/* Walked off the end without hitting MATCH or a consuming op. Treat as
empty-matchable, same as RE_MATCH. */
return FALSE;
}
static mrb_bool
@@ -845,7 +857,7 @@ compute_first_set(const re_inst *code, uint32_t code_len,
}
mrb_regexp_pattern*
re_compile(mrb_state *mrb, const char *pattern, mrb_int len, uint32_t flags)
mrb_re_compile(mrb_state *mrb, const char *pattern, mrb_int len, uint32_t flags)
{
re_compiler c;
memset(&c, 0, sizeof(c));
@@ -972,7 +984,7 @@ re_compile(mrb_state *mrb, const char *pattern, mrb_int len, uint32_t flags)
}
void
re_free(mrb_state *mrb, mrb_regexp_pattern *pat)
mrb_re_free(mrb_state *mrb, mrb_regexp_pattern *pat)
{
if (pat) {
mrb_free(mrb, pat->code);
+26 -23
View File
@@ -170,16 +170,16 @@ add_thread(pike_state *s, re_threadlist *list,
case RE_WBOUND:
{
mrb_bool before = (sp > s->str) && re_is_word_char((uint8_t)sp[-1]);
mrb_bool after = (sp < s->str_end) && re_is_word_char((uint8_t)*sp);
mrb_bool before = (sp > s->str) && mrb_re_is_word_char((uint8_t)sp[-1]);
mrb_bool after = (sp < s->str_end) && mrb_re_is_word_char((uint8_t)*sp);
if (before != after) { pc++; continue; }
}
return;
case RE_NWBOUND:
{
mrb_bool before = (sp > s->str) && re_is_word_char((uint8_t)sp[-1]);
mrb_bool after = (sp < s->str_end) && re_is_word_char((uint8_t)*sp);
mrb_bool before = (sp > s->str) && mrb_re_is_word_char((uint8_t)sp[-1]);
mrb_bool after = (sp < s->str_end) && mrb_re_is_word_char((uint8_t)*sp);
if (before == after) { pc++; continue; }
}
return;
@@ -301,7 +301,7 @@ pike_vm(mrb_state *mrb, const mrb_regexp_pattern *pat,
next.count = 0;
int ch = (uint8_t)*sp;
int advance = re_utf8_charlen(sp, str_end);
int advance = mrb_re_utf8_charlen(sp, str_end);
for (int i = 0; i < curr.count; i++) {
re_thread *th = &curr.threads[i];
@@ -388,8 +388,10 @@ pike_vm(mrb_state *mrb, const mrb_regexp_pattern *pat,
*/
static mrb_bool
bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
const char *sp, uint32_t pc, int *captures, int ncap, int *steps)
const char *sp, uint32_t pc, int *captures, int ncap, int *steps,
int depth)
{
if (depth > MRB_REGEXP_RECURSION_LIMIT) return FALSE;
while (pc < pat->code_len) {
if (++(*steps) > MRB_REGEXP_STEP_LIMIT) return FALSE;
@@ -402,22 +404,22 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
case RE_ANY:
if (sp >= str_end || *sp == '\n') return FALSE;
sp += re_utf8_charlen(sp, str_end); pc++;
sp += mrb_re_utf8_charlen(sp, str_end); pc++;
break;
case RE_ANY_NL:
if (sp >= str_end) return FALSE;
sp += re_utf8_charlen(sp, str_end); pc++;
sp += mrb_re_utf8_charlen(sp, str_end); pc++;
break;
case RE_CLASS:
if (sp >= str_end || !class_match(&pat->classes[inst.a], (uint8_t)*sp)) return FALSE;
sp += re_utf8_charlen(sp, str_end); pc++;
sp += mrb_re_utf8_charlen(sp, str_end); pc++;
break;
case RE_NCLASS:
if (sp >= str_end || class_match(&pat->classes[inst.a], (uint8_t)*sp)) return FALSE;
sp += re_utf8_charlen(sp, str_end); pc++;
sp += mrb_re_utf8_charlen(sp, str_end); pc++;
break;
case RE_MATCH:
@@ -428,12 +430,12 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
break;
case RE_SPLIT:
if (bt_match(pat, str, str_end, sp, pc + 1, captures, ncap, steps)) return TRUE;
if (bt_match(pat, str, str_end, sp, pc + 1, captures, ncap, steps, depth + 1)) return TRUE;
pc = inst.offset;
break;
case RE_SPLITNG:
if (bt_match(pat, str, str_end, sp, inst.offset, captures, ncap, steps)) return TRUE;
if (bt_match(pat, str, str_end, sp, inst.offset, captures, ncap, steps, depth + 1)) return TRUE;
pc++;
break;
@@ -443,7 +445,7 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
if (slot < ncap) {
int old = captures[slot];
captures[slot] = (int)(sp - str);
if (bt_match(pat, str, str_end, sp, pc + 1, captures, ncap, steps)) return TRUE;
if (bt_match(pat, str, str_end, sp, pc + 1, captures, ncap, steps, depth + 1)) return TRUE;
captures[slot] = old;
}
return FALSE;
@@ -471,8 +473,8 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
case RE_WBOUND:
{
mrb_bool before = (sp > str) && re_is_word_char((uint8_t)sp[-1]);
mrb_bool after = (sp < str_end) && re_is_word_char((uint8_t)*sp);
mrb_bool before = (sp > str) && mrb_re_is_word_char((uint8_t)sp[-1]);
mrb_bool after = (sp < str_end) && mrb_re_is_word_char((uint8_t)*sp);
if (before == after) return FALSE;
}
pc++;
@@ -480,8 +482,8 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
case RE_NWBOUND:
{
mrb_bool before = (sp > str) && re_is_word_char((uint8_t)sp[-1]);
mrb_bool after = (sp < str_end) && re_is_word_char((uint8_t)*sp);
mrb_bool before = (sp > str) && mrb_re_is_word_char((uint8_t)sp[-1]);
mrb_bool after = (sp < str_end) && mrb_re_is_word_char((uint8_t)*sp);
if (before != after) return FALSE;
}
pc++;
@@ -490,6 +492,7 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
case RE_BACKREF:
{
int group = inst.a;
if (group * 2 + 1 >= ncap) return FALSE;
int gs = captures[group * 2];
int ge = captures[group * 2 + 1];
if (gs < 0 || ge < 0) return FALSE;
@@ -502,13 +505,13 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
break;
case RE_LOOKAHEAD:
if (!bt_match(pat, str, str_end, sp, pc + 1, captures, ncap, steps))
if (!bt_match(pat, str, str_end, sp, pc + 1, captures, ncap, steps, depth + 1))
return FALSE;
pc = inst.offset;
break;
case RE_NEG_LOOKAHEAD:
if (bt_match(pat, str, str_end, sp, pc + 1, captures, ncap, steps))
if (bt_match(pat, str, str_end, sp, pc + 1, captures, ncap, steps, depth + 1))
return FALSE;
pc = inst.offset;
break;
@@ -517,7 +520,7 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
{
int lb_len = inst.a;
if (sp - str < lb_len) return FALSE; /* not enough text before */
if (!bt_match(pat, str, str_end, sp - lb_len, pc + 1, captures, ncap, steps))
if (!bt_match(pat, str, str_end, sp - lb_len, pc + 1, captures, ncap, steps, depth + 1))
return FALSE;
pc = inst.offset;
}
@@ -527,7 +530,7 @@ bt_match(const mrb_regexp_pattern *pat, const char *str, const char *str_end,
{
int lb_len = inst.a;
if (sp - str >= lb_len) {
if (bt_match(pat, str, str_end, sp - lb_len, pc + 1, captures, ncap, steps))
if (bt_match(pat, str, str_end, sp - lb_len, pc + 1, captures, ncap, steps, depth + 1))
return FALSE;
}
/* if not enough text before, negative lookbehind succeeds */
@@ -567,7 +570,7 @@ backtrack_exec(mrb_state *mrb, const mrb_regexp_pattern *pat,
memset(caps, -1, sizeof(int) * ncap);
int steps = 0;
if (bt_match(pat, str, str_end, sp, 0, caps, ncap, &steps)) {
if (bt_match(pat, str, str_end, sp, 0, caps, ncap, &steps, 0)) {
if (captures) {
int copy = ncap < captures_size ? ncap : captures_size;
memcpy(captures, caps, sizeof(int) * copy);
@@ -608,7 +611,7 @@ literal_exec(const mrb_regexp_pattern *pat,
/* Public entry point */
int
re_exec(mrb_state *mrb, const mrb_regexp_pattern *pat,
mrb_re_exec(mrb_state *mrb, const mrb_regexp_pattern *pat,
const char *str, mrb_int len, mrb_int start,
int *captures, int captures_size)
{
+3 -3
View File
@@ -9,7 +9,7 @@
/* Return byte length of UTF-8 character at s.
Returns 1 for invalid sequences (treat as single byte). */
int
re_utf8_charlen(const char *s, const char *end)
mrb_re_utf8_charlen(const char *s, const char *end)
{
uint8_t c = (uint8_t)*s;
int len;
@@ -28,7 +28,7 @@ re_utf8_charlen(const char *s, const char *end)
/* Decode a UTF-8 character and return its codepoint.
*len is set to the byte length consumed. */
uint32_t
re_utf8_decode(const char *s, int *len)
mrb_re_utf8_decode(const char *s, int *len)
{
uint8_t c = (uint8_t)s[0];
uint32_t cp;
@@ -66,7 +66,7 @@ re_utf8_decode(const char *s, int *len)
/* Check if character is a "word" character (\w): [a-zA-Z0-9_] */
mrb_bool
re_is_word_char(uint32_t c)
mrb_re_is_word_char(uint32_t c)
{
if (c >= 'a' && c <= 'z') return TRUE;
if (c >= 'A' && c <= 'Z') return TRUE;
+18 -13
View File
@@ -19,7 +19,7 @@
/* Regexp data type */
static void regexp_free(mrb_state *mrb, void *ptr) {
re_free(mrb, (mrb_regexp_pattern*)ptr);
mrb_re_free(mrb, (mrb_regexp_pattern*)ptr);
}
static const struct mrb_data_type regexp_type = { "Regexp", regexp_free };
@@ -107,15 +107,17 @@ regexp_init(mrb_state *mrb, mrb_value self)
flags = parse_flags(mrb, flags_val);
}
pat = re_compile(mrb, RSTRING_PTR(pattern), RSTRING_LEN(pattern), flags);
/* Set @source and @flags before mrb_re_compile() so a Regexp that survives
a compile-time exception (e.g. picked up by ObjectSpace.each_object)
still has usable IVs for hash/eql?/inspect. */
mrb_iv_set(mrb, self, mrb_intern_lit(mrb, "@source"), pattern);
mrb_iv_set(mrb, self, mrb_intern_lit(mrb, "@flags"), mrb_int_value(mrb, (mrb_int)flags));
pat = mrb_re_compile(mrb, RSTRING_PTR(pattern), RSTRING_LEN(pattern), flags);
DATA_TYPE(self) = &regexp_type;
DATA_PTR(self) = pat;
/* store source for #source and #inspect */
mrb_iv_set(mrb, self, mrb_intern_lit(mrb, "@source"), pattern);
mrb_iv_set(mrb, self, mrb_intern_lit(mrb, "@flags"), mrb_int_value(mrb, (mrb_int)flags));
/* store named captures as hash */
if (pat->num_named > 0) {
mrb_value nc = mrb_hash_new_capa(mrb, pat->num_named);
@@ -204,7 +206,7 @@ exec_match(mrb_state *mrb, mrb_value self, mrb_value str, mrb_int pos)
int cap_size = pat->num_captures * 2;
int *captures = (int*)mrb_malloc(mrb, sizeof(int) * cap_size);
memset(captures, -1, sizeof(int) * cap_size);
int ncap = re_exec(mrb, pat, RSTRING_PTR(str), RSTRING_LEN(str), pos,
int ncap = mrb_re_exec(mrb, pat, RSTRING_PTR(str), RSTRING_LEN(str), pos,
captures, cap_size);
if (ncap == 0) {
@@ -242,7 +244,7 @@ regexp_match_p(mrb_state *mrb, mrb_value self)
mrb_regexp_pattern *pat = DATA_GET_PTR(mrb, self, &regexp_type, mrb_regexp_pattern);
if (!pat) mrb_raise(mrb, E_ARGUMENT_ERROR, "uninitialized Regexp");
int ncap = re_exec(mrb, pat, RSTRING_PTR(str), RSTRING_LEN(str), pos, NULL, 0);
int ncap = mrb_re_exec(mrb, pat, RSTRING_PTR(str), RSTRING_LEN(str), pos, NULL, 0);
return mrb_bool_value(ncap > 0);
}
@@ -279,7 +281,7 @@ regexp_case_match(mrb_state *mrb, mrb_value self)
pat = DATA_GET_PTR(mrb, self, &regexp_type, mrb_regexp_pattern);
if (!pat) return mrb_false_value();
int ncap = re_exec(mrb, pat, RSTRING_PTR(str), RSTRING_LEN(str), 0, NULL, 0);
int ncap = mrb_re_exec(mrb, pat, RSTRING_PTR(str), RSTRING_LEN(str), 0, NULL, 0);
return mrb_bool_value(ncap > 0);
}
@@ -364,6 +366,9 @@ regexp_eql(mrb_state *mrb, mrb_value self)
}
mrb_value src1 = mrb_iv_get(mrb, self, mrb_intern_lit(mrb, "@source"));
mrb_value src2 = mrb_iv_get(mrb, other, mrb_intern_lit(mrb, "@source"));
if (!mrb_string_p(src1) || !mrb_string_p(src2)) {
return mrb_bool_value(mrb_obj_eq(mrb, self, other));
}
if (!mrb_str_equal(mrb, src1, src2)) return mrb_false_value();
return mrb_bool_value(get_iflags(mrb, self) == get_iflags(mrb, other));
}
@@ -375,7 +380,7 @@ static mrb_value
regexp_hash(mrb_state *mrb, mrb_value self)
{
mrb_value src = mrb_iv_get(mrb, self, mrb_intern_lit(mrb, "@source"));
uint32_t h = mrb_str_hash(mrb, src);
uint32_t h = mrb_string_p(src) ? mrb_str_hash(mrb, src) : 0;
h ^= get_iflags(mrb, self) * 0x9e3779b9; /* mix flags into hash */
return mrb_int_value(mrb, (mrb_int)h);
}
@@ -721,7 +726,7 @@ regexp_gsub_str(mrb_state *mrb, mrb_value self)
while (pos <= slen) {
memset(captures, -1, sizeof(int) * cap_size);
int n = re_exec(mrb, pat, s, slen, pos, captures, cap_size);
int n = mrb_re_exec(mrb, pat, s, slen, pos, captures, cap_size);
if (n == 0) break;
/* save last match for $~ */
@@ -795,7 +800,7 @@ regexp_sub_str(mrb_state *mrb, mrb_value self)
int *captures = (int*)mrb_malloc(mrb, sizeof(int) * cap_size);
memset(captures, -1, sizeof(int) * cap_size);
int n = re_exec(mrb, pat, s, slen, 0, captures, cap_size);
int n = mrb_re_exec(mrb, pat, s, slen, 0, captures, cap_size);
if (n == 0) {
mrb_free(mrb, captures);
clear_match_globals(mrb);
@@ -853,7 +858,7 @@ regexp_scan(mrb_state *mrb, mrb_value self)
while (pos <= slen) {
memset(captures, -1, sizeof(int) * cap_size);
int n = re_exec(mrb, pat, s, slen, pos, captures, cap_size);
int n = mrb_re_exec(mrb, pat, s, slen, pos, captures, cap_size);
if (n == 0) break;
last_ncap = cap_size;
+49 -1
View File
@@ -47,6 +47,14 @@ assert("Regexp - character class") do
assert_equal "abc", md[0]
end
assert("Regexp - \\b inside character class is backspace") do
# Outside [...], \b is the word boundary assertion; inside [...]
# it must mean U+0008 (backspace), matching MRI/Onigmo.
assert_equal "Ruby", "Ruby".gsub(/[\b]/, "X")
assert_equal "aXc", "a\bc".gsub(/[\b]/, "X")
assert_equal ["\b", "\t", "\n"], "ABC\b\t\n".scan(/[\b-\n]/)
end
assert("Regexp - dot") do
re = Regexp.new("a.c")
assert_true re.match?("abc")
@@ -173,6 +181,17 @@ assert("Regexp#hash") do
assert_not_equal r1.hash, r3.hash
end
assert("Regexp#hash/== on uninitialized regexp") do
# Regexp.allocate yields an object with no @source IV; hash/== must
# not crash (regression: ObjectSpace.each_object could expose a
# half-initialized Regexp after Regexp.new raised a compile error).
r = Regexp.allocate
assert_kind_of Integer, r.hash
assert_true r == r
assert_false r == Regexp.allocate
assert_false r == Regexp.new("abc")
end
assert("Regexp#options") do
assert_equal 0, Regexp.new("abc").options
assert_equal Regexp::IGNORECASE, Regexp.new("abc", Regexp::IGNORECASE).options
@@ -362,7 +381,7 @@ assert("MatchData#named_captures") do
end
assert("Regexp - named captures survive /x preprocessing") do
# Regression: with /x, re_compile freed the stripped buffer that
# Regression: with /x, mrb_re_compile freed the stripped buffer that
# named_captures[i].name pointed into.
re = /(?<n>\d+) # comment
\s* (?<u>\w+) /x
@@ -444,3 +463,32 @@ assert("$1-$9 cleared on no match") do
/xyz/ =~ "abc"
assert_nil $1
end
assert("Regexp - consecutive optional quantifiers (#6853)") do
# insert_inst was over-incrementing jump offsets that pointed *at* the
# insertion site, sending earlier "skip this atom" SPLITs into the next
# atom's body. Two adjacent zero-matchable atoms then both failed even
# when both should match zero characters.
assert_equal ["a", nil], /\Aa(b)?c?\z/.match("a").to_a
assert_equal ["ab", "b"], /\Aa(b)?c?\z/.match("ab").to_a
assert_equal ["ac", nil], /\Aa(b)?c?\z/.match("ac").to_a
assert_equal ["abc", "b"], /\Aa(b)?c?\z/.match("abc").to_a
assert_equal [""], /a?b?/.match("").to_a
assert_equal [""], /a*b*/.match("").to_a
assert_equal [""], /a?b?c?d?/.match("").to_a
end
assert("Regexp - empty-matchable patterns find earliest match position") do
# When a regex can match zero characters via epsilon transitions, the
# first-byte skip-ahead optimization is unsafe: skipping past bytes
# that aren't in the first-byte set would also skip past valid
# empty-match positions.
md = /a?/.match("b")
assert_equal "", md[0]
assert_equal 0, md.begin(0)
md = /a?b?/.match("c")
assert_equal "", md[0]
assert_equal 0, md.begin(0)
end
+30
View File
@@ -163,4 +163,34 @@ class String
end
self
end
##
# call-seq:
# str.scrub -> new_str
# str.scrub(repl) -> new_str
# str.scrub {|bytes| block } -> new_str
#
# Returns a copy of +self+ with each maximal run of invalid UTF-8 bytes
# replaced by +repl+ (U+FFFD if +repl+ is omitted), or by the value
# returned from the block when one is given. The block receives the
# invalid bytes as a String.
#
# "abc\x80def".scrub #=> "abc\u{FFFD}def"
# "abc\x80def".scrub("?") #=> "abc?def"
# "\xE3\x81".scrub #=> "\u{FFFD}"
# "\x80\x81".scrub { |b| b.bytes.map { |c| "<%02X>" % c }.join }
# #=> "<80><81>"
def scrub(repl = nil, &block)
return __scrub(repl) unless block
chunks = __scrub_chunks
return chunks[0] if chunks.length == 1
result = chunks[0].dup
i = 1
while i < chunks.length
result << yield(chunks[i]).to_s
result << chunks[i + 1] if i + 1 < chunks.length
i += 2
end
result
end
end
+157
View File
@@ -1020,6 +1020,143 @@ str_ord(mrb_state* mrb, mrb_value str)
return mrb_fixnum_value(c);
}
/* Returns the byte length of a valid UTF-8 char starting at p, or -1 for
any invalid sequence (illegal lead byte, truncated tail, invalid
continuation byte, overlong encoding, UTF-16 surrogate, or codepoint
above U+10FFFF). Like utf8code() but reports rather than raises. */
static mrb_int
str_scrub_char_len(const unsigned char *p, const unsigned char *e)
{
if (p[0] < 0x80) return 1;
mrb_int len = mrb_utf8len_table[p[0]>>3];
if (len < 2 || len > e - p) return -1;
for (mrb_int i = 1; i < len; i++) {
if ((p[i] & 0xc0) != 0x80) return -1;
}
mrb_int cp;
if (len == 2) {
cp = ((p[0] & 0x1f) << 6) | (p[1] & 0x3f);
if (cp < 0x80) return -1;
}
else if (len == 3) {
cp = ((p[0] & 0x0f) << 12) | ((p[1] & 0x3f) << 6) | (p[2] & 0x3f);
if (cp < 0x800) return -1;
if (cp >= 0xD800 && cp <= 0xDFFF) return -1;
}
else { /* len == 4 */
cp = ((p[0] & 0x07) << 18) | ((p[1] & 0x3f) << 12)
| ((p[2] & 0x3f) << 6) | (p[3] & 0x3f);
if (cp < 0x10000 || cp > 0x10FFFF) return -1;
}
return len;
}
static void
str_scrub_validate_replacement(mrb_state *mrb, mrb_value repl)
{
const unsigned char *p = (const unsigned char*)RSTRING_PTR(repl);
const unsigned char *e = p + RSTRING_LEN(repl);
while (p < e) {
mrb_int len = str_scrub_char_len(p, e);
if (len < 0) {
mrb_raise(mrb, E_ARGUMENT_ERROR, "replacement must be valid UTF-8");
}
p += len;
}
}
/* Core of String#scrub for the no-block case. Returns a new string with
each maximal run of invalid UTF-8 bytes replaced by `repl` (or U+FFFD
if `repl` is nil). Already-valid strings are returned via mrb_str_dup. */
static mrb_value
str_scrub_core(mrb_state *mrb, mrb_value self)
{
mrb_value repl = mrb_nil_value();
mrb_get_args(mrb, "|S!", &repl);
const char *replace;
mrb_int replace_len;
if (mrb_nil_p(repl)) {
replace = "\xEF\xBF\xBD"; /* U+FFFD REPLACEMENT CHARACTER */
replace_len = 3;
}
else {
str_scrub_validate_replacement(mrb, repl);
replace = RSTRING_PTR(repl);
replace_len = RSTRING_LEN(repl);
}
struct RString *s = mrb_str_ptr(self);
if (RSTR_SINGLE_BYTE_P(s) || RSTR_BINARY_P(s)) {
return mrb_str_dup(mrb, self);
}
const unsigned char *p = (const unsigned char*)RSTR_PTR(s);
const unsigned char *e = p + RSTR_LEN(s);
const unsigned char *valid_start = p;
const unsigned char *q = p;
mrb_value result = mrb_nil_value(); /* lazily allocated on first invalid byte */
while (q < e) {
mrb_int len = str_scrub_char_len(q, e);
if (len < 0) {
if (mrb_nil_p(result)) {
result = mrb_str_new(mrb, NULL, 0);
}
mrb_str_cat(mrb, result, (const char*)valid_start, q - valid_start);
mrb_str_cat(mrb, result, replace, replace_len);
q++;
while (q < e && str_scrub_char_len(q, e) < 0) q++;
valid_start = q;
}
else {
q += len;
}
}
if (mrb_nil_p(result)) {
return mrb_str_dup(mrb, self); /* already valid */
}
mrb_str_cat(mrb, result, (const char*)valid_start, q - valid_start);
return result;
}
/* Splits self into alternating valid/invalid byte runs and returns them
as an Array of strings ([valid, invalid, valid, ...], odd length).
Used by the block form of String#scrub in mrblib; the block can then
map each invalid run to a replacement of its choosing without the C
side having to call back into the VM. */
static mrb_value
str_scrub_chunks(mrb_state *mrb, mrb_value self)
{
mrb_value ary = mrb_ary_new(mrb);
struct RString *s = mrb_str_ptr(self);
if (RSTR_SINGLE_BYTE_P(s) || RSTR_BINARY_P(s)) {
mrb_ary_push(mrb, ary, mrb_str_dup(mrb, self));
return ary;
}
const unsigned char *p = (const unsigned char*)RSTR_PTR(s);
const unsigned char *e = p + RSTR_LEN(s);
const unsigned char *valid_start = p;
const unsigned char *q = p;
while (q < e) {
mrb_int len = str_scrub_char_len(q, e);
if (len < 0) {
mrb_ary_push(mrb, ary, mrb_str_new(mrb, (const char*)valid_start, q - valid_start));
const unsigned char *invalid_start = q;
q++;
while (q < e && str_scrub_char_len(q, e) < 0) q++;
mrb_ary_push(mrb, ary, mrb_str_new(mrb, (const char*)invalid_start, q - invalid_start));
valid_start = q;
}
else {
q += len;
}
}
mrb_ary_push(mrb, ary, mrb_str_new(mrb, (const char*)valid_start, q - valid_start));
return ary;
}
/* Internal helper for String#codepoints - returns array of character codepoints */
static mrb_value
str_codepoints(mrb_state *mrb, mrb_value str)
@@ -1068,6 +1205,24 @@ str_codepoints(mrb_state *mrb, mrb_value self)
}
return result;
}
/* Non-UTF-8 builds: scrub is a no-op. The replacement arg is accepted
(and validated only as a String) for API parity with the UTF-8 build. */
static mrb_value
str_scrub_core(mrb_state *mrb, mrb_value self)
{
mrb_value repl = mrb_nil_value();
mrb_get_args(mrb, "|S!", &repl);
return mrb_str_dup(mrb, self);
}
static mrb_value
str_scrub_chunks(mrb_state *mrb, mrb_value self)
{
mrb_value ary = mrb_ary_new(mrb);
mrb_ary_push(mrb, ary, mrb_str_dup(mrb, self));
return ary;
}
#endif
static mrb_bool
@@ -2282,6 +2437,8 @@ static const mrb_mt_entry string_ext_rom_entries[] = {
MRB_MT_ENTRY(str_b, MRB_SYM(b), MRB_ARGS_NONE()),
MRB_MT_ENTRY(str_lines, MRB_SYM(__lines), MRB_ARGS_NONE()),
MRB_MT_ENTRY(str_codepoints, MRB_SYM(__codepoints), MRB_ARGS_NONE()),
MRB_MT_ENTRY(str_scrub_core, MRB_SYM(__scrub), MRB_ARGS_OPT(1)),
MRB_MT_ENTRY(str_scrub_chunks, MRB_SYM(__scrub_chunks), MRB_ARGS_NONE()),
MRB_MT_ENTRY(str_lstrip, MRB_SYM(lstrip), MRB_ARGS_NONE()),
MRB_MT_ENTRY(str_rstrip, MRB_SYM(rstrip), MRB_ARGS_NONE()),
MRB_MT_ENTRY(str_strip, MRB_SYM(strip), MRB_ARGS_NONE()),
+57
View File
@@ -791,3 +791,60 @@ assert('String#-@') do
a = -(a.freeze)
assert_true(a.frozen?)
end
assert('String#scrub default replacement (U+FFFD)') do
# scrub has UTF-8 semantics; on builds without MRB_UTF8_STRING it
# degrades to a no-op (verified separately below).
skip unless "".length == 1
assert_equal "\u{FFFD}", "\xE3\x81".scrub
assert_equal "abc\u{FFFD}def", "abc\x80def".scrub
assert_equal "\u{FFFD}", "\x80\x81\x82".scrub # run collapsed
assert_equal "", "".scrub
assert_equal "hello", "hello".scrub # already valid
assert_equal "あい", "あい".scrub # already valid multibyte
end
assert('String#scrub rejects malformed sequences') do
skip unless "".length == 1
# overlong, UTF-16 surrogate, codepoint above U+10FFFF
assert_equal "\u{FFFD}", "\xC0\xAF".scrub # overlong "/"
assert_equal "\u{FFFD}", "\xED\xA0\x80".scrub # surrogate U+D800
assert_equal "\u{FFFD}", "\xF4\x90\x80\x80".scrub # > U+10FFFF
end
assert('String#scrub with replacement string') do
skip unless "".length == 1
assert_equal "abc?def", "abc\x80def".scrub("?")
assert_equal "abcdef", "abc\x80def".scrub("")
assert_equal "abc<bad>def", "abc\x80def".scrub("<bad>")
end
assert('String#scrub raises on invalid replacement') do
skip unless "".length == 1
assert_raise(ArgumentError) { "abc\x80".scrub("\xFF") }
end
assert('String#scrub with block') do
skip unless "".length == 1
assert_equal "abc<80>def",
"abc\x80def".scrub { |b| "<" + b.bytes.first.to_s(16) + ">" }
# Block not called when string is already valid
called = false
"hello".scrub { |_| called = true; "X" }
assert_false called
# Multiple invalid runs each get their own block invocation
result = "a\x80b\x81c".scrub { |b| "[#{b.bytes.first}]" }
assert_equal "a[128]b[129]c", result
# Non-String block return values are coerced via to_s (mruby leniency;
# CRuby raises TypeError instead). Locking this in so the choice is
# explicit and doesn't drift accidentally.
assert_equal "abc42def", "abc\x80def".scrub { 42 }
end
assert('String#scrub is a no-op without MRB_UTF8_STRING') do
skip if "".length == 1
# Method is still defined and returns a (string-equal) copy.
assert_equal "abc\x80def", "abc\x80def".scrub
assert_equal "abc\x80def", "abc\x80def".scrub("?")
assert_equal "abc\x80def", "abc\x80def".scrub { |_| "?" }
end
+4 -8
View File
@@ -41,7 +41,6 @@ struct mrb_task_queue;
* - Removed started flag (inferred from c.status): 1 byte
* - Unified wakeup_tick/join/mutex into single union: 4 bytes
* - Removed redundant proc field (stored in c.ci->proc): 8 bytes
* - Unified timeslice/result into state union: ~4 bytes
* Total savings: ~18 bytes per task (14% reduction)
*/
typedef struct mrb_task {
@@ -49,6 +48,7 @@ typedef struct mrb_task {
uint8_t priority; /* Priority (0-255, 0=highest) */
uint8_t status; /* Current status (TASKSTATUS enum) */
uint8_t reason; /* Wait reason (TASKREASON enum) */
volatile uint8_t timeslice; /* Remaining ticks while RUNNING */
mrb_value name; /* Optional task name */
/* Wait-specific data - mutually exclusive based on reason field */
@@ -61,11 +61,7 @@ typedef struct mrb_task {
mrb_value self; /* Ruby Task object reference */
/* State-specific data - mutually exclusive based on status */
union {
volatile uint8_t timeslice; /* Remaining ticks (RUNNING only) */
mrb_value result; /* Task return value (DORMANT only) */
} state;
mrb_value result; /* Task return value */
struct mrb_context c; /* Execution context (stack, callinfo, etc) */
} mrb_task;
@@ -166,8 +162,8 @@ task_check_scheduler_lock(mrb_state *mrb)
}
/* Priority-queue insert/delete - defined in task.c */
void q_insert_task(mrb_state *mrb, mrb_task *t);
void q_delete_task(mrb_state *mrb, mrb_task *t);
void mrb_task_q_insert(mrb_state *mrb, mrb_task *t);
void mrb_task_q_delete(mrb_state *mrb, mrb_task *t);
/* Task::Queue class registration - defined in task_queue.c */
void mrb_init_task_queue(mrb_state *mrb, struct RClass *task_class);
+113 -44
View File
@@ -94,6 +94,16 @@ mrb_task_mark_all(mrb_state *mrb)
for (i = 0; i < e; i++) {
mrb_gc_mark_value(mrb, c->stbase[i]);
}
/* Clear the dead slots above the live range, matching
mark_context_stack() in gc.c. A preempted task whose live range
later shrinks (a frame returned) would otherwise leave stale
object pointers in those slots; the objects get swept while the
pointers survive, and a subsequent mark of the resumed task trips
the MRB_TT_FREE assertion in mrb_gc_mark (issue #6870). */
size_t stend = c->stend - c->stbase;
for (; i < stend; i++) {
SET_NIL_VALUE(c->stbase[i]);
}
}
/* Mark call stack */
@@ -113,9 +123,7 @@ mrb_task_mark_all(mrb_state *mrb)
/* Mark task-specific values */
mrb_gc_mark_value(mrb, t->self);
if (t->status == MRB_TASK_STATUS_DORMANT) {
mrb_gc_mark_value(mrb, t->state.result);
}
mrb_gc_mark_value(mrb, t->result);
mrb_gc_mark_value(mrb, t->name);
t = t->next;
@@ -148,7 +156,7 @@ q_get_queue(mrb_state *mrb, mrb_task *t)
/* Insert task into queue based on priority (higher priority = lower number = earlier in queue) */
void
q_insert_task(mrb_state *mrb, mrb_task *t)
mrb_task_q_insert(mrb_state *mrb, mrb_task *t)
{
mrb_task **q = q_get_queue(mrb, t);
mrb_task *curr = *q;
@@ -172,7 +180,7 @@ q_insert_task(mrb_state *mrb, mrb_task *t)
/* Delete task from its current queue */
void
q_delete_task(mrb_state *mrb, mrb_task *t)
mrb_task_q_delete(mrb_state *mrb, mrb_task *t)
{
mrb_task **q = q_get_queue(mrb, t);
mrb_task *curr = *q;
@@ -202,10 +210,10 @@ task_cleanup_if_stopped(mrb_state *mrb, mrb_task *t)
if (t->status == MRB_TASK_STATUS_DORMANT || t->c.status == MRB_TASK_STOPPED) {
/* Task is terminated but still in queue - remove it */
mrb_task_disable_irq();
q_delete_task(mrb, t);
mrb_task_q_delete(mrb, t);
if (t->status != MRB_TASK_STATUS_DORMANT) {
t->status = MRB_TASK_STATUS_DORMANT;
q_insert_task(mrb, t);
mrb_task_q_insert(mrb, t);
}
mrb_task_enable_irq();
return TRUE;
@@ -286,11 +294,11 @@ wake_up_join_waiters(mrb_state *mrb, mrb_task *completed_task)
while (curr != NULL) {
mrb_task *next = curr->next;
if (curr->reason == MRB_TASK_REASON_JOIN && curr->wait.join == completed_task) {
q_delete_task(mrb, curr);
mrb_task_q_delete(mrb, curr);
curr->status = MRB_TASK_STATUS_READY;
curr->reason = MRB_TASK_REASON_NONE;
curr->wait.join = NULL;
q_insert_task(mrb, curr);
mrb_task_q_insert(mrb, curr);
/* If a higher-priority waiter is resumed from task context,
* request a context switch after leaving the critical section. */
if (mrb->c != mrb->root_c && !switching_) {
@@ -310,12 +318,33 @@ static void
task_change_state(mrb_state *mrb, mrb_task *t, uint8_t new_status)
{
mrb_task_disable_irq();
q_delete_task(mrb, t);
mrb_task_q_delete(mrb, t);
t->status = new_status;
q_insert_task(mrb, t);
mrb_task_q_insert(mrb, t);
mrb_task_enable_irq();
}
typedef struct execute_task_vm_args {
mrb_task *t;
const struct RProc *proc;
const mrb_code *pc;
} execute_task_vm_args;
static mrb_value
execute_task_vm(mrb_state *mrb, void *ud)
{
execute_task_vm_args *args = (execute_task_vm_args*)ud;
mrb->task.exception_as_result = TRUE;
args->t->result = mrb_vm_exec(mrb, args->proc, args->pc);
if (mrb->exc) {
args->t->result = mrb_obj_value(mrb->exc);
mrb->exc = NULL;
}
mrb->task.exception_as_result = FALSE;
return args->t->result;
}
/* Execute a single task - core task execution logic */
static void
execute_task(mrb_state *mrb, mrb_task *t)
@@ -325,8 +354,8 @@ execute_task(mrb_state *mrb, mrb_task *t)
uint8_t prev_cci;
/* Set task as running */
t->timeslice = MRB_TIMESLICE_TICK_COUNT;
t->status = MRB_TASK_STATUS_RUNNING;
t->state.timeslice = MRB_TIMESLICE_TICK_COUNT;
/* Switch to task context */
prev_c = mrb->c;
@@ -350,8 +379,13 @@ execute_task(mrb_state *mrb, mrb_task *t)
/* Set vmexec flag to prevent fiber_terminate from being called */
t->c.vmexec = TRUE;
/* Execute task - PC is saved in ci->pc from previous run */
t->state.result = mrb_vm_exec(mrb, proc, pc);
/* Execute task - PC is saved in ci->pc from previous run.
Unhandled task exceptions are converted to the task result by
mrb_vm_exec() in task mode, so the scheduler protect frame stays intact. */
execute_task_vm_args args = { t, proc, pc };
mrb_bool error = FALSE;
t->result = mrb_protect_error(mrb, execute_task_vm, &args, &error);
mrb->task.exception_as_result = FALSE;
/* Clear vmexec flag */
t->c.vmexec = FALSE;
@@ -365,13 +399,25 @@ execute_task(mrb_state *mrb, mrb_task *t)
prev_c->ci = prev_ci;
prev_ci->cci = prev_cci;
/* If an abnormal path inside mrb_vm_exec() bypassed
exception_as_result and unwound via MRB_THROW (e.g. a
CINFO_SKIP frame), mrb_protect_error caught it and stored the
exception object in t->result. Force the task to terminate
cleanly so the scheduler keeps running instead of aborting -
re-raising into the scheduler would abort in pattern 1, where
no outer jmpbuf exists. The exception remains observable via
mrb_task_value() / Task#value. */
if (error) {
t->c.status = MRB_TASK_STOPPED;
}
/* Handle task termination */
if (t->c.status == MRB_TASK_STOPPED) {
switching_ = FALSE;
mrb_task_disable_irq();
q_delete_task(mrb, t);
mrb_task_q_delete(mrb, t);
t->status = MRB_TASK_STATUS_DORMANT;
q_insert_task(mrb, t);
mrb_task_q_insert(mrb, t);
mrb_task_enable_irq();
/* Wake up tasks waiting on join */
@@ -394,9 +440,9 @@ mrb_tick(mrb_state *mrb)
/* Decrease timeslice for running task */
t = q_ready_;
if (t && t->status == MRB_TASK_STATUS_RUNNING && t->state.timeslice > 0) {
t->state.timeslice--;
if (t->state.timeslice == 0) {
if (t && t->status == MRB_TASK_STATUS_RUNNING && t->timeslice > 0) {
t->timeslice--;
if (t->timeslice == 0) {
switching_ = TRUE; /* Trigger context switch */
}
}
@@ -421,10 +467,10 @@ mrb_tick(mrb_state *mrb)
if (curr->reason == MRB_TASK_REASON_SLEEP) {
if ((int32_t)(curr->wait.wakeup_tick - tick_) <= 0) {
/* Time to wake up */
q_delete_task(mrb, curr);
mrb_task_q_delete(mrb, curr);
curr->status = MRB_TASK_STATUS_READY;
curr->reason = MRB_TASK_REASON_NONE;
q_insert_task(mrb, curr);
mrb_task_q_insert(mrb, curr);
switching_ = TRUE;
}
else if (curr->wait.wakeup_tick < next_wakeup) {
@@ -439,11 +485,14 @@ mrb_tick(mrb_state *mrb)
}
}
/* Main scheduler loop */
MRB_API mrb_value
mrb_task_run(mrb_state *mrb)
/* Body of the main scheduler loop. Wrapped by mrb_task_run() under
mrb_protect_error so an exception raised from a task body unwinds
cleanly without leaving `loop_running` set. */
static mrb_value
task_run_body(mrb_state *mrb, void *ud)
{
mrb_task *t;
(void)ud;
while (1) {
t = q_ready_;
@@ -480,10 +529,27 @@ mrb_task_run(mrb_state *mrb)
mrb_incremental_gc(mrb);
}
}
return mrb_nil_value();
}
/* Main scheduler loop */
MRB_API mrb_value
mrb_task_run(mrb_state *mrb)
{
if (mrb->task.loop_running) {
return mrb_nil_value();
}
mrb->task.loop_running = TRUE;
mrb_bool error = FALSE;
mrb_value result = mrb_protect_error(mrb, task_run_body, NULL, &error);
mrb->task.loop_running = FALSE;
if (error) {
mrb_exc_raise(mrb, result);
}
return result;
}
/* Single-step task execution for WASM event loop integration */
MRB_API mrb_value
mrb_task_run_once(mrb_state *mrb)
@@ -551,7 +617,7 @@ sleep_us_impl(mrb_state *mrb, uint32_t usec)
mrb_task_disable_irq();
/* Remove from ready queue */
q_delete_task(mrb, t);
mrb_task_q_delete(mrb, t);
/* Move to waiting queue */
t->status = MRB_TASK_STATUS_WAITING;
@@ -572,7 +638,7 @@ sleep_us_impl(mrb_state *mrb, uint32_t usec)
(int32_t)(t->wait.wakeup_tick - wakeup_tick_) < 0) {
wakeup_tick_ = t->wait.wakeup_tick;
}
q_insert_task(mrb, t);
mrb_task_q_insert(mrb, t);
mrb_task_enable_irq();
@@ -597,9 +663,9 @@ mrb_f_sleep(mrb_state *mrb, mrb_value self)
mrb_task *t = q_ready_;
if (t) {
mrb_task_disable_irq();
q_delete_task(mrb, t);
mrb_task_q_delete(mrb, t);
t->status = MRB_TASK_STATUS_SUSPENDED;
q_insert_task(mrb, t);
mrb_task_q_insert(mrb, t);
mrb_task_enable_irq();
switching_ = TRUE;
}
@@ -666,7 +732,7 @@ task_create_common(mrb_state *mrb, const struct RProc *proc,
task_init_context(mrb, t, proc);
mrb_task_disable_irq();
q_insert_task(mrb, t);
mrb_task_q_insert(mrb, t);
mrb_task_enable_irq();
if (q_ready_ && q_ready_->status == MRB_TASK_STATUS_RUNNING) {
@@ -1024,8 +1090,8 @@ mrb_task_set_priority(mrb_state *mrb, mrb_value self)
/* Re-sort in queue if task is ready */
if (t->status == MRB_TASK_STATUS_READY || t->status == MRB_TASK_STATUS_RUNNING) {
q_delete_task(mrb, t);
q_insert_task(mrb, t);
mrb_task_q_delete(mrb, t);
mrb_task_q_insert(mrb, t);
}
mrb_task_enable_irq();
@@ -1095,22 +1161,22 @@ mrb_task_join(mrb_state *mrb, mrb_value self)
/* If task is already dormant, return immediately */
if (t->status == MRB_TASK_STATUS_DORMANT) {
return t->state.result;
return t->result;
}
/* Wait for task to complete */
mrb_task_disable_irq();
q_delete_task(mrb, current);
mrb_task_q_delete(mrb, current);
current->status = MRB_TASK_STATUS_WAITING;
current->reason = MRB_TASK_REASON_JOIN;
current->wait.join = t;
q_insert_task(mrb, current);
mrb_task_q_insert(mrb, current);
mrb_task_enable_irq();
/* Trigger context switch */
switching_ = TRUE;
return t->state.result;
return t->result;
}
/*
@@ -1162,7 +1228,7 @@ mrb_execute_proc_synchronously(mrb_state *mrb, mrb_value proc_val, mrb_int argc,
/* 3. Move task from DORMANT to READY */
mrb_task_disable_irq();
t->status = MRB_TASK_STATUS_READY;
q_insert_task(mrb, t);
mrb_task_q_insert(mrb, t);
mrb_task_enable_irq();
/* 4. Execute the task in a dedicated loop (no context switching) */
@@ -1170,23 +1236,23 @@ mrb_execute_proc_synchronously(mrb_state *mrb, mrb_value proc_val, mrb_int argc,
mrb->c = &t->c;
while (t->c.status != MRB_TASK_STOPPED) {
t->state.result = mrb_vm_exec(mrb, mrb->c->ci->proc, mrb->c->ci->pc);
t->result = mrb_vm_exec(mrb, mrb->c->ci->proc, mrb->c->ci->pc);
}
/* If there's an unhandled exception after VM stops, save it as result */
if (mrb->exc) {
t->state.result = mrb_obj_value(mrb->exc);
t->result = mrb_obj_value(mrb->exc);
}
/* 5. Get result and clean up */
mrb_value result = t->state.result;
mrb_value result = t->result;
if (mrb_obj_ptr(result) == mrb->exc) {
mrb->exc = NULL; /* Clear exception */
}
/* 6. Free the temporary task's resources */
mrb_task_disable_irq();
q_delete_task(mrb, t);
mrb_task_q_delete(mrb, t);
mrb_task_enable_irq();
/* Prevent double-free: clear Data object's type before freeing task */
@@ -1362,10 +1428,10 @@ terminate_task_internal(mrb_state *mrb, mrb_task *t)
if (t->status == MRB_TASK_STATUS_DORMANT) return;
mrb_task_disable_irq();
q_delete_task(mrb, t);
mrb_task_q_delete(mrb, t);
t->status = MRB_TASK_STATUS_DORMANT;
t->c.status = MRB_TASK_STOPPED;
q_insert_task(mrb, t);
mrb_task_q_insert(mrb, t);
mrb_task_enable_irq();
wake_up_join_waiters(mrb, t);
@@ -1417,7 +1483,7 @@ mrb_task_value(mrb_state *mrb, mrb_value task)
mrb_task *t = (mrb_task*)mrb_data_check_get_ptr(mrb, task, &mrb_task_type);
if (!t) return mrb_nil_value();
return t->state.result;
return t->result;
}
/*
@@ -1505,6 +1571,8 @@ mrb_mruby_task_gem_init(mrb_state *mrb)
/* Initialize main task to NULL and scheduler_lock to 0 */
mrb->task.main_task = NULL;
mrb->task.scheduler_lock = 0;
mrb->task.loop_running = FALSE;
mrb->task.exception_as_result = FALSE;
task_class = mrb_define_class_id(mrb, MRB_SYM(Task), mrb->object_class);
MRB_SET_INSTANCE_TT(task_class, MRB_TT_DATA);
@@ -1536,6 +1604,7 @@ mrb_mruby_task_gem_init(mrb_state *mrb)
mrb_define_method_id(mrb, task_class, MRB_SYM(resume), mrb_task_resume, MRB_ARGS_NONE());
mrb_define_method_id(mrb, task_class, MRB_SYM(terminate), mrb_task_terminate, MRB_ARGS_NONE());
mrb_define_method_id(mrb, task_class, MRB_SYM(join), mrb_task_join, MRB_ARGS_NONE());
mrb_define_method_id(mrb, task_class, MRB_SYM(value), mrb_task_value, MRB_ARGS_NONE());
/* Kernel methods (module functions like CRuby)
* Note: sleep and usleep override mruby-sleep's implementation to be task-aware
+6 -6
View File
@@ -36,11 +36,11 @@ queue_wake_one_waiter(mrb_state *mrb, mrb_task_queue *q)
while (curr) {
mrb_task *next = curr->next;
if (curr->reason == MRB_TASK_REASON_QUEUE && curr->wait.queue == q) {
q_delete_task(mrb, curr);
mrb_task_q_delete(mrb, curr);
curr->status = MRB_TASK_STATUS_READY;
curr->reason = MRB_TASK_REASON_NONE;
curr->wait.queue = NULL;
q_insert_task(mrb, curr);
mrb_task_q_insert(mrb, curr);
switching_ = TRUE;
break;
}
@@ -59,11 +59,11 @@ queue_wake_all_waiters(mrb_state *mrb, mrb_task_queue *q)
while (curr) {
mrb_task *next = curr->next;
if (curr->reason == MRB_TASK_REASON_QUEUE && curr->wait.queue == q) {
q_delete_task(mrb, curr);
mrb_task_q_delete(mrb, curr);
curr->status = MRB_TASK_STATUS_READY;
curr->reason = MRB_TASK_REASON_NONE;
curr->wait.queue = NULL;
q_insert_task(mrb, curr);
mrb_task_q_insert(mrb, curr);
woke_any = TRUE;
}
curr = next;
@@ -154,11 +154,11 @@ queue_pop_try(mrb_state *mrb, mrb_value self)
/* Move current task to WAITING */
mrb_task *current = MRB2TASK(mrb);
mrb_task_disable_irq();
q_delete_task(mrb, current);
mrb_task_q_delete(mrb, current);
current->status = MRB_TASK_STATUS_WAITING;
current->reason = MRB_TASK_REASON_QUEUE;
current->wait.queue = q;
q_insert_task(mrb, current);
mrb_task_q_insert(mrb, current);
mrb_task_enable_irq();
switching_ = TRUE;
+25
View File
@@ -81,6 +81,10 @@ end
assert("Task#suspend doesn't raise") do
task = Task.new { }
assert_nothing_raised { task.suspend }
# Clean up: a suspended task left in q_suspended_ keeps a later
# Task.run from terminating (the scheduler idles waiting on it
# instead of exiting).
task.terminate
end
assert("Task#resume doesn't raise") do
@@ -183,3 +187,24 @@ assert("Task.new with block doesn't execute immediately") do
# Block should not execute until scheduler runs
assert_false executed
end
assert("Task.run inside Task.run is a noop") do
assert_nothing_raised do
Task.new { Task.run }
Task.run
end
end
assert("Task#value returns exception object for unhandled task errors") do
child = nil
Task.new do
child = Task.new { raise "boom" }
end
Task.run
result = child.value
assert_kind_of RuntimeError, result
assert_equal "boom", result.message
end
+8
View File
@@ -1197,6 +1197,14 @@ mrb_ary_splice(mrb_state *mrb, mrb_value ary, mrb_int head, mrb_int len, mrb_val
}
r = ary_dup(mrb, a);
argv = ARY_PTR(r);
/* ary_dup -> ary_replace converts `a` to shared as a
copy-on-write optimization when len > ARY_REPLACE_SHARED_MIN.
Subsequent ARY_CAPA(a) reads would land on aux.shared's
pointer bits instead of the actual capacity, so the
expand-capa check below silently mis-sizes and value_move
walks past the buffer. Re-modify here to unshare before
mutating `a` in place. */
ary_modify(mrb, a);
}
}
else if (mrb_undef_p(rpl)) {
+1 -1
View File
@@ -3966,7 +3966,7 @@ mrb_method_added(mrb_state *mrb, struct RClass *c, mrb_sym mid)
}
}
mrb_value
static mrb_value
define_method_m(mrb_state *mrb, struct RClass *c, int vis)
{
mrb_sym mid;
+27 -30
View File
@@ -1216,36 +1216,33 @@ incremental_sweep_phase(mrb_state *mrb, mrb_gc *gc, size_t limit)
size_t tried_sweep = 0;
while (page && (tried_sweep < limit)) {
RVALUE *p = page->objects;
RVALUE *e = p + MRB_HEAP_PAGE_SIZE;
size_t freed = 0;
mrb_bool dead_slot = TRUE;
if (is_minor_gc(gc) && page->old) {
/* skip a slot which doesn't contain any young object */
p = e;
dead_slot = FALSE;
}
while (p<e) {
if (is_dead(gc, &p->as.basic)) {
if (p->as.basic.tt != MRB_TT_FREE) {
obj_free(mrb, &p->as.basic, FALSE);
if (p->as.basic.tt == MRB_TT_FREE) {
else {
RVALUE *p = page->objects;
RVALUE *e = p + MRB_HEAP_PAGE_SIZE;
while (p<e) {
if (is_dead(gc, &p->as.basic)) {
if (p->as.basic.tt != MRB_TT_FREE) {
obj_free(mrb, &p->as.basic, FALSE);
mrb_assert(p->as.basic.tt == MRB_TT_FREE);
p->as.free.next = page->freelist;
page->freelist = p;
freed++;
}
else {
dead_slot = FALSE;
}
}
else {
if (!is_generational(gc))
paint_partial_white(gc, &p->as.basic); /* next gc target */
dead_slot = FALSE;
}
p++;
}
else {
if (!is_generational(gc))
paint_partial_white(gc, &p->as.basic); /* next gc target */
dead_slot = FALSE;
}
p++;
}
/* free dead slot */
@@ -1802,21 +1799,21 @@ gc_stat(mrb_state *mrb, mrb_value self)
mrb_gc *gc = &mrb->gc;
mrb_value hash = mrb_hash_new_capa(mrb, 8);
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "live")), mrb_int_value(mrb, (mrb_int)gc->live));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "debt")), mrb_int_value(mrb, gc->gc_debt));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "state")), mrb_int_value(mrb, (mrb_int)gc->state));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "generational")), mrb_bool_value(gc->generational));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "full")), mrb_bool_value(gc->full));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "step_limit")), mrb_int_value(mrb, (mrb_int)gc->step_limit));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "malloc_increase")), mrb_int_value(mrb, (mrb_int)gc->malloc_increase));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "malloc_threshold")), mrb_int_value(mrb, (mrb_int)gc->malloc_threshold));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "symbol_count")), mrb_int_value(mrb, (mrb_int)(MRB_PRESYM_MAX + mrb->symidx)));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "dynamic_symbol_count")), mrb_int_value(mrb, (mrb_int)mrb->dynamic_sym_count));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(live)), mrb_int_value(mrb, (mrb_int)gc->live));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(debt)), mrb_int_value(mrb, gc->gc_debt));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(state)), mrb_int_value(mrb, (mrb_int)gc->state));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(generational)), mrb_bool_value(gc->generational));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(full)), mrb_bool_value(gc->full));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(step_limit)), mrb_int_value(mrb, (mrb_int)gc->step_limit));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(malloc_increase)), mrb_int_value(mrb, (mrb_int)gc->malloc_increase));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(malloc_threshold)), mrb_int_value(mrb, (mrb_int)gc->malloc_threshold));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(symbol_count)), mrb_int_value(mrb, (mrb_int)(MRB_PRESYM_MAX + mrb->symidx)));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(dynamic_symbol_count)), mrb_int_value(mrb, (mrb_int)mrb->dynamic_sym_count));
#ifdef MRB_GC_STATS
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "total")), mrb_int_value(mrb, (mrb_int)gc->gc_total_count));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "minor")), mrb_int_value(mrb, (mrb_int)gc->minor_gc_count));
mrb_hash_set(mrb, hash, mrb_symbol_value(mrb_intern_lit(mrb, "major")), mrb_int_value(mrb, (mrb_int)gc->major_gc_count));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(total)), mrb_int_value(mrb, (mrb_int)gc->gc_total_count));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(minor)), mrb_int_value(mrb, (mrb_int)gc->minor_gc_count));
mrb_hash_set(mrb, hash, mrb_symbol_value(MRB_SYM(major)), mrb_int_value(mrb, (mrb_int)gc->major_gc_count));
#endif
return hash;
+8
View File
@@ -351,6 +351,14 @@ flo_idiv(mrb_state *mrb, mrb_value xv)
mrb_float x = mrb_float(xv);
mrb_check_num_exact(mrb, x);
mrb_int y = mrb_as_int(mrb, mrb_get_arg1(mrb));
/* (mrb_int)x is UB when x is outside mrb_int range. */
if (!FIXABLE_FLOAT(x)) {
#ifdef MRB_USE_BIGINT
return mrb_bint_div(mrb, mrb_bint_new_float(mrb, x), mrb_int_value(mrb, y));
#else
mrb_int_overflow(mrb, "div");
#endif
}
return mrb_div_int_value(mrb, (mrb_int)x, y);
}
+2 -7
View File
@@ -665,7 +665,8 @@ assign_class_name(mrb_state *mrb, struct RObject *obj, mrb_sym sym, mrb_value v)
{
if (namespace_p(mrb_type(v))) {
struct RObject *c = mrb_obj_ptr(v);
if (obj != c && ISUPPER(mrb_sym_name_len(mrb, sym, NULL)[0])) {
const char *name = mrb_sym_name_len(mrb, sym, NULL);
if (obj != c && name && ISUPPER(name[0])) {
mrb_sym id_classname = MRB_SYM(__classname__);
mrb_value o = mrb_obj_iv_get(mrb, c, id_classname);
@@ -1443,9 +1444,7 @@ mrb_const_set(mrb_state *mrb, mrb_value mod, mrb_sym sym, mrb_value v)
mrb_class_name_class(mrb, mrb_class_ptr(mod), mrb_class_ptr(v), sym);
}
mrb_obj_iv_set(mrb, mrb_obj_ptr(mod), sym, v);
#ifndef MRB_NO_CONST_CACHE
mrb_const_cache_clear(mrb);
#endif
if (!mrb->bootstrapping) {
mrb_value name = mrb_symbol_value(sym);
@@ -1472,9 +1471,7 @@ mrb_const_remove(mrb_state *mrb, mrb_value mod, mrb_sym sym)
{
mod_const_check(mrb, mod);
mrb_iv_remove(mrb, mod, sym);
#ifndef MRB_NO_CONST_CACHE
mrb_const_cache_clear(mrb);
#endif
}
/*
@@ -1492,9 +1489,7 @@ MRB_API void
mrb_define_const_id(mrb_state *mrb, struct RClass *mod, mrb_sym name, mrb_value v)
{
mrb_obj_iv_set(mrb, (struct RObject*)mod, name, v);
#ifndef MRB_NO_CONST_CACHE
mrb_const_cache_clear(mrb);
#endif
}
/*
+43 -1
View File
@@ -1588,17 +1588,58 @@ prepare_tagged_break(mrb_state *mrb, uint32_t tag, const mrb_callinfo *return_ci
#define CALL_CODE_HOOKS() do { insn = BYTECODE_DECODER(*ci->pc); CODE_FETCH_HOOK(mrb, irep, ci->pc, regs); } while (0)
#ifdef MRB_USE_TASK_SCHEDULER
/* TRUE when the current context is executing across a C call boundary, i.e.
a C function on the stack re-entered the VM (mrb_funcall / mrb_yield /
mrb_vm_run). A task cannot be suspended at such a point: the C stack
frame between the scheduler's mrb_vm_exec and the current frame cannot
be saved or restored, and returning early from the inner mrb_vm_exec
would leave the call-info stack drifted, tripping the assertion in
mrb_vm_run (issues #6864, #6868). The scheduler defers the switch until
execution unwinds back to a frame with no C boundary. This mirrors the
cooperative guard in Task.pass, which raises rather than defers.
cibase is excluded: it is the entry frame of this mrb_vm_exec. */
static mrb_bool
task_across_c_boundary(mrb_state *mrb)
{
for (mrb_callinfo *ci = mrb->c->ci; ci > mrb->c->cibase; ci--) {
if (ci->cci > 0) return TRUE;
}
return FALSE;
}
/* Defer task switches while a C-level ObjectSpace walk holds gc.iterating
true. The walk runs callbacks (which may call back into mrb_vm_exec via
mrb_yield); returning early from an inner exec while the outer C
iteration is still active drifts the call-info stack and eventually
crashes (issue #6862). Switches resume at the next OP boundary after
the walk releases gc.iterating. A pending switch is also deferred while
executing across a C call boundary (see task_across_c_boundary). A
pending MRB_TASK_STOPPED is not deferred, since the task is going away.
mrb->jmp is restored to prev_jmp before returning, exactly as the
normal return paths below do. mrb_vm_exec set mrb->jmp to its own
stack-local c_jmp on entry; leaving it dangling after this early return
means a later raise longjmps into a freed frame (issue #6863).
This macro must only be expanded where prev_jmp is in scope, i.e.
inside mrb_vm_exec (via NEXT / END_DISPATCH). */
#define RETURN_IF_TASK_STOPPED(mrb) do { \
if ((mrb)->task.switching || (mrb)->c->status == MRB_TASK_STOPPED) \
if (((mrb)->task.switching && !(mrb)->gc.iterating && \
!task_across_c_boundary(mrb)) || \
(mrb)->c->status == MRB_TASK_STOPPED) { \
(mrb)->jmp = prev_jmp; \
return mrb_nil_value(); \
} \
} while (0)
#define TASK_STOP(mrb) do { \
if (mrb->c->status != MRB_TASK_STOPPED) \
mrb->c->status = MRB_TASK_STOPPED; \
} while (0)
#define TASK_RETURN_EXCEPTION_AS_VALUE(mrb) ((mrb)->task.exception_as_result)
#else
#define RETURN_IF_TASK_STOPPED(mrb)
#define TASK_STOP(mrb)
#define TASK_RETURN_EXCEPTION_AS_VALUE(mrb) FALSE
#endif
/**
@@ -2652,6 +2693,7 @@ RETRY_TRY_BLOCK:
fiber_terminate(mrb, c, ci);
if (mrb_unlikely(!c->vmexec)) goto L_RAISE;
mrb->jmp = prev_jmp;
if (TASK_RETURN_EXCEPTION_AS_VALUE(mrb)) return mrb_obj_value(mrb->exc);
if (!prev_jmp) return mrb_obj_value(mrb->exc);
MRB_THROW(prev_jmp);
}
+2 -1
View File
@@ -1,4 +1,5 @@
$:.unshift File.dirname(File.dirname(File.expand_path(__FILE__)))
require 'shellwords'
require 'test/assert.rb'
GEMNAME = ""
@@ -16,7 +17,7 @@ def cmd_list(s)
path_list = [cmd_bin(s)]
emu = ENV['EMULATOR']
path_list.unshift emu if emu && !emu.empty?
path_list.unshift(*Shellwords.split(emu)) if emu && !emu.empty?
path_list
end
+11
View File
@@ -107,6 +107,17 @@ assert('Array#[]=', '15.2.12.5.5') do
a = [1,2,3]
a[-1,0] = a
assert_equal([1,2,1,2,3,3], a)
# passing self with length above ARY_REPLACE_SHARED_MIN (=20).
# ary_dup -> ary_replace converts the source to shared as a
# copy-on-write optimization; without re-modifying `a` afterwards,
# ARY_CAPA(a) reads from aux.shared's pointer bits and the
# expand-capa check silently mis-sizes -> heap-buffer-overflow in
# value_move. Reported via clusterfuzz mruby_fuzzer.
a = (0..30).to_a
a[3, 2] = a
assert_equal(60, a.length)
assert_equal([0, 1, 2] + (0..30).to_a + (5..30).to_a, a)
end
assert('Array#clear', '15.2.12.5.6') do
+26
View File
@@ -194,3 +194,29 @@ assert('register window of calls (#3783)') do
end
end
end
assert('bare `nil?` in if/unless uses self as receiver (#6874)') do
klass = Class.new do
def unless_form
reached = false
unless nil?
reached = true
end
reached
end
def if_form
if nil?
:yes
else
:no
end
end
end
assert_true klass.new.unless_form
assert_equal :no, klass.new.if_form
# Sanity: explicit literal nil receiver still optimized correctly.
result = if nil.nil? then :yes else :no end
assert_equal :yes, result
end
+15
View File
@@ -1043,6 +1043,21 @@ assert('pattern matching - array patterns') do
x
end
assert_equal 3, result
# array literal with splat as case value (#6854):
# the array-literal-length optimization must bail out for splat,
# since the runtime length is unknown statically.
a = [1, 2]
result = case [*a]
in [1, 2] then :match
else :nomatch
end
assert_equal :match, result
# same bug in one-line `in` pattern
assert_true ([*a] in [1, 2])
assert_false ([*a] in [1, 2, 3])
assert_true ([1, *a, 4] in [1, 1, 2, 4])
end
assert('pattern matching - find patterns') do
+1 -1
View File
@@ -42,7 +42,7 @@ assert('mrb_vformat') do
assert_equal '`S`: {a: 1, "b" => "c"}', vf.v('`S`: %S', {a: 1, "b" => ?c})
assert_equal 'percent: %', vf.z('percent: %%')
assert_equal '"I": inspect char', vf.c('%!c: inspect char', ?I)
assert_equal '709: inspect mrb_int', vf.i('%!d: inspect mrb_int', 709)
assert_equal '709: inspect mrb_int', vf.i('%!i: inspect mrb_int', 709)
assert_equal '"a\x00b\xff"', vf.l('%!l', "a\000b\xFFc\000d", 4)
assert_equal ':"&.": inspect symbol', vf.n('%!n: inspect symbol', :'&.')
assert_equal 'inspect "String"', vf.v('inspect %!v', 'String')