mirror of
https://github.com/LLVMParty/llvm-nanobind
synced 2026-06-21 13:43:38 +00:00
Refresh public examples
This commit is contained in:
@@ -13,14 +13,20 @@ _Note_: This project is 90%+ vibe coded. It is mostly an experiment to see what
|
||||
- Comprehensive LLVM-C API coverage (~7300 lines of bindings)
|
||||
- Memory-safe: validity tokens prevent use-after-free crashes
|
||||
- Type-safe: auto-generated `.pyi` stubs for IDE support
|
||||
- Tested: 25+ lit tests, 15 golden master test pairs
|
||||
- Tested: golden-master tests, regression scripts, pytest, and vendored `llvm-c-test` lit tests
|
||||
|
||||
## Installation
|
||||
|
||||
This package requires LLVM 21+ to be installed. The build will automatically find LLVM if it's in your PATH, or you can specify the path:
|
||||
Released wheels bundle LLVM for supported platforms, so normal installation is:
|
||||
|
||||
```bash
|
||||
export CMAKE_PREFIX_PATH=/path/to/llvm
|
||||
pip install llvm-nanobind
|
||||
```
|
||||
|
||||
Source builds need LLVM 21.1.6 available to CMake. Set `LLVM_ROOT` if LLVM is not in a standard location:
|
||||
|
||||
```bash
|
||||
export LLVM_ROOT=/path/to/llvm
|
||||
pip install .
|
||||
```
|
||||
|
||||
@@ -31,26 +37,28 @@ See [llvm-nanobind-example](https://github.com/LLVMParty/llvm-nanobind-example)
|
||||
```python
|
||||
import llvm
|
||||
|
||||
# Create a simple function that returns 42
|
||||
# Create a simple function that returns 42.
|
||||
with llvm.create_context() as ctx:
|
||||
i32 = ctx.types.i32
|
||||
fn_type = ctx.types.function(i32, [])
|
||||
|
||||
with ctx.create_module("example") as mod:
|
||||
# Create function type: i32 ()
|
||||
i32 = ctx.int32_type()
|
||||
fn_type = ctx.function_type(i32, [])
|
||||
|
||||
# Create function and basic block
|
||||
fn = mod.add_function("get_answer", fn_type)
|
||||
bb = fn.append_basic_block("entry")
|
||||
|
||||
# Build return instruction
|
||||
with ctx.create_builder() as builder:
|
||||
builder.position_at_end(bb)
|
||||
builder.ret(llvm.const_int(i32, 42))
|
||||
|
||||
# Print the IR
|
||||
entry = fn.append_basic_block("entry")
|
||||
|
||||
with entry.create_builder() as builder:
|
||||
builder.ret(i32.constant(42))
|
||||
|
||||
assert mod.verify(), mod.get_verification_error()
|
||||
print(mod)
|
||||
```
|
||||
|
||||
The same code is kept as a runnable smoke-tested script in `examples/quick_start.py`:
|
||||
|
||||
```bash
|
||||
uv run python examples/quick_start.py
|
||||
```
|
||||
|
||||
## Development
|
||||
|
||||
### Setup
|
||||
@@ -122,6 +130,15 @@ uv run coverage combine
|
||||
uv run coverage report --include="llvm_c_test/*"
|
||||
```
|
||||
|
||||
### Examples
|
||||
|
||||
Runnable examples live in `examples/` and are covered by `tests/test_examples.py`:
|
||||
|
||||
```bash
|
||||
uv run python examples/quick_start.py
|
||||
uv run python examples/transform_replace_add.py
|
||||
```
|
||||
|
||||
## Documentation
|
||||
|
||||
Type stubs are auto-generated and provide IDE intellisense. After building, find them at:
|
||||
|
||||
@@ -6,18 +6,10 @@ This document explains the principles behind the llvm-nanobind API refactor.
|
||||
|
||||
## Why This Refactor Matters
|
||||
|
||||
The current API mirrors the LLVM C API directly. While this made initial implementation easier, it creates a **non-Pythonic** experience:
|
||||
The earliest API mirrored the LLVM C API directly. While this made initial implementation easier, it created a **non-Pythonic** experience. The current API is object-oriented and discoverable:
|
||||
|
||||
```python
|
||||
# Current: C-style global functions
|
||||
i32 = ctx.int32_type()
|
||||
const = llvm.const_int(i32, 42)
|
||||
llvm.add_attribute_at_index(func, 0, attr)
|
||||
llvm.set_metadata(inst, kind, md)
|
||||
```
|
||||
|
||||
```python
|
||||
# Target: Pythonic object-oriented API
|
||||
# Current: Pythonic object-oriented API
|
||||
i32 = ctx.types.i32
|
||||
const = i32.constant(42)
|
||||
func.add_attribute(0, attr)
|
||||
@@ -72,15 +64,11 @@ In Python, these naturally group:
|
||||
**Why**: Reduces boilerplate and reveals structure:
|
||||
|
||||
```python
|
||||
# Verbose: each type requires a method call
|
||||
i8 = ctx.int8_type()
|
||||
i16 = ctx.int16_type()
|
||||
i32 = ctx.int32_type()
|
||||
|
||||
# Concise: types are organized under a namespace
|
||||
# Types are organized under a namespace
|
||||
i8 = ctx.types.i8
|
||||
i16 = ctx.types.i16
|
||||
i32 = ctx.types.i32
|
||||
ptr = ctx.types.ptr
|
||||
```
|
||||
|
||||
The `ctx.types` namespace makes it clear these are all type-related, and IDE autocomplete shows all available types.
|
||||
@@ -150,46 +138,24 @@ With global functions, chaining is impossible.
|
||||
|
||||
### Visual Comparison
|
||||
|
||||
**Current (C-style)**:
|
||||
```python
|
||||
import llvm
|
||||
|
||||
with llvm.create_context() as ctx:
|
||||
i32 = ctx.int32_type()
|
||||
i64 = ctx.int64_type()
|
||||
fn_ty = ctx.function_type(i32, [i32, i32])
|
||||
|
||||
const_42 = llvm.const_int(i32, 42)
|
||||
const_0 = llvm.const_null(i32)
|
||||
|
||||
with ctx.create_module("test") as mod:
|
||||
fn = mod.add_function("add", fn_ty)
|
||||
llvm.add_attribute_at_index(fn, 0, attr)
|
||||
|
||||
with ctx.create_builder() as builder:
|
||||
phi = builder.phi(i32)
|
||||
llvm.phi_add_incoming(phi, val, bb) # Easy to misuse
|
||||
```
|
||||
|
||||
**Target (Pythonic)**:
|
||||
**Current (Pythonic)**:
|
||||
```python
|
||||
import llvm
|
||||
|
||||
with llvm.create_context() as ctx:
|
||||
i32 = ctx.types.i32
|
||||
i64 = ctx.types.i64
|
||||
fn_ty = ctx.types.function(i32, [i32, i32])
|
||||
|
||||
const_42 = i32.constant(42)
|
||||
const_0 = i32.null()
|
||||
|
||||
with ctx.create_module("test") as mod:
|
||||
fn = mod.add_function("add", fn_ty)
|
||||
fn.add_attribute(0, attr)
|
||||
lhs, rhs = fn.params
|
||||
entry = fn.append_basic_block("entry")
|
||||
|
||||
with ctx.create_builder() as builder:
|
||||
phi = builder.phi(i32)
|
||||
phi.add_incoming(val, bb) # Clear: operating on the phi
|
||||
with entry.create_builder() as builder:
|
||||
result = builder.add(lhs, rhs, "sum")
|
||||
builder.ret(result)
|
||||
|
||||
assert mod.verify(), mod.get_verification_error()
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
+12
-16
@@ -81,7 +81,7 @@ All lifetime-related errors:
|
||||
|
||||
```python
|
||||
with llvm.create_context() as ctx:
|
||||
val = llvm.const_int(ctx.int32_type(), 42)
|
||||
val = ctx.types.i32.constant(42)
|
||||
|
||||
# Context is destroyed
|
||||
val.is_constant # raises LLVMMemoryError - PROGRAM TERMINATES
|
||||
@@ -213,8 +213,7 @@ with llvm.create_context() as ctx:
|
||||
func = mod.add_function("foo", func_type)
|
||||
entry = func.append_basic_block("entry")
|
||||
|
||||
with ctx.create_builder() as builder:
|
||||
builder.position_at_end(entry)
|
||||
with entry.create_builder() as builder:
|
||||
inst = builder.add(a, b, name="sum")
|
||||
|
||||
# Save references
|
||||
@@ -261,17 +260,16 @@ with llvm.create_context() as ctx:
|
||||
entry = func.append_basic_block("entry")
|
||||
|
||||
# Builder can outlive or be shorter-lived than blocks
|
||||
with ctx.create_builder() as builder:
|
||||
builder.position_at_end(entry)
|
||||
with entry.create_builder() as builder:
|
||||
inst = builder.add(a, b)
|
||||
# Builder disposed, but entry and inst still valid
|
||||
|
||||
inst.name # Works fine
|
||||
|
||||
# Can create new builder later
|
||||
with ctx.create_builder() as builder2:
|
||||
builder2.position_before(inst)
|
||||
# Insert more instructions
|
||||
# Can create new builder later, positioned before an instruction
|
||||
with inst.create_builder() as builder2:
|
||||
# Insert more instructions before inst
|
||||
...
|
||||
```
|
||||
|
||||
### Type Mismatches Raise AssertionError
|
||||
@@ -280,8 +278,8 @@ with llvm.create_context() as ctx:
|
||||
import llvm
|
||||
|
||||
with llvm.create_context() as ctx:
|
||||
int_ty = ctx.int32_type()
|
||||
float_ty = ctx.float_type()
|
||||
int_ty = ctx.types.i32
|
||||
float_ty = ctx.types.f32
|
||||
|
||||
# Correct usage
|
||||
width = int_ty.int_width # Works: 32
|
||||
@@ -357,8 +355,7 @@ with llvm.create_context() as ctx:
|
||||
bb1 = func.append_basic_block("bb1")
|
||||
bb2 = func.append_basic_block("bb2")
|
||||
|
||||
with ctx.create_builder() as builder:
|
||||
builder.position_at_end(bb1)
|
||||
with bb1.create_builder() as builder:
|
||||
inst1 = builder.add(a, b, name="sum")
|
||||
inst2 = builder.mul(inst1, c, name="prod")
|
||||
|
||||
@@ -383,8 +380,7 @@ with llvm.create_context() as ctx:
|
||||
entry = func.append_basic_block("entry")
|
||||
other = func.append_basic_block("other")
|
||||
|
||||
with ctx.create_builder() as builder:
|
||||
builder.position_at_end(entry)
|
||||
with entry.create_builder() as builder:
|
||||
add = builder.add(a, b, name="sum")
|
||||
mul = builder.mul(add, c, name="prod")
|
||||
builder.ret(mul)
|
||||
@@ -551,7 +547,7 @@ with ctx.parse_bitcode_from_bytes(bitcode) as src:
|
||||
with ctx.create_module(src.name) as dst:
|
||||
# TypeCloner needs the destination module's context
|
||||
dst_ctx = llvm.get_module_context(dst)
|
||||
int_ty = dst_ctx.int32_type() # Create type in correct context
|
||||
int_ty = dst_ctx.types.i32 # Create type in correct context
|
||||
```
|
||||
|
||||
Previously, `get_module_context()` was broken - it always returned the global context instead of the module's actual context. This caused problems with context-specific features like custom syncscopes (e.g., `syncscope("agent")`).
|
||||
|
||||
+97
-189
@@ -1,6 +1,6 @@
|
||||
# Porting C++ LLVM Code to Python Bindings
|
||||
|
||||
This guide documents the experience of porting the LLVM Obfuscator passes from C++ to Python. It covers API differences, missing functionality, confusing patterns, and workarounds discovered during the port.
|
||||
This guide documents the experience of porting the LLVM Obfuscator passes from C++ to Python. It covers current API differences, transformation patterns, and remaining gaps discovered during the port.
|
||||
|
||||
## Table of Contents
|
||||
|
||||
@@ -14,7 +14,7 @@ This guide documents the experience of porting the LLVM Obfuscator passes from C
|
||||
8. [Global Variables](#global-variables)
|
||||
9. [Constants](#constants)
|
||||
10. [Memory Management](#memory-management)
|
||||
11. [Missing APIs](#missing-apis)
|
||||
11. [Remaining API Differences](#remaining-api-differences)
|
||||
12. [API Naming Differences](#api-naming-differences)
|
||||
13. [Common Pitfalls](#common-pitfalls)
|
||||
|
||||
@@ -79,21 +79,16 @@ Type* ptr = builder.getPtrTy();
|
||||
|
||||
**Python:**
|
||||
```python
|
||||
i32 = ctx.types.i32 # Property, not method
|
||||
i64 = ctx.types.i64 # Property, not method
|
||||
ptr = ctx.types.ptr() # METHOD - must call!
|
||||
i32 = ctx.types.i32
|
||||
i64 = ctx.types.i64
|
||||
ptr = ctx.types.ptr
|
||||
```
|
||||
|
||||
**Critical Difference:** Most types are properties (`i32`, `i64`, `i8`, `f32`, etc.) but `ptr()` is a METHOD that must be called:
|
||||
Most common scalar types are properties on `ctx.types`. Composite types are factory methods:
|
||||
|
||||
```python
|
||||
# WRONG
|
||||
ptr_ty = ctx.types.ptr # This is a bound method object!
|
||||
array_ty = ctx.types.array(ptr_ty, 2) # TypeError!
|
||||
|
||||
# CORRECT
|
||||
ptr_ty = ctx.types.ptr() # Call the method
|
||||
array_ty = ctx.types.array(ptr_ty, 2) # Works
|
||||
array_ty = ctx.types.array(ptr, 2)
|
||||
fn_ty = ctx.types.function(i32, [ptr])
|
||||
```
|
||||
|
||||
### Creating Array Types
|
||||
@@ -170,21 +165,16 @@ unsigned numOps = inst->getNumOperands();
|
||||
|
||||
**Python:**
|
||||
```python
|
||||
op0 = inst.get_operand(0) # Method, not property
|
||||
op0 = inst.get_operand(0)
|
||||
op1 = inst.get_operand(1)
|
||||
num_ops = inst.num_operands # Property
|
||||
num_ops = inst.num_operands
|
||||
|
||||
# Or iterate over a snapshot of operands:
|
||||
for op in inst.operands:
|
||||
...
|
||||
```
|
||||
|
||||
**Note:** There is NO `.operands` iterator property. You must use index-based access:
|
||||
|
||||
```python
|
||||
# WRONG
|
||||
for op in inst.operands: # AttributeError!
|
||||
|
||||
# CORRECT
|
||||
for i in range(inst.num_operands):
|
||||
op = inst.get_operand(i)
|
||||
```
|
||||
Use `get_operand()`/`set_operand()` when you need indexed mutation. Use `inst.operands` for simple iteration.
|
||||
|
||||
### Setting Operands
|
||||
|
||||
@@ -207,7 +197,8 @@ if (inst->isTerminator()) { ... }
|
||||
|
||||
**Python:**
|
||||
```python
|
||||
if inst.is_terminator_inst: # Note the suffix
|
||||
if inst.is_terminator:
|
||||
...
|
||||
```
|
||||
|
||||
Other properties:
|
||||
@@ -223,7 +214,8 @@ BasicBlock* bb = inst->getParent();
|
||||
|
||||
**Python:**
|
||||
```python
|
||||
bb = inst.block # NOT .parent!
|
||||
bb = inst.block # explicit instruction/basic-block relationship
|
||||
bb = inst.parent # alias
|
||||
```
|
||||
|
||||
---
|
||||
@@ -294,9 +286,17 @@ bb.move_before(other_bb)
|
||||
|
||||
### Splitting Blocks
|
||||
|
||||
**MISSING API:** There is no `splitBasicBlock()` or `splitBefore()` method in the Python bindings. You cannot split a basic block at an instruction.
|
||||
**C++:**
|
||||
```cpp
|
||||
BasicBlock* tail = bb->splitBasicBlock(inst, "tail");
|
||||
```
|
||||
|
||||
**Workaround:** Create a new block and manually move instructions (complex, not fully supported).
|
||||
**Python:**
|
||||
```python
|
||||
tail = bb.split_basic_block(inst, "tail")
|
||||
# or split before the instruction instead of after it
|
||||
tail = bb.split_basic_block_before(inst, "tail")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
@@ -616,18 +616,11 @@ Constant* str = ConstantDataArray::getString(ctx, "hello", true);
|
||||
**Python:**
|
||||
```python
|
||||
str_const = llvm.const_string(ctx, "hello", dont_null_terminate=False)
|
||||
raw_const = llvm.const_string(ctx, b"\xff\x80B", dont_null_terminate=True)
|
||||
assert raw_const.type.array_length == 3
|
||||
```
|
||||
|
||||
**CRITICAL BUG:** String constants are encoded as UTF-8 internally. If your string contains bytes > 127, the resulting array will be larger than expected:
|
||||
|
||||
```python
|
||||
# String with high bytes
|
||||
data = "Hello\xff\x80" # 7 characters
|
||||
const = llvm.const_string(ctx, data, dont_null_terminate=True)
|
||||
# const.type might be [10 x i8] instead of [7 x i8]!
|
||||
```
|
||||
|
||||
This makes string encryption impossible when encrypted bytes exceed ASCII range.
|
||||
Use `bytes` when you need byte-preserving constants. The `str` overload intentionally uses UTF-8 text encoding.
|
||||
|
||||
### Array Constants
|
||||
|
||||
@@ -640,11 +633,11 @@ Constant* dataArr = ConstantDataArray::get(ctx, arrayRef);
|
||||
**Python:**
|
||||
```python
|
||||
arr = llvm.const_array(elem_ty, [elem1, elem2])
|
||||
data_arr = llvm.const_data_array(elem_ty, string_data)
|
||||
data_arr = llvm.const_data_array(ctx.types.i8, b"\xff\x80B")
|
||||
# Equivalent type helper overload:
|
||||
data_arr = ctx.types.i8.const_data_array(b"\xff\x80B")
|
||||
```
|
||||
|
||||
**Same UTF-8 bug applies to `const_data_array`.**
|
||||
|
||||
### Struct Constants
|
||||
|
||||
**C++:**
|
||||
@@ -668,21 +661,13 @@ s = llvm.const_named_struct(struct_ty, [field1, field2])
|
||||
oldValue->replaceAllUsesWith(newValue);
|
||||
```
|
||||
|
||||
**Python:** There is NO `replace_all_uses_with()` method on `Value`. You must implement it manually:
|
||||
|
||||
**Python:**
|
||||
```python
|
||||
def replace_all_uses_with(old_value, new_value):
|
||||
"""Replace all uses of old_value with new_value."""
|
||||
uses_to_replace = []
|
||||
for use in old_value.uses:
|
||||
uses_to_replace.append(use.user)
|
||||
|
||||
for user in uses_to_replace:
|
||||
for i in range(user.num_operands):
|
||||
if user.get_operand(i) == old_value:
|
||||
user.set_operand(i, new_value)
|
||||
old_value.replace_all_uses_with(new_value)
|
||||
```
|
||||
|
||||
The replacement must be from the same context and have the exact same LLVM type.
|
||||
|
||||
### Deleting Instructions
|
||||
|
||||
**C++:**
|
||||
@@ -690,17 +675,12 @@ def replace_all_uses_with(old_value, new_value):
|
||||
inst->eraseFromParent();
|
||||
```
|
||||
|
||||
**Python:** Two-step process required:
|
||||
|
||||
**Python:**
|
||||
```python
|
||||
inst.remove_from_parent() # Unlink from block
|
||||
inst.delete_instruction() # Actually delete
|
||||
inst.erase_from_parent()
|
||||
```
|
||||
|
||||
If you call `delete_instruction()` without `remove_from_parent()` first, LLVM will assert:
|
||||
```
|
||||
Assertion failed: (!getParent() && "Instruction still linked in the program!")
|
||||
```
|
||||
Low-level `remove_from_parent()` and `delete_instruction()` still exist, but most code should use `erase_from_parent()`.
|
||||
|
||||
### Instruction Uses
|
||||
|
||||
@@ -720,33 +700,29 @@ for use in inst.uses:
|
||||
|
||||
---
|
||||
|
||||
## Missing APIs
|
||||
## Remaining API Differences
|
||||
|
||||
The following LLVM C++ APIs have no Python equivalent:
|
||||
The following are current differences from the LLVM C++ API:
|
||||
|
||||
### Basic Block Operations
|
||||
- `BasicBlock::splitBasicBlock()` - Split block at instruction
|
||||
- `BasicBlock::splitBasicBlockBefore()` - Split before instruction
|
||||
- Split helpers are available as `bb.split_basic_block(...)` and `bb.split_basic_block_before(...)`.
|
||||
|
||||
### Instruction Operations
|
||||
- `Instruction::insertBefore()` - Insert instruction before another
|
||||
- `Instruction::insertAfter()` - Insert instruction after another
|
||||
- `Instruction::moveBefore()` - Move instruction
|
||||
- `Instruction::moveAfter()` - Move instruction
|
||||
- `Instruction::clone()` - Clone instruction
|
||||
- Instruction movement is available as `inst.move_before(...)` and `inst.move_after(...)`.
|
||||
- Instruction cloning is available as `inst.instruction_clone()`.
|
||||
- Direct `insertBefore()` / `insertAfter()`-style detached insertion APIs are still limited.
|
||||
|
||||
### Value Operations
|
||||
- `Value::replaceAllUsesWith()` - Replace all uses (must implement manually)
|
||||
- `Value::replaceAllUsesWith()` is available as `value.replace_all_uses_with(...)`.
|
||||
|
||||
### Module Operations
|
||||
- `Module::materializeAll()` - Materialize lazy module
|
||||
- `Module::materializeAll()` is not currently exposed.
|
||||
|
||||
### Type Checking
|
||||
- `isa<T>()`, `dyn_cast<T>()` - No type casting, use `.opcode` or `.is_*` properties
|
||||
- `isa<T>()` / `dyn_cast<T>()` are not exposed; use `.opcode` or `.is_*` properties.
|
||||
|
||||
### Function Operations
|
||||
- `Function::viewCFG()` - View control flow graph
|
||||
- `Function::viewCFGOnly()` - View CFG without instructions
|
||||
- `Function::viewCFG()` and `Function::viewCFGOnly()` are not exposed.
|
||||
|
||||
---
|
||||
|
||||
@@ -771,9 +747,9 @@ The following LLVM C++ APIs have no Python equivalent:
|
||||
| `getTerminator()` | `.terminator` |
|
||||
| `getNumOperands()` | `.num_operands` |
|
||||
| `getOperand(i)` | `.get_operand(i)` |
|
||||
| `isTerminator()` | `.is_terminator_inst` |
|
||||
| `isTerminator()` | `.is_terminator` |
|
||||
| `getIntegerBitWidth()` | `.int_width` |
|
||||
| `eraseFromParent()` | `.remove_from_parent()` + `.delete_instruction()` |
|
||||
| `eraseFromParent()` | `.erase_from_parent()` |
|
||||
|
||||
---
|
||||
|
||||
@@ -791,48 +767,40 @@ with ctx.parse_ir(ir_text) as mod:
|
||||
mod.functions # Works
|
||||
```
|
||||
|
||||
### 2. Forgetting to Call `ptr()`
|
||||
### 2. Pointer Types Are Properties
|
||||
|
||||
```python
|
||||
# WRONG
|
||||
ptr_ty = ctx.types.ptr # Bound method!
|
||||
|
||||
# CORRECT
|
||||
ptr_ty = ctx.types.ptr() # Type object
|
||||
ptr_ty = ctx.types.ptr # default opaque pointer type
|
||||
ptr_as1 = ctx.types.addrspace_ptr(1)
|
||||
```
|
||||
|
||||
### 3. Using Wrong Property for Parent Block
|
||||
### 3. Parent Block Alias
|
||||
|
||||
```python
|
||||
# WRONG
|
||||
bb = inst.parent # AttributeError
|
||||
|
||||
# CORRECT
|
||||
bb = inst.block
|
||||
bb = inst.block # explicit instruction/basic-block relationship
|
||||
bb = inst.parent # alias, if you prefer C++ terminology
|
||||
```
|
||||
|
||||
### 4. Trying to Iterate Operands
|
||||
### 4. Operand Iteration vs Indexed Mutation
|
||||
|
||||
```python
|
||||
# WRONG
|
||||
for op in inst.operands: # No such property
|
||||
for op in inst.operands:
|
||||
...
|
||||
|
||||
# CORRECT
|
||||
# Use indexes when replacing operands.
|
||||
for i in range(inst.num_operands):
|
||||
op = inst.get_operand(i)
|
||||
if inst.get_operand(i) == old:
|
||||
inst.set_operand(i, new)
|
||||
```
|
||||
|
||||
### 5. Forgetting Two-Step Deletion
|
||||
### 5. Prefer Single-Step Deletion
|
||||
|
||||
```python
|
||||
# WRONG - will crash
|
||||
inst.delete_instruction()
|
||||
|
||||
# CORRECT
|
||||
inst.remove_from_parent()
|
||||
inst.delete_instruction()
|
||||
inst.erase_from_parent()
|
||||
```
|
||||
|
||||
Only use `remove_from_parent()` / `delete_instruction()` when you intentionally need the lower-level operations.
|
||||
|
||||
### 6. Integer Overflow in Constants
|
||||
|
||||
```python
|
||||
@@ -843,13 +811,13 @@ val = i64_ty.constant(0xFFFFFFFFFFFFFFFF) # Too large!
|
||||
val = i64_ty.constant(-1) # Same bit pattern, works
|
||||
```
|
||||
|
||||
### 7. String Constants with High Bytes
|
||||
### 7. Raw Byte Constants
|
||||
|
||||
```python
|
||||
# Encrypted string with bytes > 127
|
||||
encrypted = bytes([0xFF, 0x80, 0x42]).decode('latin-1')
|
||||
const = llvm.const_string(ctx, encrypted, True)
|
||||
# Type will be wrong due to UTF-8 encoding!
|
||||
# Encrypted bytes or other binary payloads: pass bytes, not str.
|
||||
encrypted = bytes([0xFF, 0x80, 0x42])
|
||||
const = llvm.const_string(ctx, encrypted, dont_null_terminate=True)
|
||||
assert const.type.array_length == len(encrypted)
|
||||
```
|
||||
|
||||
### 8. Exception Types
|
||||
@@ -873,16 +841,8 @@ Available exceptions: `LLVMError`, `LLVMParseError`, `LLVMAssertionError`, `LLVM
|
||||
```python
|
||||
def replace_instruction(old_inst, new_value):
|
||||
"""Safely replace an instruction with a new value."""
|
||||
# Replace all uses
|
||||
for use in list(old_inst.uses):
|
||||
user = use.user
|
||||
for i in range(user.num_operands):
|
||||
if user.get_operand(i) == old_inst:
|
||||
user.set_operand(i, new_value)
|
||||
|
||||
# Delete old instruction
|
||||
old_inst.remove_from_parent()
|
||||
old_inst.delete_instruction()
|
||||
old_inst.replace_all_uses_with(new_value)
|
||||
old_inst.erase_from_parent()
|
||||
```
|
||||
|
||||
### Safe Iteration with Modification
|
||||
@@ -1015,87 +975,35 @@ result = builder.select(condition, true_val, false_val, "name")
|
||||
|
||||
5. **Iteration**: Iterating over functions, blocks, and instructions with `for` loops works seamlessly.
|
||||
|
||||
### What Needs Improvement
|
||||
### Current Status and Remaining Gaps
|
||||
|
||||
#### Critical Issues
|
||||
The original porting effort exposed several blockers. The current bindings have since added the core transformation conveniences:
|
||||
|
||||
1. **UTF-8 String Encoding Bug**: `const_string` and `const_data_array` encode strings as UTF-8, making it impossible to work with raw byte arrays. This is a **blocking issue** for any binary manipulation use case.
|
||||
|
||||
**Recommendation**: Add `const_bytes(ctx, bytes_object)` or accept `bytes` type in addition to `str`.
|
||||
- raw `bytes` support for `const_string()` / `const_data_array()`
|
||||
- `value.replace_all_uses_with(new_value)`
|
||||
- `bb.split_basic_block(...)` and `bb.split_basic_block_before(...)`
|
||||
- `inst.erase_from_parent()`
|
||||
- `ctx.types.ptr` as a property
|
||||
- `inst.parent` and `inst.is_terminator` aliases
|
||||
- `inst.operands`
|
||||
- `inst.move_before(...)` / `inst.move_after(...)`
|
||||
- `inst.instruction_clone()`
|
||||
|
||||
2. **Missing `replaceAllUsesWith`**: This is a fundamental LLVM operation used constantly in transforms. Its absence forces users to implement it manually (error-prone).
|
||||
|
||||
**Recommendation**: Add `value.replace_all_uses_with(new_value)`.
|
||||
Remaining limitations are more specialized:
|
||||
|
||||
3. **Missing `splitBasicBlock`**: Cannot split blocks, which is essential for many transformations.
|
||||
|
||||
**Recommendation**: Add `bb.split_before(inst)` returning the new block.
|
||||
- Direct detached-instruction insertion APIs are still limited.
|
||||
- The API is intentionally flat: use `.opcode` / `.is_*` predicates instead of C++ `isa<T>()` / `dyn_cast<T>()`.
|
||||
- Context managers remain important: `ctx.parse_ir(...)` and `ctx.create_module(...)` return managers that must be entered with `with`.
|
||||
|
||||
4. **Two-Step Instruction Deletion**: Having to call `remove_from_parent()` then `delete_instruction()` is error-prone and non-obvious.
|
||||
|
||||
**Recommendation**: Add `inst.erase_from_parent()` that does both, matching C++ API.
|
||||
### Documentation References
|
||||
|
||||
#### API Inconsistencies
|
||||
|
||||
1. **`ptr()` vs `i32`**: Why is `ctx.types.ptr()` a method but `ctx.types.i32` a property? This inconsistency causes confusion.
|
||||
|
||||
**Recommendation**: Make all types properties, or all methods. Consistency matters more than either choice.
|
||||
|
||||
2. **`set_constant()` vs `linkage =`**: Global variable `is_constant` uses a setter method, but `linkage` uses property assignment. Pick one pattern.
|
||||
|
||||
**Recommendation**: Prefer property setters: `gv.is_constant = False`.
|
||||
|
||||
3. **`.block` vs `.parent`**: Instructions use `.block` to get parent, but the C++ API and intuition suggest `.parent`.
|
||||
|
||||
**Recommendation**: Add `.parent` as an alias for `.block`.
|
||||
|
||||
4. **`.is_terminator_inst` suffix**: The `_inst` suffix is inconsistent with other boolean properties.
|
||||
|
||||
**Recommendation**: Use `.is_terminator` instead.
|
||||
|
||||
#### Missing Conveniences
|
||||
|
||||
1. **No `.operands` iterator**: Having to use index-based access is verbose:
|
||||
```python
|
||||
# Current (verbose)
|
||||
for i in range(inst.num_operands):
|
||||
op = inst.get_operand(i)
|
||||
|
||||
# Desired
|
||||
for op in inst.operands:
|
||||
```
|
||||
|
||||
**Recommendation**: Add `operands` property returning an iterator.
|
||||
|
||||
2. **No instruction movement**: Can't move instructions between blocks or reorder them.
|
||||
|
||||
**Recommendation**: Add `inst.move_before(other)`, `inst.move_after(other)`.
|
||||
|
||||
3. **No instruction cloning**: Can't clone an instruction.
|
||||
|
||||
**Recommendation**: Add `inst.clone()`.
|
||||
|
||||
### Documentation Gaps
|
||||
|
||||
1. **No comprehensive API reference**: Had to discover APIs through `dir()` and trial-and-error.
|
||||
|
||||
2. **Exception types not documented**: Had to guess that `LLVMError` exists, not `LLVMException`.
|
||||
|
||||
3. **Context manager behavior not obvious**: The `ModuleManager` pattern needs explicit documentation.
|
||||
- The generated stub `.venv/Lib/site-packages/llvm/__init__.pyi` is the exact API reference for the currently built bindings.
|
||||
- `devdocs/api-reference.md` documents validity/precondition rules.
|
||||
- `README.md` and `examples/` contain runnable examples.
|
||||
|
||||
### Overall Assessment
|
||||
|
||||
**Rating: 7/10 for basic use, 5/10 for advanced transforms**
|
||||
|
||||
The bindings are **good for code generation** (creating new IR from scratch) where the Builder API shines. They're **challenging for transforms** (modifying existing IR) due to missing fundamental operations like RAUW, block splitting, and instruction movement.
|
||||
|
||||
The API is **Pythonic in places** (properties, context managers, iteration) but **inconsistent in others** (method vs property for similar operations). The UTF-8 string bug is a **critical issue** that blocks an entire category of use cases.
|
||||
|
||||
For someone porting C++ LLVM code, expect to:
|
||||
- Spend significant time discovering API names through trial-and-error
|
||||
- Implement workarounds for missing functionality
|
||||
- Hit crashes that require debugging LLVM assertions
|
||||
- Accept that some transformations simply aren't possible
|
||||
The bindings are suitable for code generation and many transformation tasks. When porting C++ LLVM code, expect to translate C++ class-specific APIs into the binding's flat wrapper model, but fundamental SSA and basic-block operations are now available.
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@ Visual reference materials for understanding llvm-nanobind.
|
||||
│ │ Python Classes │ Lifetime Management │ Type Helpers │ │
|
||||
│ │ ───────────────── │ ──────────────────── │ ──────────── │ │
|
||||
│ │ Module │ Validity tokens │ ctx.types.i32 │ │
|
||||
│ │ Function │ Context managers │ ctx.types.ptr() │ │
|
||||
│ │ Function │ Context managers │ ctx.types.ptr │ │
|
||||
│ │ BasicBlock │ Ref counting │ ctx.types.array │ │
|
||||
│ │ Value/Instruction │ │ │ │
|
||||
│ │ Builder │ │ │ │
|
||||
@@ -40,9 +40,6 @@ Visual reference materials for understanding llvm-nanobind.
|
||||
│ LLVMContextCreate() LLVMBuildAdd() LLVMGetBasicBlocks() │
|
||||
│ LLVMParseIRInContext() LLVMBuildBr() LLVMGetInstructions() │
|
||||
│ LLVMCreateBuilder() LLVMBuildRet() LLVMReplaceAllUsesWith() │
|
||||
│ ▲ │
|
||||
│ │ │
|
||||
│ NOT YET BOUND! │
|
||||
├─────────────────────────────────────────────────────────────────────────────┤
|
||||
│ LLVM C++ Core │
|
||||
│ │
|
||||
@@ -124,21 +121,14 @@ KEY INVARIANTS:
|
||||
┌────────────────────────────────┐
|
||||
│ 4. REPLACE │
|
||||
│ │
|
||||
│ for use in old.uses: │
|
||||
│ user.set_operand(i, new) │
|
||||
│ │
|
||||
│ [This is manual RAUW!] │
|
||||
│ old.replace_all_uses_with(new)│
|
||||
└───────────────┬────────────────┘
|
||||
│
|
||||
▼
|
||||
┌────────────────────────────────┐
|
||||
│ 5. CLEAN UP │
|
||||
│ │
|
||||
│ old.remove_from_parent() │
|
||||
│ old.delete_instruction() │
|
||||
│ │
|
||||
│ [Two-step dance!] │
|
||||
│ │
|
||||
│ old.erase_from_parent() │
|
||||
│ assert mod.verify() │
|
||||
└───────────────┬────────────────┘
|
||||
│
|
||||
@@ -335,14 +325,14 @@ PRIORITY 2: API CONSISTENCY
|
||||
│ ptr() vs ptr ───────────── Inconsistent access pattern │
|
||||
│ set_constant() vs = ────── Mixed setter styles │
|
||||
│ .block vs .parent ──────── Confusing naming │
|
||||
│ .is_terminator_inst ────── Unnecessary suffix │
|
||||
│ .is_terminator ─────────── Terminator predicate │
|
||||
│ │
|
||||
────────────────────────────────────────────────────────────
|
||||
|
||||
PRIORITY 3: CONVENIENCES
|
||||
────────────────────────────────────────────────────────────
|
||||
│ │
|
||||
│ .operands iterator ─────── Pythonic iteration │
|
||||
│ .operands ──────────────── Pythonic operand iteration │
|
||||
│ move_before/after ──────── Instruction movement │
|
||||
│ .clone() ───────────────── Instruction copying │
|
||||
│ .num_successors ────────── Direct count access │
|
||||
|
||||
@@ -32,8 +32,8 @@ i8 = ctx.types.i8 # byte
|
||||
i32 = ctx.types.i32 # int
|
||||
i64 = ctx.types.i64 # long
|
||||
|
||||
# Pointer - NOTE: must call!
|
||||
ptr = ctx.types.ptr() # NOT ctx.types.ptr
|
||||
# Pointer
|
||||
ptr = ctx.types.ptr
|
||||
|
||||
# Composite types
|
||||
arr = ctx.types.array(elem_ty, count)
|
||||
@@ -81,24 +81,24 @@ bb.move_after(other_bb)
|
||||
```python
|
||||
# Properties
|
||||
inst.opcode == llvm.Opcode.Add
|
||||
inst.is_terminator_inst
|
||||
inst.is_terminator
|
||||
inst.name
|
||||
inst.type
|
||||
bb = inst.block # NOT .parent
|
||||
bb = inst.block # explicit parent block
|
||||
bb = inst.parent # alias
|
||||
|
||||
# Operands (NO .operands iterator!)
|
||||
for i in range(inst.num_operands):
|
||||
op = inst.get_operand(i)
|
||||
inst.set_operand(i, new_val)
|
||||
# Operands
|
||||
for op in inst.operands:
|
||||
...
|
||||
inst.set_operand(i, new_val) # indexed mutation
|
||||
|
||||
# Uses
|
||||
for use in inst.uses:
|
||||
user = use.user
|
||||
# user is an instruction that uses inst
|
||||
|
||||
# Deletion (TWO steps!)
|
||||
inst.remove_from_parent()
|
||||
inst.delete_instruction()
|
||||
# Deletion
|
||||
inst.erase_from_parent()
|
||||
```
|
||||
|
||||
## Builder
|
||||
@@ -150,8 +150,9 @@ phi.add_incoming(val, from_bb)
|
||||
c = i32.constant(42)
|
||||
c = i32.constant(-1) # signed OK
|
||||
|
||||
# Strings (CAUTION: UTF-8 encoding!)
|
||||
# Strings and raw bytes
|
||||
c = llvm.const_string(ctx, "text", dont_null_terminate=False)
|
||||
raw = llvm.const_string(ctx, b"\xff\x80B", dont_null_terminate=True)
|
||||
|
||||
# Null pointer
|
||||
null = llvm.ConstantPointerNull.get(ptr_ty)
|
||||
@@ -181,7 +182,7 @@ for i in range(phi.num_incoming):
|
||||
gv = mod.add_global(ty, "name")
|
||||
gv.initializer = const_val
|
||||
gv.linkage = llvm.Linkage.Private
|
||||
gv.set_constant(True) # method, not property!
|
||||
gv.is_global_constant = True
|
||||
|
||||
# Iterate
|
||||
for gv in mod.globals:
|
||||
@@ -193,17 +194,10 @@ for gv in mod.globals:
|
||||
|
||||
## Common Patterns
|
||||
|
||||
### Replace All Uses With (MANUAL!)
|
||||
### Replace All Uses With
|
||||
|
||||
```python
|
||||
def replace_all_uses_with(old_val, new_val):
|
||||
"""Replace all uses of old_val with new_val."""
|
||||
uses = list(old_val.uses) # snapshot
|
||||
for use in uses:
|
||||
user = use.user
|
||||
for i in range(user.num_operands):
|
||||
if user.get_operand(i) == old_val:
|
||||
user.set_operand(i, new_val)
|
||||
old_val.replace_all_uses_with(new_val)
|
||||
```
|
||||
|
||||
### Safe Instruction Replacement
|
||||
@@ -211,9 +205,8 @@ def replace_all_uses_with(old_val, new_val):
|
||||
```python
|
||||
def replace_instruction(old_inst, new_val):
|
||||
"""Replace instruction with new value and delete."""
|
||||
replace_all_uses_with(old_inst, new_val)
|
||||
old_inst.remove_from_parent()
|
||||
old_inst.delete_instruction()
|
||||
old_inst.replace_all_uses_with(new_val)
|
||||
old_inst.erase_from_parent()
|
||||
```
|
||||
|
||||
### Collect Then Modify
|
||||
@@ -250,13 +243,10 @@ subs = find_by_opcode(func, llvm.Opcode.Sub)
|
||||
|
||||
| What you try | What happens | Fix |
|
||||
|-------------|--------------|-----|
|
||||
| `ctx.types.ptr` | Gets bound method, not type | `ctx.types.ptr()` |
|
||||
| `inst.parent` | AttributeError | `inst.block` |
|
||||
| `for op in inst.operands:` | AttributeError | Manual iteration |
|
||||
| `inst.delete_instruction()` | LLVM assertion | Remove first |
|
||||
| `mod = ctx.parse_ir(...)` | Gets manager, not module | Use `with ... as mod` |
|
||||
| Bytes > 127 in strings | UTF-8 expansion | **Blocked** - no fix |
|
||||
| `inst.is_terminator` | AttributeError | `inst.is_terminator_inst` |
|
||||
| `bb.terminator` on unterminated block | Raises `LLVMAssertionError` | Check `bb.has_terminator` first |
|
||||
| `inst.set_operand(i, value_of_wrong_type)` | Raises `LLVMAssertionError` | Use a value with the exact expected type |
|
||||
| `llvm.const_string(ctx, "\xff")` for raw bytes | Uses text/UTF-8 path | Pass `bytes`: `b"\xff"` |
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -249,16 +249,9 @@ def exercise_3_simple_transform():
|
||||
doubled_const = builder.add(op1, op1, "doubled")
|
||||
new_add = builder.add(op0, doubled_const, inst.name + ".new")
|
||||
|
||||
# Replace uses
|
||||
for use in list(inst.uses):
|
||||
user = use.user
|
||||
for i in range(user.num_operands):
|
||||
if user.get_operand(i) == inst:
|
||||
user.set_operand(i, new_add)
|
||||
|
||||
# Remove old instruction
|
||||
inst.remove_from_parent()
|
||||
inst.delete_instruction()
|
||||
# Replace uses and remove the old instruction
|
||||
inst.replace_all_uses_with(new_add)
|
||||
inst.erase_from_parent()
|
||||
|
||||
show_ir("After transformation", mod.to_string())
|
||||
|
||||
@@ -270,15 +263,11 @@ def exercise_3_simple_transform():
|
||||
3. REPLACED: Uses of old instruction with new one
|
||||
4. CLEANED UP: Removed old instruction
|
||||
|
||||
Notice the manual replace_all_uses_with pattern:
|
||||
for use in list(inst.uses):
|
||||
user = use.user
|
||||
for i in range(user.num_operands):
|
||||
if user.get_operand(i) == inst:
|
||||
user.set_operand(i, new_add)
|
||||
The current bindings expose the common LLVM RAUW pattern directly:
|
||||
inst.replace_all_uses_with(new_add)
|
||||
inst.erase_from_parent()
|
||||
|
||||
This is verbose because the binding doesn't expose RAUW directly.
|
||||
That's one of the planned improvements!
|
||||
This keeps the transform focused on the IR rewrite, not bookkeeping.
|
||||
""")
|
||||
|
||||
pause("Press Enter for the next exercise...")
|
||||
@@ -372,15 +361,9 @@ def exercise_4_mba_by_hand():
|
||||
result = builder.add(xor_val, mul_val, "mba.result")
|
||||
print(f" Step 5: %mba.result = add %mba.xor, %mba.mul")
|
||||
|
||||
# Replace uses
|
||||
for use in list(inst.uses):
|
||||
user = use.user
|
||||
for i in range(user.num_operands):
|
||||
if user.get_operand(i) == inst:
|
||||
user.set_operand(i, result)
|
||||
|
||||
inst.remove_from_parent()
|
||||
inst.delete_instruction()
|
||||
# Replace uses and remove the old subtraction.
|
||||
inst.replace_all_uses_with(result)
|
||||
inst.erase_from_parent()
|
||||
|
||||
print()
|
||||
show_ir("After MBA", mod.to_string())
|
||||
@@ -558,11 +541,11 @@ def summary():
|
||||
- Replace: substitute old with new (RAUW)
|
||||
- Clean up: delete old, verify
|
||||
|
||||
3. {CYAN}API Quirks{RESET}
|
||||
3. {CYAN}API Patterns{RESET}
|
||||
- Context managers for modules and builders
|
||||
- No direct RAUW (must implement manually)
|
||||
- Two-step instruction deletion
|
||||
- Inconsistent property vs method patterns
|
||||
- `replace_all_uses_with` for SSA rewrites
|
||||
- `erase_from_parent` for instruction deletion
|
||||
- Properties such as `inst.operands` and `ctx.types.ptr`
|
||||
|
||||
4. {CYAN}Obfuscation Techniques{RESET}
|
||||
- MBA: hide operations in equivalent bitwise soup
|
||||
|
||||
@@ -132,7 +132,7 @@ BINDING_API = [
|
||||
{
|
||||
"question": "How do you get the opaque pointer type for address space 0 in the bindings?",
|
||||
"options": [
|
||||
"ctx.types.ptr()",
|
||||
"ctx.types.pointer",
|
||||
"ctx.types.ptr",
|
||||
"ctx.types.pointer(0)",
|
||||
"ctx.types.addrspace_ptr(0)",
|
||||
@@ -150,11 +150,10 @@ BINDING_API = [
|
||||
"Call inst.remove_from_parent() then inst.delete_instruction()",
|
||||
"Set inst to None",
|
||||
],
|
||||
"correct": 3,
|
||||
"explanation": "The two-step process is error-prone! If you call delete_instruction() "
|
||||
"while the instruction is still in a block, LLVM will assert/crash. "
|
||||
"The plan proposes adding erase_from_parent() that does both atomically, "
|
||||
"matching the C++ API.",
|
||||
"correct": 2,
|
||||
"explanation": "Use inst.erase_from_parent(), which matches the common C++ eraseFromParent() pattern. "
|
||||
"Lower-level remove_from_parent() and delete_instruction() still exist, "
|
||||
"but most code should not need the two-step sequence.",
|
||||
},
|
||||
{
|
||||
"question": "What does replace_all_uses_with() do?",
|
||||
@@ -166,9 +165,8 @@ BINDING_API = [
|
||||
],
|
||||
"correct": 2,
|
||||
"explanation": "RAUW is fundamental to SSA-based transformations. When you create a new "
|
||||
"value that should replace an old one, you use RAUW to update every "
|
||||
"instruction that uses the old value. This is NOT currently bound - "
|
||||
"the passes implement it manually, which is error-prone and slow.",
|
||||
"value that should replace an old one, you use replace_all_uses_with() "
|
||||
"to update every instruction that uses the old value.",
|
||||
},
|
||||
{
|
||||
"question": "What happens if you access a Module after exiting its 'with' block?",
|
||||
@@ -192,10 +190,9 @@ BINDING_API = [
|
||||
"for i in range(inst.num_operands): op = inst.get_operand(i)",
|
||||
"for op in inst:",
|
||||
],
|
||||
"correct": 3,
|
||||
"explanation": "There's no .operands iterator! This is listed as a missing convenience. "
|
||||
"You must use index-based access. The plan proposes adding an operands "
|
||||
"property that returns an iterator for Pythonic access.",
|
||||
"correct": 1,
|
||||
"explanation": "inst.operands provides Pythonic iteration over an operand snapshot. "
|
||||
"Use index-based get_operand()/set_operand() when you need to replace operands.",
|
||||
},
|
||||
]
|
||||
|
||||
@@ -243,18 +240,16 @@ OBFUSCATION = [
|
||||
"we preserve the semantics without relying on control flow.",
|
||||
},
|
||||
{
|
||||
"question": "The string encryption pass was abandoned because:",
|
||||
"question": "When creating encrypted string data, which input type preserves raw bytes?",
|
||||
"options": [
|
||||
"Strings can't be encrypted",
|
||||
"The bindings encode strings as UTF-8, corrupting bytes > 127",
|
||||
"LLVM doesn't support string constants",
|
||||
"It was too slow",
|
||||
"str decoded with latin-1",
|
||||
"bytes passed to const_string() or const_data_array()",
|
||||
"A Python list of characters",
|
||||
"LLVM does not support byte constants",
|
||||
],
|
||||
"correct": 2,
|
||||
"explanation": "const_string() and const_data_array() pass strings through UTF-8 encoding. "
|
||||
"Encrypted bytes often exceed 127, which expand to multi-byte UTF-8 "
|
||||
"sequences. The resulting array is larger than expected, breaking the "
|
||||
"decryption logic. This is listed as a critical blocker in the plan.",
|
||||
"explanation": "Use bytes for binary payloads: llvm.const_string(ctx, data, ...) or "
|
||||
"llvm.const_data_array(i8, data). The str overload is for text and follows UTF-8 encoding.",
|
||||
},
|
||||
]
|
||||
|
||||
@@ -263,14 +258,14 @@ CRITIQUE = [
|
||||
"question": "Which improvement has the HIGHEST priority according to the plan?",
|
||||
"options": [
|
||||
"Adding documentation for exceptions",
|
||||
"Making ptr a property instead of method",
|
||||
"Documenting that ptr is a property",
|
||||
"Binding LLVMReplaceAllUsesWith",
|
||||
"Adding an .operands iterator",
|
||||
],
|
||||
"correct": 3,
|
||||
"explanation": "The plan categorizes issues by priority. Priority 1 (Critical Blockers) "
|
||||
"includes RAUW, erase_from_parent, split_basic_block, and raw bytes support. "
|
||||
"These block real use cases. API consistency issues (like ptr()) are P2, "
|
||||
"These block real use cases. API consistency issues were P2, "
|
||||
"conveniences are P3, documentation is P4.",
|
||||
},
|
||||
{
|
||||
|
||||
@@ -187,9 +187,8 @@ for bb in func.basic_blocks:
|
||||
result = builder.add(xor_val, mul_val)
|
||||
|
||||
# 4. REPLACE: Swap old for new
|
||||
replace_all_uses_with(inst, result)
|
||||
inst.remove_from_parent()
|
||||
inst.delete_instruction()
|
||||
inst.replace_all_uses_with(result)
|
||||
inst.erase_from_parent()
|
||||
|
||||
# 5. CLEAN UP: Verify
|
||||
assert mod.verify()
|
||||
@@ -296,7 +295,7 @@ func.name # not get_name()
|
||||
|
||||
# Methods for operations with side effects:
|
||||
builder.add(a, b) # creates instruction
|
||||
inst.remove_from_parent() # modifies IR
|
||||
inst.erase_from_parent() # modifies IR
|
||||
```
|
||||
|
||||
**Context Managers:**
|
||||
@@ -319,34 +318,38 @@ for func in mod.functions:
|
||||
...
|
||||
```
|
||||
|
||||
**The Inconsistencies (and why they matter):**
|
||||
**Consistency improvements:**
|
||||
|
||||
The porting guide documents several inconsistencies:
|
||||
Several early inconsistencies have been cleaned up. Current code can use:
|
||||
|
||||
| Inconsistent | Expected |
|
||||
|-------------|----------|
|
||||
| `ctx.types.ptr()` (method) | `ctx.types.ptr` (property like `i32`) |
|
||||
| `gv.set_constant(False)` | `gv.is_constant = False` |
|
||||
| `inst.block` | `inst.parent` (matches C++) |
|
||||
| `inst.is_terminator_inst` | `inst.is_terminator` |
|
||||
```python
|
||||
ptr_ty = ctx.types.ptr
|
||||
for op in inst.operands:
|
||||
...
|
||||
inst.is_terminator
|
||||
inst.parent # alias for inst.block
|
||||
gv.is_global_constant = True
|
||||
```
|
||||
|
||||
These aren't just aesthetic. Inconsistencies create cognitive load. Every time you use the API, you have to remember which pattern applies.
|
||||
The porting guide still documents remaining differences from the C++ API, but the common transformation path now uses Pythonic properties and single-step helpers.
|
||||
|
||||
**Exercise 2.3:**
|
||||
You want to iterate over all operands of an instruction. Based on the consistency issues, predict: is there an `.operands` iterator?
|
||||
You want to replace one operand of an instruction. Should you use `inst.operands` or indexed access?
|
||||
|
||||
<details>
|
||||
<summary>Answer</summary>
|
||||
|
||||
No! You must use:
|
||||
Use `inst.operands` for read-only iteration:
|
||||
```python
|
||||
for i in range(inst.num_operands):
|
||||
op = inst.get_operand(i)
|
||||
for op in inst.operands:
|
||||
print(op)
|
||||
```
|
||||
|
||||
This is listed as a missing convenience. It should be:
|
||||
Use indexed access for mutation:
|
||||
```python
|
||||
for op in inst.operands: # Proposed improvement
|
||||
for i in range(inst.num_operands):
|
||||
if inst.get_operand(i) == old:
|
||||
inst.set_operand(i, new)
|
||||
```
|
||||
</details>
|
||||
|
||||
@@ -516,32 +519,22 @@ This is defense in depth: the state values are already random, but their *check
|
||||
|
||||
The porting process revealed API gaps:
|
||||
|
||||
| Gap | Why It Matters |
|
||||
| Previously Missing API | Current Python API |
|
||||
|-----|----------------|
|
||||
| No `replace_all_uses_with` | Core SSA operation—must implement manually |
|
||||
| No `erase_from_parent` | Error-prone two-step deletion |
|
||||
| No `split_basic_block` | Can't implement some transforms (blocked) |
|
||||
| UTF-8 encoding bug | Can't handle encrypted bytes > 127 (blocked) |
|
||||
| Replace all uses | `value.replace_all_uses_with(new_value)` |
|
||||
| Delete instruction | `inst.erase_from_parent()` |
|
||||
| Split block | `bb.split_basic_block(...)` / `bb.split_basic_block_before(...)` |
|
||||
| Raw byte constants | `llvm.const_string(ctx, b"...", ...)` and `llvm.const_data_array(i8, b"...")` |
|
||||
|
||||
**The UTF-8 Bug in Detail:**
|
||||
**Raw bytes example:**
|
||||
|
||||
```python
|
||||
# You want to create a constant with bytes [0xFF, 0x80, 0x42]
|
||||
encrypted = bytes([0xFF, 0x80, 0x42]).decode('latin-1') # "\xff\x80B"
|
||||
encrypted = bytes([0xFF, 0x80, 0x42])
|
||||
const = llvm.const_string(ctx, encrypted, dont_null_terminate=True)
|
||||
|
||||
# Expected: [3 x i8] c"\xff\x80B"
|
||||
# Actual: [5 x i8] c"\xc3\xbf\xc2\x80B" (UTF-8 encoded!)
|
||||
assert const.type.array_length == len(encrypted)
|
||||
```
|
||||
|
||||
The bindings encode strings as UTF-8 before passing to LLVM. Bytes > 127 expand to multi-byte sequences.
|
||||
|
||||
This blocks string encryption passes because:
|
||||
1. Encrypt a string → get arbitrary bytes
|
||||
2. Store in LLVM → UTF-8 encodes, changes length
|
||||
3. Decryption code reads wrong length → crash
|
||||
|
||||
**Fix needed:** Accept `bytes` type directly, pass raw bytes to LLVM.
|
||||
Use the `bytes` overload for binary payloads. The `str` overload is for text and follows UTF-8 encoding.
|
||||
|
||||
---
|
||||
|
||||
@@ -555,12 +548,12 @@ Now you understand the work. Let's develop your ability to *critique* it.
|
||||
|
||||
An API should have predictable patterns. Test: Can you predict the name of something you haven't used?
|
||||
|
||||
| Test | Prediction | Actual | Consistent? |
|
||||
|------|------------|--------|-------------|
|
||||
| Get parent block of instruction | `.parent` | `.block` | No |
|
||||
| Get all operands | `.operands` | (doesn't exist) | No |
|
||||
| Is it a terminator? | `.is_terminator` | `.is_terminator_inst` | No |
|
||||
| Get pointer type | `.ptr` | `.ptr()` | No |
|
||||
| Test | Current API |
|
||||
|------|-------------|
|
||||
| Get parent block of instruction | `.block` or `.parent` |
|
||||
| Get all operands | `.operands` |
|
||||
| Is it a terminator? | `.is_terminator` |
|
||||
| Get pointer type | `ctx.types.ptr` |
|
||||
|
||||
**Your Task:**
|
||||
Add rows to this table. What else would you try to predict? Check the porting guide—does it exist, and does it match your expectation?
|
||||
@@ -571,9 +564,9 @@ A good API makes it hard to do the wrong thing. The "pit of success" means the e
|
||||
|
||||
| Operation | Current API | Pit of Success? |
|
||||
|-----------|-------------|-----------------|
|
||||
| Delete instruction | `inst.remove_from_parent(); inst.delete_instruction()` | No—forgetting step 1 crashes |
|
||||
| Parse module | `ctx.parse_ir()` returns context manager, not module | Partial—must use `with`, but error is confusing |
|
||||
| Create pointer type | `ctx.types.ptr()` | No—forgetting `()` silently fails later |
|
||||
| Delete instruction | `inst.erase_from_parent()` | Yes |
|
||||
| Parse module | `ctx.parse_ir()` returns context manager, not module | Partial—must use `with` |
|
||||
| Create pointer type | `ctx.types.ptr` | Yes |
|
||||
|
||||
**Your Task:**
|
||||
For each failure, design a better API. What would make the easy path correct?
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
# Transformation API Improvements
|
||||
|
||||
## Status
|
||||
|
||||
Complete. This file is kept as the original improvement plan, updated with the API names that shipped. See `progress.md` for completion status and `devdocs/porting-guide.md` for current usage examples.
|
||||
|
||||
## Overview
|
||||
|
||||
This task tracks improvements to the Python bindings to better support IR transformation use cases (as opposed to just code generation). These issues were discovered while porting LLVM obfuscator passes to Python.
|
||||
|
||||
See `devdocs/porting-guide.md` for the full context and detailed examples.
|
||||
This task tracked improvements to the Python bindings to better support IR transformation use cases (as opposed to just code generation). These issues were discovered while porting LLVM obfuscator passes to Python.
|
||||
|
||||
## Priority 1: Critical Blockers
|
||||
|
||||
@@ -12,21 +14,17 @@ These issues completely block certain use cases.
|
||||
|
||||
### 1.1 Raw Bytes Support for Constants
|
||||
|
||||
**Problem**: `const_string` and `const_data_array` encode strings as UTF-8, causing byte sequences with values > 127 to expand. This makes binary manipulation impossible.
|
||||
**Problem**: raw binary payloads must not go through text/UTF-8 encoding.
|
||||
|
||||
**Shipped solution**: `const_string` and `const_data_array` accept `bytes` and pass raw bytes to LLVM without encoding.
|
||||
|
||||
```python
|
||||
# This creates [22 x i8] instead of [14 x i8]!
|
||||
encrypted = bytes([0xFF, 0x80, 0x42, ...]).decode('latin-1')
|
||||
const = llvm.const_string(ctx, encrypted, dont_null_terminate=True)
|
||||
```
|
||||
raw = bytes([0xFF, 0x80, 0x42])
|
||||
const = llvm.const_string(ctx, raw, dont_null_terminate=True)
|
||||
assert const.type.array_length == len(raw)
|
||||
|
||||
**Solution**: Accept `bytes` type in addition to `str`, passing raw bytes to LLVM without encoding.
|
||||
|
||||
```python
|
||||
# Proposed API
|
||||
const = llvm.const_bytes(ctx, b'\xff\x80\x42...')
|
||||
# or
|
||||
const = llvm.const_data_array(i8, b'\xff\x80\x42...') # Accept bytes
|
||||
arr = llvm.const_data_array(i8, raw)
|
||||
assert arr.type.array_length == len(raw)
|
||||
```
|
||||
|
||||
**Files to modify**: Bindings for `LLVMConstStringInContext2`, `LLVMConstArray`
|
||||
@@ -37,22 +35,11 @@ const = llvm.const_data_array(i8, b'\xff\x80\x42...') # Accept bytes
|
||||
|
||||
### 1.2 Add `Value.replace_all_uses_with()`
|
||||
|
||||
**Problem**: No way to replace all uses of a value. Users must implement manually:
|
||||
**Problem**: replacing all uses of a value is fundamental to SSA transforms.
|
||||
|
||||
**Shipped solution**: bind `LLVMReplaceAllUsesWith`.
|
||||
|
||||
```python
|
||||
# Current workaround (error-prone)
|
||||
def replace_all_uses_with(old_value, new_value):
|
||||
for use in list(old_value.uses):
|
||||
user = use.user
|
||||
for i in range(user.num_operands):
|
||||
if user.get_operand(i) == old_value:
|
||||
user.set_operand(i, new_value)
|
||||
```
|
||||
|
||||
**Solution**: Bind `LLVMReplaceAllUsesWith`.
|
||||
|
||||
```python
|
||||
# Proposed API
|
||||
old_inst.replace_all_uses_with(new_value)
|
||||
```
|
||||
|
||||
@@ -62,15 +49,15 @@ old_inst.replace_all_uses_with(new_value)
|
||||
|
||||
---
|
||||
|
||||
### 1.3 Add `BasicBlock.split_before()`
|
||||
### 1.3 Add Basic Block Splitting
|
||||
|
||||
**Problem**: Cannot split a basic block at an instruction point.
|
||||
**Problem**: transformations often need to split a basic block at an instruction point.
|
||||
|
||||
**Solution**: Bind `LLVMSplitBasicBlock` or implement via C++ API.
|
||||
**Shipped solution**:
|
||||
|
||||
```python
|
||||
# Proposed API
|
||||
new_bb = bb.split_before(inst) # Returns new block containing inst onwards
|
||||
new_bb = bb.split_basic_block(inst, "tail")
|
||||
new_bb = bb.split_basic_block_before(inst, "tail")
|
||||
```
|
||||
|
||||
**C API**: Not directly available - may need custom C wrapper or use `LLVMInsertBasicBlock` + instruction movement.
|
||||
@@ -81,22 +68,12 @@ new_bb = bb.split_before(inst) # Returns new block containing inst onwards
|
||||
|
||||
### 1.4 Single-Step Instruction Deletion
|
||||
|
||||
**Problem**: Deleting an instruction requires two calls, and doing it wrong crashes:
|
||||
**Problem**: deleting an instruction should be a single safe operation.
|
||||
|
||||
**Shipped solution**:
|
||||
|
||||
```python
|
||||
# Current (crashes if order wrong)
|
||||
inst.remove_from_parent()
|
||||
inst.delete_instruction()
|
||||
|
||||
# If you forget remove_from_parent():
|
||||
# Assertion failed: (!getParent() && "Instruction still linked in the program!")
|
||||
```
|
||||
|
||||
**Solution**: Add `erase_from_parent()` that does both atomically.
|
||||
|
||||
```python
|
||||
# Proposed API
|
||||
inst.erase_from_parent() # Unlinks and deletes in one call
|
||||
inst.erase_from_parent()
|
||||
```
|
||||
|
||||
**C API**: `LLVMInstructionEraseFromParent` (already exists!)
|
||||
@@ -111,20 +88,14 @@ These issues cause confusion and bugs but have workarounds.
|
||||
|
||||
### 2.1 Make `ptr` a Property Like Other Types
|
||||
|
||||
**Problem**: `ctx.types.ptr()` is a method, but `ctx.types.i32` is a property.
|
||||
**Problem**: the default opaque pointer type should be as simple as integer types.
|
||||
|
||||
**Shipped solution**:
|
||||
|
||||
```python
|
||||
ptr_ty = ctx.types.ptr # WRONG - returns bound method!
|
||||
ptr_ty = ctx.types.ptr() # Correct
|
||||
|
||||
i32_ty = ctx.types.i32 # Correct - returns type directly
|
||||
```
|
||||
|
||||
**Solution**: Make `ptr` a property that returns the opaque pointer type.
|
||||
|
||||
```python
|
||||
# Proposed API
|
||||
ptr_ty = ctx.types.ptr # Property, like i32
|
||||
ptr_ty = ctx.types.ptr
|
||||
ptr_as1 = ctx.types.addrspace_ptr(1)
|
||||
i32_ty = ctx.types.i32
|
||||
```
|
||||
|
||||
**Note**: If `ptr(addrspace)` is needed for address spaces, keep method but add property for default.
|
||||
@@ -136,16 +107,15 @@ ptr_ty = ctx.types.ptr # Property, like i32
|
||||
**Problem**: Mixed patterns for setting properties on globals:
|
||||
|
||||
```python
|
||||
gv.set_constant(False) # Method
|
||||
gv.linkage = llvm.Linkage.X # Property setter
|
||||
gv.is_global_constant = False
|
||||
gv.linkage = llvm.Linkage.Internal
|
||||
```
|
||||
|
||||
**Solution**: Use property setters consistently.
|
||||
|
||||
```python
|
||||
# Proposed API
|
||||
gv.is_constant = False
|
||||
gv.linkage = llvm.Linkage.X
|
||||
gv.is_global_constant = False
|
||||
gv.linkage = llvm.Linkage.Internal
|
||||
```
|
||||
|
||||
---
|
||||
@@ -155,8 +125,8 @@ gv.linkage = llvm.Linkage.X
|
||||
**Problem**: Instructions use `.block` to get parent, but `.parent` is more intuitive and matches C++.
|
||||
|
||||
```python
|
||||
bb = inst.block # Current
|
||||
bb = inst.parent # Expected (doesn't exist)
|
||||
bb = inst.block
|
||||
bb = inst.parent # alias
|
||||
```
|
||||
|
||||
**Solution**: Add `.parent` as an alias for `.block`.
|
||||
@@ -168,8 +138,7 @@ bb = inst.parent # Expected (doesn't exist)
|
||||
**Problem**: The `_inst` suffix is inconsistent with other boolean properties.
|
||||
|
||||
```python
|
||||
inst.is_terminator_inst # Current
|
||||
inst.is_terminator # Expected
|
||||
inst.is_terminator
|
||||
```
|
||||
|
||||
**Solution**: Add `.is_terminator` (keep old name for compatibility or deprecate).
|
||||
@@ -185,11 +154,6 @@ These would make the API more Pythonic and reduce boilerplate.
|
||||
**Problem**: Must use index-based access to iterate operands.
|
||||
|
||||
```python
|
||||
# Current (verbose)
|
||||
for i in range(inst.num_operands):
|
||||
op = inst.get_operand(i)
|
||||
|
||||
# Desired
|
||||
for op in inst.operands:
|
||||
```
|
||||
|
||||
@@ -229,8 +193,7 @@ inst.move_after(other_inst)
|
||||
**Solution**: Bind instruction cloning.
|
||||
|
||||
```python
|
||||
# Proposed API
|
||||
new_inst = inst.clone()
|
||||
new_inst = inst.instruction_clone()
|
||||
```
|
||||
|
||||
**C API**: `LLVMInstructionClone`
|
||||
@@ -242,10 +205,6 @@ new_inst = inst.clone()
|
||||
**Problem**: Must convert to list to count successors.
|
||||
|
||||
```python
|
||||
# Current
|
||||
num = len(list(term.successors))
|
||||
|
||||
# Desired
|
||||
num = term.num_successors
|
||||
```
|
||||
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
"""Build a tiny LLVM module with llvm-nanobind.
|
||||
|
||||
Run from the repository root with:
|
||||
|
||||
uv run python examples/quick_start.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import llvm
|
||||
|
||||
|
||||
def build_module() -> str:
|
||||
"""Return LLVM IR for a function that returns 42."""
|
||||
with llvm.create_context() as ctx:
|
||||
i32 = ctx.types.i32
|
||||
fn_type = ctx.types.function(i32, [])
|
||||
|
||||
with ctx.create_module("example") as mod:
|
||||
fn = mod.add_function("get_answer", fn_type)
|
||||
entry = fn.append_basic_block("entry")
|
||||
|
||||
with entry.create_builder() as builder:
|
||||
builder.ret(i32.constant(42))
|
||||
|
||||
assert mod.verify(), mod.get_verification_error()
|
||||
return str(mod)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
print(build_module(), end="")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,60 @@
|
||||
"""Parse LLVM IR and replace every integer add with an equivalent sub.
|
||||
|
||||
This demonstrates the current transformation APIs:
|
||||
|
||||
- `inst.operands`
|
||||
- `inst.replace_all_uses_with(...)`
|
||||
- `inst.erase_from_parent()`
|
||||
|
||||
Run from the repository root with:
|
||||
|
||||
uv run python examples/transform_replace_add.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import textwrap
|
||||
|
||||
import llvm
|
||||
|
||||
|
||||
INPUT_IR = """
|
||||
define i32 @add_one(i32 %x) {
|
||||
entry:
|
||||
%sum = add i32 %x, 1
|
||||
ret i32 %sum
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
def transform(ir_text: str) -> str:
|
||||
with llvm.create_context() as ctx:
|
||||
with ctx.parse_ir(textwrap.dedent(ir_text).strip() + "\n") as mod:
|
||||
for func in mod.functions:
|
||||
if func.is_declaration:
|
||||
continue
|
||||
|
||||
additions = [
|
||||
inst
|
||||
for bb in func.basic_blocks
|
||||
for inst in bb.instructions
|
||||
if inst.opcode == llvm.Opcode.Add
|
||||
]
|
||||
|
||||
for inst in additions:
|
||||
lhs, rhs = inst.operands
|
||||
with inst.create_builder() as builder:
|
||||
replacement = builder.sub(lhs, rhs.type.constant(-1), inst.name + ".repl")
|
||||
inst.replace_all_uses_with(replacement)
|
||||
inst.erase_from_parent()
|
||||
|
||||
assert mod.verify(), mod.get_verification_error()
|
||||
return str(mod)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
print(transform(INPUT_IR), end="")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Smoke tests for public example scripts.
|
||||
|
||||
These keep README-linked examples honest as the bindings evolve.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
|
||||
|
||||
def run_example(path: str) -> str:
|
||||
result = subprocess.run(
|
||||
[sys.executable, path],
|
||||
cwd=PROJECT_ROOT,
|
||||
text=True,
|
||||
capture_output=True,
|
||||
check=True,
|
||||
)
|
||||
return result.stdout
|
||||
|
||||
|
||||
def assert_quick_start_output(output: str) -> None:
|
||||
assert "define i32 @get_answer()" in output
|
||||
assert "ret i32 42" in output
|
||||
|
||||
|
||||
def test_quick_start_example() -> None:
|
||||
output = run_example("examples/quick_start.py")
|
||||
assert_quick_start_output(output)
|
||||
|
||||
|
||||
def test_readme_quick_start_snippet() -> None:
|
||||
readme = (PROJECT_ROOT / "README.md").read_text(encoding="utf-8")
|
||||
match = re.search(r"```python\n(?P<code>.*?get_answer.*?)\n```", readme, re.DOTALL)
|
||||
assert match is not None, "README quick start Python snippet not found"
|
||||
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-c", match.group("code")],
|
||||
cwd=PROJECT_ROOT,
|
||||
text=True,
|
||||
capture_output=True,
|
||||
check=True,
|
||||
)
|
||||
assert_quick_start_output(result.stdout)
|
||||
|
||||
|
||||
def test_transform_replace_add_example() -> None:
|
||||
output = run_example("examples/transform_replace_add.py")
|
||||
assert "define i32 @add_one(i32 %x)" in output
|
||||
assert "%sum.repl = sub i32 %x, -1" in output
|
||||
assert "ret i32 %sum.repl" in output
|
||||
assert " add i32 " not in output
|
||||
Reference in New Issue
Block a user