/* * Copyright (c) 2017 Trail of Bits, Inc. * * Licensed under the Apache License, Version 2.0 (the "License"); * you may not use this file except in compliance with the License. * You may obtain a copy of the License at * * http://www.apache.org/licenses/LICENSE-2.0 * * Unless required by applicable law or agreed to in writing, software * distributed under the License is distributed on an "AS IS" BASIS, * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. * See the License for the specific language governing permissions and * limitations under the License. */ #include "remill/Arch/Arch.h" #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include "remill/Arch/Name.h" #include "remill/BC/ABI.h" #include "remill/BC/Compat/Attributes.h" #include "remill/BC/Compat/DebugInfo.h" #include "remill/BC/Compat/GlobalValue.h" #include "remill/BC/Util.h" #include "remill/BC/Version.h" #include "remill/OS/OS.h" DEFINE_string(arch, REMILL_ARCH, "Architecture of the code being translated. " "Valid architectures: x86, amd64 (with or without " "`_avx` or `_avx512` appended), aarch64"); DECLARE_string(os); namespace remill { class ArchImpl { public: // State type. llvm::StructType *state_type{nullptr}; // Memory pointer type. llvm::PointerType *memory_type{nullptr}; // Lifted function type. llvm::FunctionType *lifted_function_type{nullptr}; // Metadata type ID for remill registers. unsigned reg_md_id{0}; std::vector> registers; std::vector reg_by_offset; std::unordered_map reg_by_name; }; namespace { static unsigned AddressSize(ArchName arch_name) { switch (arch_name) { case kArchInvalid: LOG(FATAL) << "Cannot get address size for invalid arch."; return 0; case kArchMSP430: return 16; case kArchX86: case kArchX86_AVX: case kArchX86_AVX512: case kArchSparc32: return 32; case kArchAMD64: case kArchAMD64_AVX: case kArchAMD64_AVX512: case kArchAArch64LittleEndian: case kArchSparc64: return 64; } return 0; } } // namespace Arch::Arch(llvm::LLVMContext *context_, OSName os_name_, ArchName arch_name_) : os_name(os_name_), arch_name(arch_name_), address_size(AddressSize(arch_name_)), context(context_) {} Arch::~Arch(void) {} // Returns `true` if memory access are little endian byte ordered. bool Arch::MemoryAccessIsLittleEndian(void) const { return true; } // Returns `true` if a given instruction might have a delay slot. bool Arch::MayHaveDelaySlot(const Instruction &) const { return false; } // Returns `true` if a given instruction might have a delay slot. bool Arch::NextInstructionIsDelayed(const Instruction &, const Instruction &, bool) const { return false; } llvm::Triple Arch::BasicTriple(void) const { llvm::Triple triple; switch (os_name) { case kOSInvalid: LOG(FATAL) << "Cannot get triple OS."; break; case kOSLinux: triple.setOS(llvm::Triple::Linux); triple.setEnvironment(llvm::Triple::GNU); triple.setVendor(llvm::Triple::PC); triple.setObjectFormat(llvm::Triple::ELF); break; case kOSmacOS: triple.setOS(llvm::Triple::MacOSX); triple.setEnvironment(llvm::Triple::UnknownEnvironment); triple.setVendor(llvm::Triple::Apple); triple.setObjectFormat(llvm::Triple::MachO); break; case kOSWindows: triple.setOS(llvm::Triple::Win32); triple.setEnvironment(llvm::Triple::MSVC); triple.setVendor(llvm::Triple::UnknownVendor); triple.setObjectFormat(llvm::Triple::COFF); break; case kOSSolaris: triple.setOS(llvm::Triple::Solaris); triple.setEnvironment(llvm::Triple::UnknownEnvironment); triple.setVendor(llvm::Triple::UnknownVendor); triple.setObjectFormat(llvm::Triple::ELF); break; } return triple; } auto Arch::Build(llvm::LLVMContext *context_, OSName os_name_, ArchName arch_name_) -> ArchPtr { switch (arch_name_) { case kArchInvalid: LOG(FATAL) << "Unrecognized architecture."; return nullptr; case kArchAArch64LittleEndian: { DLOG(INFO) << "Using architecture: AArch64, feature set: Little Endian"; return GetAArch64(context_, os_name_, arch_name_); } case kArchX86: { DLOG(INFO) << "Using architecture: X86"; return GetX86(context_, os_name_, arch_name_); } case kArchX86_AVX: { DLOG(INFO) << "Using architecture: X86, feature set: AVX"; return GetX86(context_, os_name_, arch_name_); } case kArchX86_AVX512: { DLOG(INFO) << "Using architecture: X86, feature set: AVX512"; return GetX86(context_, os_name_, arch_name_); } case kArchAMD64: { DLOG(INFO) << "Using architecture: AMD64"; return GetX86(context_, os_name_, arch_name_); } case kArchAMD64_AVX: { DLOG(INFO) << "Using architecture: AMD64, feature set: AVX"; return GetX86(context_, os_name_, arch_name_); } case kArchAMD64_AVX512: { DLOG(INFO) << "Using architecture: AMD64, feature set: AVX512"; return GetX86(context_, os_name_, arch_name_); } case kArchSparc32: { DLOG(INFO) << "Using architecture: 32-bit SPARC"; return GetSPARC(context_, os_name_, arch_name_); } case kArchSparc64: { DLOG(INFO) << "Using architecture: 64-bit SPARC"; return GetSPARC64(context_, os_name_, arch_name_); } case kArchMSP430: { DLOG(INFO) << "Using architecture: TI MSP430"; return GetMSP430(context_, os_name_, arch_name_); } } } auto Arch::Get(llvm::LLVMContext &context, std::string_view os, std::string_view arch_name) -> ArchPtr { return Arch::Build(&context, GetOSName(os), GetArchName(arch_name)); } auto Arch::Get(llvm::LLVMContext &context, OSName os, ArchName arch_name) -> ArchPtr { return Arch::Build(&context, os, arch_name); } auto Arch::GetHostArch(llvm::LLVMContext &ctx) -> ArchPtr { return Arch::Build(&ctx, GetOSName(REMILL_OS), GetArchName(REMILL_ARCH)); } auto Arch::GetTargetArch(llvm::LLVMContext &ctx) -> ArchPtr { return Arch::Build(&ctx, GetOSName(FLAGS_os), GetArchName(FLAGS_arch)); } // Return the type of the state structure. llvm::StructType *Arch::StateStructType(void) const { CHECK(impl) << "Have you not run `PrepareModule` on a loaded semantics module?"; return impl->state_type; } // Pointer to a state structure type. llvm::PointerType *Arch::StatePointerType(void) const { CHECK(impl) << "Have you not run `PrepareModule` on a loaded semantics module?"; return llvm::PointerType::get(impl->state_type, 0); } // Return the type of an address, i.e. `addr_t` in the semantics. llvm::IntegerType *Arch::AddressType(void) const { return llvm::IntegerType::get(*context, address_size); } // The type of memory. llvm::PointerType *Arch::MemoryPointerType(void) const { CHECK(impl) << "Have you not run `PrepareModule` on a loaded semantics module?"; return impl->memory_type; } // Return the type of a lifted function. llvm::FunctionType *Arch::LiftedFunctionType(void) const { CHECK(impl) << "Have you not run `PrepareModule` on a loaded semantics module?"; return impl->lifted_function_type; } // Return information about the register at offset `offset` in the `State` // structure. const Register *Arch::RegisterAtStateOffset(uint64_t offset) const { auto ®_by_offset = impl->reg_by_offset; if (offset >= reg_by_offset.size()) { return nullptr; } else { return reg_by_offset[offset]; // May be `nullptr`. } } // Apply `cb` to every register. void Arch::ForEachRegister(std::function cb) const { for (const auto ® : impl->registers) { cb(reg.get()); } } // Return information about a register, given its name. const Register *Arch::RegisterByName(std::string_view name_) const { std::string name(name_.data(), name_.size()); auto [curr_val_it, added] = impl->reg_by_name.emplace(std::move(name), nullptr); if (added) { return nullptr; } else { return curr_val_it->second; } } namespace { // NOTE(lukas): Structure that allows global caching of `Arch` objects, // as key llvm::LLVMContext * is used. Eventually this should // be removed in favor of Arch::Build/Get* but some old code may // depend on this caching behaviour. struct AvailableArchs { using ArchMap = std::unordered_map>; static ArchMap cached; static const Arch *GetOrCreate(llvm::LLVMContext *ctx, OSName os, ArchName name) { auto &arch = cached[ctx]; if (!arch) { arch = Create(ctx, os, name); } return arch.get(); } static const Arch *Get(llvm::LLVMContext *ctx) { if (auto arch_it = cached.find(ctx); arch_it != cached.end()) return arch_it->second.get(); return nullptr; } static Arch::ArchPtr Create(llvm::LLVMContext *ctx, OSName os, ArchName name) { return Arch::Build(ctx, os, name); } }; AvailableArchs::ArchMap AvailableArchs::cached = {}; } // namespace remill::Arch::ArchPtr Arch::GetModuleArch(const llvm::Module &module) { const llvm::Triple triple = llvm::Triple(module.getTargetTriple()); return remill::Arch::Build(&module.getContext(), GetOSName(triple), GetArchName(triple)); } bool Arch::IsX86(void) const { switch (arch_name) { case remill::kArchX86: case remill::kArchX86_AVX: case remill::kArchX86_AVX512: return true; default: return false; } } bool Arch::IsAMD64(void) const { switch (arch_name) { case remill::kArchAMD64: case remill::kArchAMD64_AVX: case remill::kArchAMD64_AVX512: return true; default: return false; } } bool Arch::IsAArch64(void) const { return remill::kArchAArch64LittleEndian == arch_name; } bool Arch::IsSPARC32(void) const { return remill::kArchSparc32 == arch_name; } bool Arch::IsSPARC64(void) const { return remill::kArchSparc64 == arch_name; } bool Arch::IsMSP430(void) const { return remill::kArchMSP430 == arch_name; } bool Arch::IsWindows(void) const { return remill::kOSWindows == os_name; } bool Arch::IsLinux(void) const { return remill::kOSLinux == os_name; } bool Arch::IsMacOS(void) const { return remill::kOSmacOS == os_name; } bool Arch::IsSolaris(void) const { return remill::kOSSolaris == os_name; } namespace { // These variables must always be defined within `__remill_basic_block`. static bool BlockHasSpecialVars(llvm::Function *basic_block) { return FindVarInFunction(basic_block, kStateVariableName, true) && FindVarInFunction(basic_block, kMemoryVariableName, true) && FindVarInFunction(basic_block, kPCVariableName, true) && FindVarInFunction(basic_block, kNextPCVariableName, true) && FindVarInFunction(basic_block, kBranchTakenVariableName, true); } // Add attributes to llvm::Argument in a way portable across LLVMs static void AddNoAliasToArgument(llvm::Argument *arg) { IF_LLVM_LT_390(arg->addAttr(llvm::AttributeSet::get( arg->getContext(), arg->getArgNo() + 1, llvm::Attribute::NoAlias));); IF_LLVM_GTE_390(arg->addAttr(llvm::Attribute::NoAlias);); } } // namespace Register::Register(const std::string &name_, uint64_t offset_, uint64_t size_, llvm::Type *type_, const Register *parent_, const ArchImpl *arch_) : name(name_), offset(offset_), size(size_), type(type_), constant_name( llvm::ConstantDataArray::getString(type->getContext(), name_)), parent(parent_), arch(arch_) {} // Returns the enclosing register of size AT LEAST `size`, or `nullptr`. const Register *Register::EnclosingRegisterOfSize(uint64_t size_) const { auto enclosing = this; for (; enclosing && enclosing->size < size_; enclosing = enclosing->parent) { /* Empty. */; } return enclosing; } // Returns the largest enclosing register containing the current register. const Register *Register::EnclosingRegister(void) const { auto enclosing = this; while (enclosing->parent) { enclosing = enclosing->parent; } return enclosing; } // Returns the list of directly enclosed registers. For example, // `RAX` will directly enclose `EAX` but nothing else. `AX` will directly // enclose `AH` and `AL`. const std::vector &Register::EnclosedRegisters(void) const { return children; } namespace { // Compute the total offset of a GEP chain. static uint64_t TotalOffset(const llvm::DataLayout &dl, llvm::Value *base, llvm::Type *state_ptr_type) { uint64_t total_offset = 0; const auto state_size = dl.getTypeAllocSize(state_ptr_type->getPointerElementType()); while (base) { if (auto gep = llvm::dyn_cast(base); gep) { llvm::APInt accumulated_offset(dl.getPointerSizeInBits(0), 0, false); CHECK(gep->accumulateConstantOffset(dl, accumulated_offset)); auto curr_offset = accumulated_offset.getZExtValue(); CHECK_LT(curr_offset, state_size); total_offset += curr_offset; CHECK_LT(total_offset, state_size); base = gep->getPointerOperand(); } else if (auto bc = llvm::dyn_cast(base); bc) { base = bc->getOperand(0); } else if (auto itp = llvm::dyn_cast(base); itp) { base = itp->getOperand(0); } else if (auto pti = llvm::dyn_cast(base); pti) { base = pti->getOperand(0); } else if (base->getType() == state_ptr_type) { break; } else { LOG(FATAL) << "Unexpected value " << LLVMThingToString(base) << " in State structure indexing chain"; base = nullptr; } } return total_offset; } static llvm::Value * FinishAddressOf(llvm::IRBuilder<> &ir, const llvm::DataLayout &dl, llvm::Type *state_ptr_type, size_t state_size, const Register *reg, unsigned addr_space, llvm::Value *gep) { auto gep_offset = TotalOffset(dl, gep, state_ptr_type); auto gep_type_at_offset = gep->getType()->getPointerElementType(); CHECK_LT(gep_offset, state_size); const auto index_type = reg->gep_index_list[0]->getType(); const auto goal_ptr_type = llvm::PointerType::get(reg->type, addr_space); // Best case: we've found a value field in the structure that // is located at the correct byte offset. if (gep_offset == reg->offset) { if (gep_type_at_offset == reg->type) { return gep; } else if (auto const_gep = llvm::dyn_cast(gep); const_gep) { return llvm::ConstantExpr::getBitCast(const_gep, goal_ptr_type); } else { return ir.CreateBitCast(gep, goal_ptr_type); } } const auto diff = reg->offset - gep_offset; // Next best case: the difference between what we want and what we have // is a multiple of the size of the register, so we can cast to the // `goal_ptr_type` and index. if (((diff / reg->size) * reg->size) == diff) { llvm::Value *elem_indexes[] = { llvm::ConstantInt::get(index_type, diff / reg->size, false)}; if (auto const_gep = llvm::dyn_cast(gep); const_gep) { const_gep = llvm::ConstantExpr::getBitCast(const_gep, goal_ptr_type); return llvm::ConstantExpr::getGetElementPtr(reg->type, const_gep, elem_indexes); } else { const auto arr = ir.CreateBitCast(gep, goal_ptr_type); return ir.CreateGEP(reg->type, arr, elem_indexes); } } // Worst case is that we have to fall down to byte-granularity // pointer arithmetic. const auto byte_type = llvm::IntegerType::getInt8Ty(goal_ptr_type->getContext()); llvm::Value *elem_indexes[] = { llvm::ConstantInt::get(index_type, diff, false)}; if (auto const_gep = llvm::dyn_cast(gep); const_gep) { const_gep = llvm::ConstantExpr::getBitCast( const_gep, llvm::PointerType::get(byte_type, addr_space)); const_gep = llvm::ConstantExpr::getGetElementPtr(byte_type, const_gep, elem_indexes); return llvm::ConstantExpr::getBitCast(const_gep, goal_ptr_type); } else { gep = ir.CreateBitCast(gep, llvm::PointerType::get(byte_type, addr_space)); gep = ir.CreateGEP(byte_type, gep, elem_indexes); return ir.CreateBitCast(gep, goal_ptr_type); } } } // namespace // Generate a GEP that will let us load/store to this register, given // a `State *`. llvm::Value *Register::AddressOf(llvm::Value *state_ptr, llvm::BasicBlock *add_to_end) const { llvm::IRBuilder<> ir(add_to_end); return AddressOf(state_ptr, ir); } llvm::Value *Register::AddressOf(llvm::Value *state_ptr, llvm::IRBuilder<> &ir) const { auto &context = type->getContext(); CHECK_EQ(&context, &(state_ptr->getContext())); const auto state_ptr_type = llvm::dyn_cast(state_ptr->getType()); CHECK_NOTNULL(state_ptr_type); const auto addr_space = state_ptr_type->getAddressSpace(); const auto state_type = llvm::dyn_cast(state_ptr_type->getPointerElementType()); CHECK_NOTNULL(state_type); const auto module = ir.GetInsertBlock()->getParent()->getParent(); const auto &dl = module->getDataLayout(); llvm::Value *gep = nullptr; if (auto const_state_ptr = llvm::dyn_cast(state_ptr); const_state_ptr) { gep = llvm::ConstantExpr::getInBoundsGetElementPtr( state_type, const_state_ptr, gep_index_list); } else { gep = ir.CreateInBoundsGEP(state_type, state_ptr, gep_index_list); } auto state_size = dl.getTypeAllocSize(state_type); auto ret = FinishAddressOf(ir, dl, state_ptr_type, state_size, this, addr_space, gep); // Add the metadata to `inst`. if (auto inst = llvm::dyn_cast(ret); inst) { #if LLVM_VERSION_NUMBER >= LLVM_VERSION(3, 6) auto reg_name_md = llvm::ValueAsMetadata::get(constant_name); auto reg_name_node = llvm::MDNode::get(context, reg_name_md); #else auto reg_name_node = llvm::MDNode::get(context, reg->constant_name); #endif inst->setMetadata(arch->reg_md_id, reg_name_node); inst->setName(name); } return ret; } // Converts an LLVM module object to have the right triple / data layout // information for the target architecture. // void Arch::PrepareModuleDataLayout(llvm::Module *mod) const { mod->setDataLayout(DataLayout().getStringRepresentation()); mod->setTargetTriple(Triple().str()); // Go and remove compile-time attributes added into the semantics. These // can screw up later compilation. We purposefully compile semantics with // things like auto-vectorization disabled so that it keeps the bitcode // to a simpler subset of the available LLVM instuction set. If/when we // compile this bitcode back into machine code, we may want to use those // features, and clang will complain if we try to do so if these metadata // remain present. auto &context = mod->getContext(); llvm::AttributeSet target_attribs; target_attribs = target_attribs.addAttribute( context, IF_LLVM_LT_500_(llvm::AttributeSet::FunctionIndex) "target-features"); target_attribs = target_attribs.addAttribute( context, IF_LLVM_LT_500_(llvm::AttributeSet::FunctionIndex) "target-cpu"); for (llvm::Function &func : *mod) { auto attribs = func.getAttributes(); attribs = attribs.removeAttributes( context, llvm::AttributeLoc::FunctionIndex, target_attribs); func.setAttributes(attribs); } } void Arch::PrepareModule(llvm::Module *mod) const { CHECK_EQ(&(mod->getContext()), context); PrepareModuleDataLayout(mod); } const Register *Arch::AddRegister( const char *reg_name_, llvm::Type *val_type, size_t offset, const char *parent_reg_name) const { const std::string reg_name(reg_name_); auto ® = impl->reg_by_name[reg_name]; if (reg) { return reg; } const auto dl = this->DataLayout(); // If this is a sub-register, then link it in. const Register *parent_reg = nullptr; if (parent_reg_name) { parent_reg = impl->reg_by_name[parent_reg_name]; } llvm::SmallVector gep_index_list; gep_index_list.push_back( llvm::Constant::getNullValue(llvm::Type::getInt32Ty(*context))); auto [gep_offset, gep_type_at_offset] = BuildIndexes( dl, impl->state_type, 0, offset, gep_index_list); if (!val_type) { CHECK_EQ(gep_offset, offset); val_type = gep_type_at_offset; } auto reg_impl = new Register(reg_name, offset, dl.getTypeAllocSize(val_type), val_type, parent_reg, impl.get()); reg_impl->gep_index_list = std::move(gep_index_list); reg_impl->gep_offset = gep_offset; reg_impl->gep_type_at_offset = gep_type_at_offset; reg = reg_impl; impl->registers.emplace_back(reg_impl); if (parent_reg) { const_cast(reg->parent)->children.push_back(reg); } // Provide easy access to registers at specific offsets in the `State` // structure. for (auto i = reg->offset; i < (reg->offset + reg->size); ++i) { auto ®_at_offset = impl->reg_by_offset[i]; if (!reg_at_offset) { reg_at_offset = reg; } else if (reg_at_offset) { CHECK_EQ(reg_at_offset->EnclosingRegister(), reg->EnclosingRegister()); reg_at_offset = reg; } } return reg; } // Get all of the register information from the prepared module. void Arch::InitFromSemanticsModule(llvm::Module *module) const { if (impl) { return; } impl.reset(new ArchImpl); CHECK(!impl->state_type); const auto &dl = module->getDataLayout(); const auto basic_block = BasicBlockFunction(module); const auto state_ptr_type = NthArgument(basic_block, kStatePointerArgNum)->getType(); const auto state_type = llvm::dyn_cast(state_ptr_type->getPointerElementType()); impl->state_type = state_type; impl->reg_by_offset.resize(dl.getTypeAllocSize(state_type)); impl->memory_type = llvm::dyn_cast( NthArgument(basic_block, kMemoryPointerArgNum)->getType()); impl->lifted_function_type = basic_block->getFunctionType(); impl->reg_md_id = context->getMDKindID("remill_register"); if (basic_block->isDeclaration()) { basic_block->setLinkage(llvm::GlobalValue::ExternalLinkage); basic_block->setVisibility(llvm::GlobalValue::DefaultVisibility); basic_block->removeFnAttr(llvm::Attribute::AlwaysInline); basic_block->removeFnAttr(llvm::Attribute::InlineHint); basic_block->addFnAttr(llvm::Attribute::OptimizeNone); basic_block->addFnAttr(llvm::Attribute::NoInline); basic_block->setVisibility(llvm::GlobalValue::DefaultVisibility); auto memory = remill::NthArgument(basic_block, kMemoryPointerArgNum); auto state = remill::NthArgument(basic_block, kStatePointerArgNum); memory->setName("memory"); state->setName("state"); remill::NthArgument(basic_block, kPCArgNum)->setName("pc"); AddNoAliasToArgument(state); AddNoAliasToArgument(memory); auto block = llvm::BasicBlock::Create(module->getContext(), "", basic_block); auto u8 = llvm::Type::getInt8Ty(*context); auto addr = llvm::Type::getIntNTy(*context, address_size); llvm::IRBuilder<> ir(block); ir.CreateAlloca(u8, nullptr, "BRANCH_TAKEN"); ir.CreateAlloca(addr, nullptr, "RETURN_PC"); ir.CreateAlloca(addr, nullptr, "MONITOR"); // NOTE(pag): `PC` and `NEXT_PC` are handled by // `PopulateBasicBlockFunction`. ir.CreateStore(state, ir.CreateAlloca(llvm::PointerType::get(impl->state_type, 0), nullptr, "STATE")); ir.CreateStore(memory, ir.CreateAlloca(impl->memory_type, nullptr, "MEMORY")); PopulateBasicBlockFunction(module, basic_block); ir.SetInsertPoint(block); ir.CreateRet(memory); } else { CHECK(!impl->reg_by_name.empty()); } CHECK(BlockHasSpecialVars(basic_block)) << "Unable to locate required variables in `__remill_basic_block`."; } } // namespace remill