mirror of
https://github.com/HullaBrian/ttd-capa
synced 2026-08-09 12:08:34 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
62c36a2c9a | ||
|
|
bf80e4a71f | ||
|
|
e4477ec15e | ||
|
|
2750eba80c | ||
|
|
61d08f628a | ||
|
|
5563641071 | ||
|
|
ac6151331b | ||
|
|
e66fb999fc | ||
|
|
2888975ff6 | ||
|
|
2d98e5e58f | ||
|
|
fc4c2ef00a | ||
|
|
75cde9091d | ||
|
|
c8c4b56b52 | ||
|
|
64afd8c401 | ||
|
|
bdf0e048c7 | ||
|
|
e716a83bfd | ||
|
|
7954276ad8 | ||
|
|
a5c5d7f240 | ||
|
|
8ca0ccdea7 | ||
|
|
80c39ade4f | ||
|
|
19dc1581ce | ||
|
|
f9db02e59e | ||
|
|
e6bb901599 | ||
|
|
40842e1c50 | ||
|
|
a9ac90da8a | ||
|
|
2c642a7ed8 | ||
|
|
0b8a582cf1 | ||
|
|
4d107384b5 | ||
|
|
6cb97d6cb1 | ||
|
|
a044cd9b42 | ||
|
|
9f0c73dced | ||
|
|
2aca37a548 | ||
|
|
affc14b111 |
@@ -0,0 +1,89 @@
|
||||
name: build
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [master, metadata-decoding]
|
||||
tags: ['v*']
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: build-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
jobs:
|
||||
extractor:
|
||||
# ttdcapa-extract.vcxproj pins PlatformToolset v145, which ships only with
|
||||
# Visual Studio 2026. windows-2025/windows-latest still carry VS 2022
|
||||
# (v143 only) and fail with MSB8020, so the VS 2026 image is required.
|
||||
# Fold this back into windows-2025 once that image carries VS 2026.
|
||||
runs-on: windows-2025-vs2026
|
||||
permissions:
|
||||
contents: write # needed by the tagged-release step below
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
# win32json is only needed to regenerate the API index, which is committed.
|
||||
# Cloning it costs several hundred MB for nothing on a normal build.
|
||||
submodules: false
|
||||
|
||||
- uses: microsoft/setup-msbuild@v2
|
||||
with:
|
||||
# Pin to the VS 2026 (18.x) install; vswhere would otherwise be free to
|
||||
# pick up an older side-by-side MSBuild that has no v145 toolset.
|
||||
vs-version: '[18.0,19.0)'
|
||||
|
||||
- uses: NuGet/setup-nuget@v2
|
||||
|
||||
- name: Cache NuGet packages
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ttd/packages
|
||||
key: nuget-${{ runner.os }}-${{ hashFiles('ttd/packages.config') }}
|
||||
|
||||
# packages.config-style restore: the .vcxproj imports the TTD and
|
||||
# nlohmann.json props/targets from ttd\packages\ by literal path.
|
||||
- name: Restore NuGet packages
|
||||
working-directory: ttd
|
||||
run: nuget restore packages.config -PackagesDirectory packages
|
||||
|
||||
- name: Build
|
||||
working-directory: ttd
|
||||
run: msbuild ttdcapa-extract.vcxproj /p:Configuration=Release /p:Platform=x64 /m /v:minimal
|
||||
|
||||
# The extractor delay-loads TTDReplay.dll and takes --ttd-dlls, so it links and
|
||||
# runs without Microsoft's replay DLLs present. They are not ours to redistribute;
|
||||
# a caller points at the copy WinDbg already installed.
|
||||
- name: Check it runs
|
||||
working-directory: ttd/bin/x64/Release
|
||||
run: |
|
||||
./ttdcapa-extract.exe --dump-sig CreateFileW
|
||||
if ($LASTEXITCODE -ne 0) { throw "dump-sig failed" }
|
||||
|
||||
- name: Stage artifacts
|
||||
run: |
|
||||
New-Item -ItemType Directory -Force dist | Out-Null
|
||||
Copy-Item ttd/bin/x64/Release/ttdcapa-extract.exe dist/
|
||||
Copy-Item ttd/data/win32-index.bin dist/
|
||||
Copy-Item docs/ttdcapa-extract.md dist/
|
||||
Get-ChildItem dist | Select-Object Name, Length
|
||||
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: ttdcapa-extract-x64
|
||||
path: dist/
|
||||
if-no-files-found: error
|
||||
|
||||
- name: Publish release
|
||||
if: startsWith(github.ref, 'refs/tags/v')
|
||||
uses: softprops/action-gh-release@v2
|
||||
with:
|
||||
files: dist/*
|
||||
body: |
|
||||
`ttdcapa-extract.exe` sweeps a TTD trace and writes a `.ttdb` report.
|
||||
|
||||
- `win32-index.bin` must sit beside the executable, or be passed with `--win32-index`.
|
||||
- `TTDReplay.dll` and `TTDReplayCPU.dll` are Microsoft's and are not redistributed
|
||||
here. Pass `--ttd-dlls <dir>` pointing at WinDbg's `amd64\ttd` folder, or place
|
||||
copies beside the executable.
|
||||
- `ttdcapa-extract.md` documents the command line and the `.ttdb` layout.
|
||||
@@ -40,3 +40,5 @@ capa-9.4.0/.venv/
|
||||
!tests/sample.ttd.json
|
||||
*.dll
|
||||
/ttd/bin
|
||||
/.vs
|
||||
/ttd/obj
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
[submodule "capa"]
|
||||
path = capa
|
||||
url = https://github.com/HullaBrian/capa
|
||||
[submodule "win32json"]
|
||||
path = win32json
|
||||
url = https://github.com/marlersoft/win32json
|
||||
|
||||
@@ -0,0 +1,201 @@
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -1,146 +1,158 @@
|
||||
# Overview
|
||||
> [!WARNING]
|
||||
> ttd-capa is still early in development. You may encounter unexpected issues and rough edges. If you hit a bug or have a suggestion, please open an issue or PR.
|
||||
|
||||
ttd-capa is a [CAPA](https://github.com/mandiant/capa) compatible capability extractor built for Time Travel Debugging (TTD) traces.
|
||||
TTD records the complete execution of a process, and exposes a vast amount of information which can be used for analysis purposes.
|
||||
# ttd-capa
|
||||
|
||||
ttd-capa is a [CAPA](https://github.com/mandiant/capa)-compatible capability extractor for Time Travel Debugging (TTD) traces. TTD records the complete execution of a process, and ttd-capa mines that recording for the capabilities a binary actually exercised at runtime.
|
||||
|
||||

|
||||
|
||||
With this tool, reverse engineers can get signifcantly more information from a binary with just a TTD trace, a tool, and a dream.
|
||||
Take a packed CobaltStrike beacon as an example (see above screenshot). With just a static CAPA scan, only a few things might be observed (left side of screenshot).
|
||||
If you're lucky, you may even see the type of packer being used (i.e. UPX). However, not much can be determined about the actual capabilities
|
||||
of the packed executable code. This is where ttd-capa shines. Once a TTD trace is recorded, ttd-capa can scan the entire trace and identify
|
||||
capabilities only revealed during runtime (right side of screenshot).
|
||||
A static CAPA scan only sees what is visible without running the sample. Against a packed binary, that is often little more than the packer itself (UPX in the screenshot above). ttd-capa scans the full trace instead, so capabilities that only appear once the sample unpacks and executes become visible (right side of the screenshot). This makes it useful for triaging packed or obfuscated malware, where the interesting behavior is hidden until runtime.
|
||||
|
||||
# How It Works
|
||||
|
||||
ttd-capa is not a standalone tool. It runs alongside CAPA, producing a report that CAPA then matches against its existing rules.
|
||||
|
||||

|
||||
ttd-capa is not a standalone tool. Rather, it is meant to be run in unison with the CAPA tool so as to extract information useful for CAPA rule matching.
|
||||
The general capability extraction process looks like:
|
||||
1. Get a TTD trace of a given sample
|
||||
2. Run ttd-capa on the trace, which generates a CAPA-compatible JSON report
|
||||
3. Run CAPA on the JSON report, which uses existing CAPA rules to extract capability information
|
||||
|
||||
As a disclaimer, I am not a professional software developer. As such, you may encounter unexpected issues and/or spaghetti code in this project. If you'd
|
||||
like to contribute, I'm more than happy to take a look at PRs, but please include readable and descriptive code/comments.
|
||||
The workflow is:
|
||||
|
||||
# How it Works
|
||||
ttd-capa uses the official Microsoft TTD C++ SDK to interact with TTD, and [nlohmann/json](https://github.com/nlohmann/json) for working with JSON.
|
||||
1. Record a TTD trace of the sample.
|
||||
2. Run ttd-capa on the trace to generate a CAPA-compatible JSON report.
|
||||
3. Run CAPA on the report to extract capabilities using the standard rule set.
|
||||
|
||||
ttd-capa will first gather a list of module load events and navigate to each. Once there, it will extract the exported functions directly from the TTD trace
|
||||
and store their associated virtual address, function name, and module name. Next, ttd-capa registeres a call callback with the TTD engine, and iterates over
|
||||
the entire trace. Every time a call occurs, ttd-capa checks if the call target is one stored in the module export map. If it is, then it will log the call along
|
||||
with the associate module, function name, parameters, and return value. Additionally, ttd-capa automatically attempts to resolve function parameters as strings,
|
||||
which has the potential to significantly increase quick wins during malware analysis.
|
||||
To build the report, ttd-capa walks the module load events in the trace, extracts each module's exported functions, then sweeps the entire trace watching for calls into those exports. Every matching call is logged with its module, function name, arguments, and return value.
|
||||
|
||||
Arguments are decoded using Microsoft's own Win32 API metadata, which gives ttd-capa each function's real parameter count and types. Because a TTD trace can be read at any point in time, `[Out]` parameters are re-read at the call's return, so values a function writes back to the caller (buffers, output handles, byte counts) are captured filled in. String arguments are resolved automatically, which tends to surface quick wins during analysis. Calls with no available metadata fall back to a register-based heuristic.
|
||||
|
||||
Because TTD records timestamps, ttd-capa can also reconstruct the order and timing of capabilities across execution, not just the set of capabilities present.
|
||||
|
||||
# Prerequisites
|
||||
- Windows x64 (the extractor links the TTD Replay runtime)
|
||||
- Visual Studio
|
||||
- Python 3.10+ for CAPA
|
||||
- A TTD `.run` trace (plus its `.idx`; the extractor builds the `.idx` on first run
|
||||
if missing). Record one with WinDbg Preview's Time-travel debugging or `tttracer.exe`.
|
||||
- x64 traces only in this version.
|
||||
|
||||
# Building ttd-capa
|
||||
1. Open `ttd/ttdcapa-extract.sln` in Visual Studio
|
||||
2. Ensure that the required nuget packages (`Microsoft.TimeTravelDebugging.Apis` and `nlohmann.json`) are installed
|
||||
3. Set the build mode to `x64` and `Release`
|
||||
4. Navigate to `Build > Build Solution` to being the build
|
||||
|
||||
After the build, ttd-capa cannot run properly without Microsoft's `TTDReplay.dll` and `TTDReplayCPU.dll` being in the same directory as `ttdcapa-extract.exe`.
|
||||
To get those DLLs, ensure you have WinDbg instealled already. Then, run the following PowerShell command to find the DLL location on your system:
|
||||
- Windows
|
||||
- Microsoft C++ Build Tools, v143 (VS2022) or newer
|
||||
- TTD DLLs (`TTDReplay.dll` and `TTDReplayCPU.dll`), which ship with WinDbg
|
||||
- Python 3.10+ (for CAPA)
|
||||
|
||||
# Building
|
||||
|
||||
Clone with submodules, since the `win32json` submodule supplies the API metadata:
|
||||
|
||||
```powershell
|
||||
git clone --recursive <repo-url>
|
||||
# or, in an existing clone:
|
||||
git submodule update --init
|
||||
```
|
||||
|
||||
Then build the extractor:
|
||||
|
||||
1. Open `ttd/ttdcapa-extract.sln` in Visual Studio.
|
||||
2. Install the required NuGet packages (`Microsoft.TimeTravelDebugging.Apis` and `nlohmann.json`).
|
||||
3. Set the configuration to `x64` / `Release`.
|
||||
4. Build the solution (`Build > Build Solution`).
|
||||
|
||||
The extractor also needs Microsoft's `TTDReplay.dll` and `TTDReplayCPU.dll` at runtime. These ship with WinDbg, not with this project. Locate them with:
|
||||
|
||||
```powershell
|
||||
Join-Path (Get-AppxPackage Microsoft.WinDbg).InstallLocation 'amd64\ttd'
|
||||
```
|
||||
|
||||
Then, copy `TTDReplay.dll` and `TTDReplayCPU.dll` into the same directory as `ttdcapa-extract.exe`.
|
||||
Then either pass `--ttd-dlls <that path>` when running, or copy both DLLs next to `ttdcapa-extract.exe`. See [docs/ttdcapa-extract.md](docs/ttdcapa-extract.md) for the full option reference.
|
||||
|
||||
# (Temporary) Installing TTD compatible CAPA
|
||||
I'm opening a PR to the main CAPA repository, so hopefully CAPA will eventually have native support
|
||||
for TTD traces. In the meantime, you can install a custom fork of CAPA. To start, clone [https://github.com/HullaBrian/capa](https://github.com/HullaBrian/capa)
|
||||
# Installing a TTD-compatible CAPA (temporary)
|
||||
|
||||
A PR to upstream CAPA is in progress, so native TTD support should land there eventually. Until then, use the fork.
|
||||
|
||||
Clone and install [HullaBrian/capa](https://github.com/HullaBrian/capa):
|
||||
|
||||
Once you have cloned the repository, create a Python virtual environment, and install CAPA:
|
||||
```powershell
|
||||
git clone https://github.com/HullaBrian/capa
|
||||
cd capa
|
||||
python -m venv .venv
|
||||
.\.venv\Scripts\Activate.ps1
|
||||
pip install -e .
|
||||
```
|
||||
|
||||
Capability rules are a separate repo maintained by the CAPA team. Clone the rules release matching
|
||||
the version of the (temporary) fork (should be 9.4.0): [https://github.com/mandiant/capa-rules/archive/refs/tags/v9.4.0.zip](https://github.com/mandiant/capa-rules/archive/refs/tags/v9.4.0.zip)
|
||||
Capability rules live in a separate repo. Download the rules release matching the fork (9.4.0): [capa-rules v9.4.0](https://github.com/mandiant/capa-rules/archive/refs/tags/v9.4.0.zip).
|
||||
|
||||
# Usage
|
||||
## Python Wrapper Script
|
||||
Included in this repository is `ttd-capa.py`, which is a wrapper script that abstracts some of the "plumbing" away and makes
|
||||
extracting the features and performing rule matching a single step.
|
||||
|
||||
## Python wrapper (recommended)
|
||||
|
||||
`ttd-capa.py` combines feature extraction and rule matching into a single step:
|
||||
|
||||
```powershell
|
||||
python ttd-capa.py <trace.run> <rules-dir> [--sample sample.exe] [-- <capa args>]
|
||||
```
|
||||
|
||||
The wrapper runs the extractor to a temp report (deleted after use), then invokes `capa -f ttd`. Anything
|
||||
after a bare `--` is forwarded to CAPA:
|
||||
It runs the extractor to a temporary report and then invokes `capa -f ttd`. Anything after a bare `--` is passed straight through to CAPA:
|
||||
|
||||
```powershell
|
||||
python ttd-capa.py <trace.run> <rules-dir> --sample sample.exe -- -vv
|
||||
```
|
||||
|
||||
Useful flags:
|
||||
- `--extractor <path>` - specify a path to the ttd-capa extractor executable
|
||||
- `--max-calls N` - cap huge traces to certain number of calls
|
||||
- `--with-stack-args` - captures parameters for function calls that are on the stack (capture more than 5+ parameters for function calls)
|
||||
- `--keep-json` - keep the generated JSON report after the Python script runs
|
||||
|
||||
- `--extractor <path>` - path to the ttd-capa extractor executable
|
||||
- `--max-calls N` - cap the number of calls processed on very large traces
|
||||
- `--max-buffer N` - bytes to keep from any one captured buffer (default 65536)
|
||||
- `--keep-json` - keep the generated JSON report after the run
|
||||
|
||||
See [docs/ttdcapa-extract.md](docs/ttdcapa-extract.md) for the complete list.
|
||||
|
||||
## Manual
|
||||
To manually extract capability features without the Python wrapper script:
|
||||
|
||||
To run the two steps yourself:
|
||||
|
||||
```powershell
|
||||
# 1) generate report from the trace
|
||||
<PATH TO TTDCAPA-EXTRACT.EXE> <PATH TO TTD TRACE> --sample <OPTIONAL SAMPLE FILE> -o <OUTPUT JSON PATH>
|
||||
# 1) generate the report from the trace
|
||||
<ttdcapa-extract.exe> <trace.run> --sample <optional sample file> -o <output.json>
|
||||
|
||||
# 2) run capa against the report
|
||||
python -m capa.main -f ttd -r <CAPA RULES DIRECTORY> <OUTPUT JSON PATH>
|
||||
python -m capa.main -f ttd -r <rules-dir> <output.json>
|
||||
```
|
||||
|
||||
# Timeline Generation
|
||||
TTD grants malware analysts the unique opportunity to not only identify statically present capabilities, but those capabilities
|
||||
only revealed during execution. Thankfully, ttd-capa exposes TTD timestamps, which means that not only can we get the capabilities
|
||||
over the entire execution of the sample, but also the order and time when they were leveraged.
|
||||
## Timeline
|
||||
|
||||
`ttd-timeline.py` uses the TTD timestamps in a report to show capabilities (or all observed API calls) in execution order:
|
||||
|
||||
```powershell
|
||||
# executed capabilities only
|
||||
python ttd-timeline.py <report.json> -r <rules-dir>
|
||||
|
||||
# all observed API calls
|
||||
python ttd-timeline.py <report.json> --calls
|
||||
```
|
||||
|
||||
Example output for a UPX-packed Cobalt Strike beacon:
|
||||
|
||||
See the following example generated by `ttd-timeline.py` on a JSON report for a UPX-packed CobaltStrike beacon. Within the timeline
|
||||
is a clearly defined order of the executed capabilities, along with the associated function parameters. This can provide quick wins
|
||||
during initial triage efforts.
|
||||
```
|
||||
TTD POS TID CAPABILITY NAMESPACE TRIGGERING CALL
|
||||
--------------------------------------------------------------------------------------------------------------
|
||||
8F6:1EF2 4 link function at runtime on Windows linking/runtime-linking kernel32.GetProcAddress(hModule=0x7ffc23060000, lpProcName='InternetConnectA') -> 0x7ffc23129740
|
||||
8F9:153A 4 link function at runtime on Windows linking/runtime-linking kernel32.GetProcAddress(hModule=0x7ffc23060000, lpProcName='InternetOpenA') -> 0x7ffc2312ab30
|
||||
...
|
||||
38DF:F52 4 create HTTP request communication/http/client wininet.InternetOpenA('Mozilla/4.0 (compatible; MSIE 5.0; Windows NT; DigExt; DTS Agent', 0x0, 0x0, 0x0, 0x0, 0x19b5040, 0xcdbc90, '/jquery-3.3.1.min.js') -> 0xcc0004
|
||||
38E5:6DA 4 connect to HTTP server communication/http/client wininet.InternetConnectA(0xcc0004, '192.168.81.129', 0x50, 0x0, 0x0, 0x3, 0x0, 0x11584c4) -> 0xcc0008
|
||||
3E3E:F52 4 create HTTP request communication/http/client wininet.InternetOpenA(lpszAgent='Mozilla/4.0 (compatible; MSIE 5.0; Windows NT; DigExt; DTS Agent', dwAccessType=0x0, lpszProxy=0x0, lpszProxyBypass=0x0, dwFlags=0x0) -> 0xcc0004
|
||||
3E3E:25B7 4 connect to HTTP server communication/http/client wininet.InternetConnectA(hInternet=0xcc0004, lpszServerName='192.168.81.129', nServerPort=0x50, lpszUserName=0x0, lpszPassword=0x0, dwService=0x3, dwFlags=0x0, dwContext=0xb684c4) -> 0xcc0008
|
||||
...
|
||||
```
|
||||
|
||||
`ttd-timeline.py` allows you to either display execution capabilities or all observed API calls within the trace in the final timeline view.
|
||||
# Supported
|
||||
|
||||
To see only the executed capabilities in the trace:
|
||||
```powershell
|
||||
python ttd-timeline.py <JSON REPORT PATH> -r <CAPA RULES PATH>
|
||||
```
|
||||
- x64 and x86 traces, including WoW64 (bitness is decided per call from the owning module's PE header)
|
||||
- Functions directly exported by loaded modules
|
||||
- Exact argument decoding for APIs covered by the public Win32 SDK metadata
|
||||
|
||||
To see all observed API calls during the trace:
|
||||
```powershell
|
||||
python ttd-timeline.py <JSON REPORT PATH> --calls
|
||||
```
|
||||
# Not Supported
|
||||
|
||||
# Limitations
|
||||
- Only x64 traces are supported at the moment (no x86 or ARM)
|
||||
- Argument captures are heuristic, so errors may occur
|
||||
- Only the functions directly exported by loaded modules are logged in the JSON report
|
||||
- Dynamically allocated code. Any code which is present during the recording only is not scanned by ttd-capa...for now... (coming soon)
|
||||
- ARM64 traces
|
||||
- COM interface methods (the call target is a vtable slot, not a named export, and the metadata has no vtable indices)
|
||||
- Variadic functions (`printf`-style): only their fixed parameters are decoded
|
||||
- Structure fields: struct parameters are recorded as pointers and not expanded
|
||||
- Exact argument decoding for `ntdll` internals, undocumented APIs, and CRT helpers, which fall back to the register heuristic and may be wrong
|
||||
|
||||
# Verifying the backend in isolation
|
||||
`tests/test_ttd_extractor.py` loads a report and dumps every feature per scope — handy to confirm a call yields
|
||||
the expected `API`/`Number`/`String` features:
|
||||
A few caveats worth knowing:
|
||||
|
||||
```powershell
|
||||
python tests\test_ttd_extractor.py [report.ttd.json]
|
||||
```
|
||||
|
||||
With no argument it uses the bundled `tests/sample.ttd.json` fixture.
|
||||
- Some string and buffer arguments come back empty even when the data is in the trace. This is a property of the memory-read interface used during the sweep, not evidence that the argument was null; the raw pointer value is still recorded. Recovering these values is possible but expensive, so it is not done yet.
|
||||
- On x86, some signatures are undecodable because a pointer-sized parameter (`SIZE_T`, `WPARAM`, `LPARAM`, etc.) is ambiguous against the x64 width stored in the index. Affected calls fall back to the heuristic; `--dump-sig` marks them `[no x86 layout]`.
|
||||
|
||||
+1
-1
Submodule capa updated: c206922f04...f8c0296883
@@ -0,0 +1,286 @@
|
||||
# `ttdcapa-extract`
|
||||
|
||||
Sweeps a Time Travel Debugging trace and records every Windows API call it made: when it
|
||||
happened, which function was called, what its arguments were, and what it returned.
|
||||
Arguments are decoded against Microsoft's Win32 API metadata where a signature exists, so
|
||||
parameters come back named and typed rather than as four guessed registers.
|
||||
|
||||
It writes either of two things, or both:
|
||||
|
||||
- **`-o report.json`** -- the JSON report [capa](https://github.com/mandiant/capa) consumes.
|
||||
- **`-b report.ttdb`** -- a compact binary layout meant to be memory-mapped and browsed.
|
||||
On a 3.4M-call trace it takes 3.6s to write against ~56s for the JSON, and loading it is
|
||||
a memory map rather than a 17s parse. The format is specified [below](#the-ttdb-format).
|
||||
|
||||
## Usage
|
||||
|
||||
```
|
||||
ttdcapa-extract <trace.run> [-o <out.json>] [-b <out.ttdb>] [options]
|
||||
ttdcapa-extract --dump-sig <ApiName>
|
||||
```
|
||||
|
||||
### Output
|
||||
|
||||
| Option | Effect |
|
||||
| --- | --- |
|
||||
| `-o`, `--output <path>` | write the JSON report |
|
||||
| `-b`, `--binary <path>` | write the binary report |
|
||||
| `--sample <path>` | hash this file and record its MD5/SHA1/SHA256 in the report |
|
||||
|
||||
Giving neither `-o` nor `-b` sweeps the trace and writes nothing, which is only useful for
|
||||
timing.
|
||||
|
||||
### Finding the replay engine
|
||||
|
||||
| Option | Effect |
|
||||
| --- | --- |
|
||||
| `--ttd-dlls <dir>` | where `TTDReplay.dll` and `TTDReplayCPU.dll` live |
|
||||
|
||||
Those DLLs are Microsoft's, shipped with WinDbg, and are not redistributed with this tool.
|
||||
Point at the copy WinDbg already installed -- normally its `amd64 td` folder:
|
||||
|
||||
```powershell
|
||||
Join-Path (Get-AppxPackage Microsoft.WinDbg).InstallLocation 'amd64 td'
|
||||
```
|
||||
|
||||
Without the flag the usual DLL search order applies, so copies placed beside the
|
||||
executable work too. `TTDReplay.dll` is delay-loaded, which is what lets the location be
|
||||
chosen at runtime; a missing engine is reported rather than crashing the process.
|
||||
|
||||
### Argument decoding
|
||||
|
||||
| Option | Effect |
|
||||
| --- | --- |
|
||||
| `--win32-index <path>` | use a specific `win32-index.bin` instead of the one beside the executable |
|
||||
| `--no-metadata` | ignore the metadata entirely and use the register/stack heuristic everywhere |
|
||||
| `--with-stack-args` | for calls with *no* signature, also grab four stack slots past the register arguments. Calls that have a signature always capture their true arity regardless |
|
||||
| `--max-buffer N` | bytes to keep from any one captured buffer (default 65536). Buffers cut short at this limit carry their real length, so a consumer can tell a short buffer from a truncated one |
|
||||
| `--max-calls N` | stop recording after N calls (default 0, unlimited). The sweep still runs to the end |
|
||||
| `--dump-sig <ApiName>` | print one function's decoded signature and exit. Needs only the metadata index, not a trace or the replay DLLs, which makes it a good smoke test |
|
||||
|
||||
### Driving it from another program
|
||||
|
||||
| Option | Effect |
|
||||
| --- | --- |
|
||||
| `--progress` | emit progress on stderr |
|
||||
| `--cancel-on-stdin` | stop the sweep when a line reading `cancel` arrives on stdin |
|
||||
|
||||
`--progress` writes `[phase] sweep` and `[phase] write` markers and, during the sweep,
|
||||
`[progress] <percent> <calls>` a few times a second. The phases matter because the sweep is
|
||||
only part of the wall clock: on a 3.4M-call trace it takes about 30s while serialising the
|
||||
report takes comparable time, so a caller tracking only the sweep would sit at 100% looking
|
||||
wedged. Percentages are clamped to be non-decreasing, since positions read from different
|
||||
threads are not perfectly ordered.
|
||||
|
||||
`--cancel-on-stdin` interrupts the replay and then writes the report as usual: everything
|
||||
recorded up to that point is kept, so a cancelled run yields a complete, valid report of a
|
||||
shorter prefix of the trace.
|
||||
|
||||
## Notes on what the data can and cannot tell you
|
||||
|
||||
**A missing string argument is not evidence the argument was null.** The sweep reads guest
|
||||
memory through the `IThreadView` the replay engine hands to a callback, and the SDK fixes
|
||||
those reads to `QueryMemoryPolicy::ThreadLocal` -- a fast lookup explicitly allowed to
|
||||
ignore memory observed by other threads or by this thread at another time. It returns
|
||||
nothing for roughly a quarter of string parameters even though the data is in the trace;
|
||||
reading the same address at the same position through a cursor returns the whole string.
|
||||
The parameter's raw pointer value is still recorded, and is usually valid.
|
||||
|
||||
**Calls without a signature have guessed arguments.** Coverage is the public Windows SDK,
|
||||
so `ntdll` internals and CRT helpers fall back to capturing argument registers (x64) or
|
||||
stack slots (x86). Their values are positional, unnamed, and the count is a guess. The
|
||||
binary format flags these per call.
|
||||
|
||||
## The `.ttdb` format
|
||||
|
||||
Designed to be **memory-mapped and read in place**: nothing is parsed on open and nothing
|
||||
is allocated per call, so opening a 3.4-million-call report costs a header validation. The
|
||||
reference implementations are `ttd/src/binreport.cpp` (writer) and, in the Binary Ninja
|
||||
debugger, `core/ttdbehavior.cpp` (reader).
|
||||
|
||||
Format version 2. All integers are **little-endian**. All *file offsets* are byte offsets
|
||||
from the start of the file, so a mapped view needs no fixups; *region offsets* (into the
|
||||
string table or the blob) are relative to that region's start, and are noted as such.
|
||||
|
||||
### Layout
|
||||
|
||||
```
|
||||
+----------------------------+ 0
|
||||
| header | 128 bytes
|
||||
+----------------------------+ callsOff
|
||||
| call records | callCount * 48 bytes, in sweep order
|
||||
+----------------------------+ paramsOff
|
||||
| parameter records | variable length, referenced by call records
|
||||
+----------------------------+ stringsOff
|
||||
| string table | stringsSize bytes, NUL-terminated UTF-8
|
||||
+----------------------------+ blobOff
|
||||
| blob | blobSize bytes: search text, strings, buffers
|
||||
+----------------------------+ end of file
|
||||
```
|
||||
|
||||
The split exists because the three kinds of data have different access patterns. Call
|
||||
records are fixed size so row *N* is a multiply. The string table is deduplicated, which is
|
||||
where the compression really comes from -- a trace has millions of calls but only thousands
|
||||
of distinct module and function names, so every name in a 3.4M-call report fits in about
|
||||
100 KB. Everything variable-length and unique to one call goes in the blob.
|
||||
|
||||
### Header (128 bytes)
|
||||
|
||||
| Offset | Type | Field | Notes |
|
||||
| ---: | --- | --- | --- |
|
||||
| 0 | `char[8]` | magic | `TTDBEHV1` |
|
||||
| 8 | `u32` | version | 2 |
|
||||
| 12 | `u32` | arch | 0 = x64, 1 = x86 |
|
||||
| 16 | `u64` | callCount | number of call records |
|
||||
| 24 | `u64` | paramBytes | size in bytes of the parameter region |
|
||||
| 32 | `u64` | pid | process id of the traced process |
|
||||
| 40 | `u64` | callsOff | file offset of the call records |
|
||||
| 48 | `u64` | paramsOff | file offset of the parameter records |
|
||||
| 56 | `u64` | stringsOff | file offset of the string table |
|
||||
| 64 | `u64` | stringsSize | size in bytes of the string table |
|
||||
| 72 | `u64` | blobOff | file offset of the blob |
|
||||
| 80 | `u64` | blobSize | size in bytes of the blob |
|
||||
| 88 | `u64` | decodedCount | calls whose parameters came from a real signature |
|
||||
| 96 | `u64` | maxSeq | highest sequence number, i.e. `callCount - 1` |
|
||||
| 104 | `u32` | tracePathStr | string-table offset of the source `.run` path |
|
||||
| 108 | `u32` | sampleNameStr | string-table offset of the main module's name |
|
||||
| 112 | `u32` | maxPositionChars | widest rendered position, for column sizing |
|
||||
| 116 | | *(reserved)* | zero |
|
||||
|
||||
`arch` describes the **traced process**, taken from the main module's PE format, not the
|
||||
machine that recorded the trace. A reader should reject a file whose magic does not match,
|
||||
whose version it does not know, or whose regions do not fit inside the file.
|
||||
|
||||
### Call record (48 bytes)
|
||||
|
||||
| Offset | Type | Field | Notes |
|
||||
| ---: | --- | --- | --- |
|
||||
| 0 | `u64` | ret | the call's return value |
|
||||
| 8 | `u32` | tid | TTD unique thread id |
|
||||
| 12 | `u32` | positionSequence | TTD position, sequence part |
|
||||
| 16 | `u32` | positionSteps | TTD position, steps part |
|
||||
| 20 | `u32` | moduleStr | string-table offset of the module name, e.g. `kernel32` |
|
||||
| 24 | `u32` | apiStr | string-table offset of the function name, e.g. `CreateFileW` |
|
||||
| 28 | `u32` | paramOff | offset into the parameter region; 0 when there are none |
|
||||
| 32 | `u32` | searchOff | offset into the blob of this call's search text |
|
||||
| 36 | `u16` | paramCount | low 15 bits; bit 15 (`0x8000`) set means *decoded* |
|
||||
| 38 | `u16` | searchLen | length in bytes of the search text |
|
||||
| 40 | `u64` | returnAddress | the instruction after the CALL, i.e. the call site |
|
||||
|
||||
**There is no sequence number field.** Recorded calls are numbered densely from zero in
|
||||
sweep order, so a record's index *is* its sequence number.
|
||||
|
||||
**Position** is conventionally rendered `%X:%X` of sequence and steps -- `45905:15A5` --
|
||||
which is the form WinDbg and the Binary Ninja debugger accept for time travel.
|
||||
|
||||
**`returnAddress`** identifies the caller, which is how you distinguish a call the sample
|
||||
made itself from one a system DLL made on its behalf.
|
||||
|
||||
**The decoded bit** matters for interpreting the parameters. When set, the extractor had a
|
||||
real signature for the function from Microsoft's Win32 metadata, so the parameters have
|
||||
correct arity, names, and types. When clear, it fell back to capturing argument registers
|
||||
(x64) or stack slots (x86) heuristically: the values are positional, unnamed, and the count
|
||||
is a guess. A decoded call may legitimately have zero parameters, which is why the flag is
|
||||
separate from the count.
|
||||
|
||||
### Parameter record (variable length)
|
||||
|
||||
Parameters for one call are stored consecutively starting at `paramsOff + paramOff`, and
|
||||
are decoded in sequence -- there is no index, so reading parameter *k* means walking the
|
||||
*k* before it. That is deliberate: a viewer only decodes parameters for rows a user
|
||||
actually looks at.
|
||||
|
||||
Fixed part, 18 bytes:
|
||||
|
||||
| Offset | Type | Field |
|
||||
| ---: | --- | --- |
|
||||
| 0 | `u8` | kind |
|
||||
| 1 | `u8` | bits |
|
||||
| 2 | `u32` | nameStr (string-table offset; 0 when unnamed) |
|
||||
| 6 | `u32` | typeStr (string-table offset; 0 when untyped) |
|
||||
| 10 | `u64` | value (the raw register or stack value) |
|
||||
|
||||
Then, in this order, only the parts whose bit is set in `bits`:
|
||||
|
||||
| Bit | Name | Adds |
|
||||
| ---: | --- | --- |
|
||||
| 0x01 | Out | *(nothing; marks an `[Out]` parameter)* |
|
||||
| 0x02 | AtReturn | *(nothing; value was re-read at the call's return)* |
|
||||
| 0x04 | HasDeref | `u64` deref -- the pointed-to value |
|
||||
| 0x08 | HasStr | `u32` strOff, `u32` strLen -- blob offset and length |
|
||||
| 0x10 | HasBytes | `u32` bytesOff, `u32` bytesLen, `u64` bytesTotal |
|
||||
| 0x20 | HasFlags | `u32` flagsOff, `u32` flagsLen -- `|`-joined names in the blob |
|
||||
|
||||
So a parameter's size is 18 bytes plus 8 for HasDeref, 8 for HasStr, 16 for HasBytes and 8
|
||||
for HasFlags, in that order.
|
||||
|
||||
**`bytesTotal`** is the buffer's real length when only a prefix was captured (the capture
|
||||
limit defaults to 64 KiB); `bytesLen` is what is actually in the file. `bytesTotal >
|
||||
bytesLen` means truncated.
|
||||
|
||||
**`AtReturn`** marks values re-read at the call's return position rather than at the call.
|
||||
This is what makes `[Out]` parameters renderable at all -- at the moment of the call they
|
||||
have not been written yet -- and it is the one thing a time-travel trace gives you that a
|
||||
live debugger cannot easily.
|
||||
|
||||
### Parameter kinds
|
||||
|
||||
`kind` is an index into this list. It describes how the value should be interpreted, not
|
||||
its C type, which is in `typeStr`.
|
||||
|
||||
| # | Name | Meaning |
|
||||
| ---: | --- | --- |
|
||||
| 0 | *(unknown)* | unclassified, or a heuristic capture with no signature |
|
||||
| 1 | `int` | plain scalar |
|
||||
| 2 | `bool` | |
|
||||
| 3 | `handle` | opaque, pointer-sized, never dereferenced |
|
||||
| 4 | `enum` | scalar with a symbolic value table; see HasFlags |
|
||||
| 5 | `float` | |
|
||||
| 6 | `double` | |
|
||||
| 7 | `str` | `char*`, NUL-terminated |
|
||||
| 8 | `wstr` | `wchar_t*`, NUL-terminated |
|
||||
| 9 | `strbuf` | `char[]`, length from a sibling parameter |
|
||||
| 10 | `wstrbuf` | `wchar_t[]` |
|
||||
| 11 | `buf` | `void*`/byte array |
|
||||
| 12 | `int*` | pointer to a scalar; see HasDeref |
|
||||
| 13 | `struct*` | pointer to a struct or union, not expanded |
|
||||
| 14 | `fnptr` | |
|
||||
| 15 | `guid` | pointer to a 16-byte GUID, rendered into HasStr |
|
||||
| 16 | `ptr` | opaque pointer |
|
||||
| 17 | `str*` | `char**`, an out-parameter receiving an allocated string |
|
||||
| 18 | `wstr*` | `wchar_t**` |
|
||||
|
||||
### String table
|
||||
|
||||
A run of NUL-terminated UTF-8 strings, deduplicated. A string-table offset is relative to
|
||||
`stringsOff`. **Offset 0 is the empty string**, so 0 doubles as "absent".
|
||||
|
||||
Only repeated text lives here: module names, function names, parameter names, parameter
|
||||
type names.
|
||||
|
||||
### Blob
|
||||
|
||||
Raw bytes, referenced by `(offset, length)` pairs relative to `blobOff`. It holds three
|
||||
things, interleaved in whatever order the writer emitted them:
|
||||
|
||||
- **Search text**, one per call, referenced by `searchOff`/`searchLen`. This is a
|
||||
lowercased rendering of `module!api` followed by each parameter as it would be displayed,
|
||||
including printable runs extracted from captured buffers. It exists so a text filter is a
|
||||
substring scan over mapped pages with nothing to build first.
|
||||
- **Parameter strings**, referenced by HasStr.
|
||||
- **Captured buffer contents**, referenced by HasBytes. Raw bytes, not hex.
|
||||
|
||||
### Reading a report
|
||||
|
||||
Validate the magic and version, check the regions fit in the file, and map it. Then:
|
||||
|
||||
- **Row *N***: read 48 bytes at `callsOff + N * 48`.
|
||||
- **Its module and function**: NUL-terminated strings at `stringsOff + moduleStr` and
|
||||
`stringsOff + apiStr`.
|
||||
- **Its parameters**: walk `paramCount & 0x7FFF` records from `paramsOff + paramOff`,
|
||||
decoding each per the bits.
|
||||
- **Filtering**: for a text match, `memmem` the needle in the `searchLen` bytes at
|
||||
`blobOff + searchOff`. For a match on module or function, resolve the name against the
|
||||
string table once to a set of offsets, then compare `moduleStr`/`apiStr` as integers --
|
||||
that is far cheaper than a string search, which is why scoped queries are faster than
|
||||
plain text ones.
|
||||
@@ -0,0 +1,641 @@
|
||||
"""
|
||||
Flatten the win32json Win32 API metadata into a compact binary index that
|
||||
ttdcapa-extract loads at startup to decode call parameters.
|
||||
|
||||
The metadata (https://github.com/marlersoft/win32json, vendored as the `win32json`
|
||||
submodule) is ~66 MB of JSON across ~300 files. Parsing that at debug time would be
|
||||
absurd, so we pre-bake it once: resolve the type graph, classify every parameter
|
||||
into an x64 ABI slot plus a small "decode kind" the C++ side can switch on, and
|
||||
write a single mmap-friendly blob.
|
||||
|
||||
python tools/build-win32-index.py [win32json/api] [-o ttd/data/win32-index.bin]
|
||||
|
||||
Regenerate whenever the win32json submodule is bumped. See
|
||||
WIN32JSON-TTD-INTEGRATION-NOTES.md for what the metadata does and does not give us
|
||||
(short version: types and semantics yes, ABI classification is ours -- that's the
|
||||
`SLOT_*` / `classify_param` logic below).
|
||||
"""
|
||||
import io
|
||||
import os
|
||||
import sys
|
||||
import json
|
||||
import glob
|
||||
import struct
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
MAGIC = b"W32IDX01"
|
||||
FORMAT_VERSION = 1
|
||||
|
||||
# --- decode kinds; keep in sync with ArgKind in ttd/src/win32meta.hpp ----------
|
||||
K_UNKNOWN = 0
|
||||
K_INTEGER = 1
|
||||
K_BOOL = 2
|
||||
K_HANDLE = 3
|
||||
K_ENUM = 4
|
||||
K_FLOAT = 5
|
||||
K_DOUBLE = 6
|
||||
K_ANSI_STRING = 7
|
||||
K_WIDE_STRING = 8
|
||||
K_ANSI_BUFFER = 9
|
||||
K_WIDE_BUFFER = 10
|
||||
K_BYTE_BUFFER = 11
|
||||
K_PTR_TO_INT = 12
|
||||
K_STRUCT_PTR = 13
|
||||
K_FUNC_PTR = 14
|
||||
K_GUID = 15
|
||||
K_POINTER = 16
|
||||
K_PTR_TO_ANSI_STRING = 17
|
||||
K_PTR_TO_WIDE_STRING = 18
|
||||
|
||||
# --- parameter attribute bits; keep in sync with ParamAttr in win32meta.hpp ---
|
||||
A_IN = 0x01
|
||||
A_OUT = 0x02
|
||||
A_OPTIONAL = 0x04
|
||||
A_CONST = 0x08
|
||||
A_RESERVED = 0x10
|
||||
A_NOT_NUL_TERM = 0x20
|
||||
A_NULNUL_TERM = 0x40
|
||||
A_COM_OUT_PTR = 0x80
|
||||
|
||||
# --- how a buffer's length is determined; keep in sync with AuxKind -----------
|
||||
AUX_NONE = 0
|
||||
AUX_BYTES_FROM_PARAM = 1
|
||||
AUX_COUNT_FROM_PARAM = 2
|
||||
AUX_COUNT_CONST = 3
|
||||
|
||||
# --- function flags; keep in sync with FuncFlag ------------------------------
|
||||
F_HIDDEN_RET_PTR = 0x01
|
||||
F_UNSUPPORTED = 0x02
|
||||
F_SET_LAST_ERROR = 0x04
|
||||
|
||||
NO_ENUM = 0xFFFFFFFF
|
||||
|
||||
# x64 sizes for the metadata's primitive names
|
||||
NATIVE_SIZE = {
|
||||
"Byte": 1, "SByte": 1, "Boolean": 1,
|
||||
"Int16": 2, "UInt16": 2, "Char": 2,
|
||||
"Int32": 4, "UInt32": 4, "Single": 4,
|
||||
"Int64": 8, "UInt64": 8, "Double": 8, "IntPtr": 8, "UIntPtr": 8,
|
||||
"Guid": 16,
|
||||
"Void": 0,
|
||||
}
|
||||
NATIVE_INT = {
|
||||
"Byte", "SByte", "Boolean", "Int16", "UInt16", "Char",
|
||||
"Int32", "UInt32", "Int64", "UInt64", "IntPtr", "UIntPtr",
|
||||
}
|
||||
|
||||
# NativeTypedefs that are really strings, not opaque handles. BSTR carries a
|
||||
# FreeFunc (SysFreeString) so the handle heuristic below would otherwise claim it.
|
||||
STRING_TYPEDEFS = {"PSTR": K_ANSI_STRING, "PWSTR": K_WIDE_STRING, "BSTR": K_WIDE_STRING}
|
||||
|
||||
MAX_PARAMS = 255 # param_count is a u8; nothing real comes close (max observed: 18)
|
||||
|
||||
|
||||
class StringTable:
|
||||
"""Deduplicating NUL-terminated string pool. Offset 0 is always ""."""
|
||||
|
||||
def __init__(self):
|
||||
self.buf = bytearray(b"\x00")
|
||||
self.offsets = {"": 0}
|
||||
|
||||
def add(self, s):
|
||||
if s is None:
|
||||
s = ""
|
||||
off = self.offsets.get(s)
|
||||
if off is None:
|
||||
off = len(self.buf)
|
||||
self.buf += s.encode("utf-8") + b"\x00"
|
||||
self.offsets[s] = off
|
||||
return off
|
||||
|
||||
|
||||
class Metadata:
|
||||
"""The win32json type graph, plus the resolution helpers built on top of it."""
|
||||
|
||||
def __init__(self, api_dir):
|
||||
self.types = {} # (api, name) -> type dict
|
||||
self.functions = [] # (api, function dict)
|
||||
self.unicode_aliases = set()
|
||||
files = sorted(glob.glob(os.path.join(api_dir, "*.json")))
|
||||
if not files:
|
||||
sys.exit(f"no .json files under {api_dir}; is the win32json submodule checked out?")
|
||||
for path in files:
|
||||
api = os.path.basename(path)[:-5]
|
||||
with open(path, encoding="utf-8") as f:
|
||||
doc = json.load(f)
|
||||
for t in doc.get("Types", []):
|
||||
# first definition wins; arch-specific duplicates are handled by
|
||||
# SupportedArchitecture, which we resolve to x64 at index time
|
||||
self.types.setdefault((api, t["Name"]), t)
|
||||
for fn in doc.get("Functions", []):
|
||||
self.functions.append((api, fn))
|
||||
self.unicode_aliases.update(doc.get("UnicodeAliases", []))
|
||||
self._size_cache = {}
|
||||
|
||||
def lookup(self, ref):
|
||||
return self.types.get((ref.get("Api"), ref.get("Name")))
|
||||
|
||||
def resolve(self, t, depth=0):
|
||||
"""Follow ApiRef -> NativeTypedef chains to a terminal type node.
|
||||
|
||||
Returns (node, typedef) where `typedef` is the last NativeTypedef we passed
|
||||
through (or None). Callers need the typedef to spot HANDLE/PSTR, which are
|
||||
only distinguishable by name -- their definitions are plain IntPtr/Byte*.
|
||||
"""
|
||||
seen = set()
|
||||
typedef = None
|
||||
while depth < 16:
|
||||
depth += 1
|
||||
if t.get("Kind") != "ApiRef":
|
||||
return t, typedef
|
||||
key = (t.get("Api"), t.get("Name"))
|
||||
if key in seen:
|
||||
return t, typedef
|
||||
seen.add(key)
|
||||
target = self.lookup(t)
|
||||
if target is None:
|
||||
return t, typedef
|
||||
if target.get("Kind") == "NativeTypedef":
|
||||
typedef = target
|
||||
t = target["Def"]
|
||||
continue
|
||||
return target, typedef
|
||||
return t, typedef
|
||||
|
||||
# --- sizeof, needed only to classify by-value aggregates ------------------
|
||||
|
||||
def sizeof(self, t, depth=0):
|
||||
"""x64 size in bytes, or None if it can't be determined.
|
||||
|
||||
Only aggregates passed/returned by value need this (236 params and 18
|
||||
return types across the whole surface), so a None here costs us one
|
||||
function, not correctness everywhere.
|
||||
"""
|
||||
if depth > 24:
|
||||
return None
|
||||
kind = t.get("Kind")
|
||||
if kind == "Native":
|
||||
return NATIVE_SIZE.get(t.get("Name"))
|
||||
if kind in ("PointerTo", "LPArray", "FunctionPointer"):
|
||||
return 8
|
||||
if kind == "Array":
|
||||
child = self.sizeof(t["Child"], depth + 1)
|
||||
count = (t.get("Shape") or {}).get("Size")
|
||||
if child is None or not count:
|
||||
return None
|
||||
return child * count
|
||||
if kind == "ApiRef":
|
||||
node, _ = self.resolve(t)
|
||||
if node.get("Kind") == "ApiRef":
|
||||
return None # unresolvable reference
|
||||
return self.sizeof(node, depth + 1)
|
||||
if kind == "Com":
|
||||
return 8
|
||||
if kind == "Enum":
|
||||
return self.enum_width(t)
|
||||
if kind in ("Struct", "Union"):
|
||||
return self._sizeof_record(t, depth)
|
||||
return None
|
||||
|
||||
def _sizeof_record(self, t, depth):
|
||||
key = id(t)
|
||||
cached = self._size_cache.get(key)
|
||||
if cached is not None:
|
||||
return cached[0]
|
||||
self._size_cache[key] = (None,) # cycle guard
|
||||
declared = t.get("Size") or 0
|
||||
if declared:
|
||||
self._size_cache[key] = (declared,)
|
||||
return declared
|
||||
pack = t.get("PackingSize") or 0
|
||||
offset = 0
|
||||
max_align = 1
|
||||
for field in t.get("Fields", []):
|
||||
fsize = self.sizeof(field["Type"], depth + 1)
|
||||
if fsize is None:
|
||||
self._size_cache[key] = (None,)
|
||||
return None
|
||||
align = min(fsize if fsize in (1, 2, 4, 8, 16) else 8, pack) if pack else fsize
|
||||
align = max(1, min(align if align in (1, 2, 4, 8, 16) else 8, 8))
|
||||
max_align = max(max_align, align)
|
||||
if t["Kind"] == "Union":
|
||||
offset = max(offset, fsize)
|
||||
else:
|
||||
offset = (offset + align - 1) // align * align + fsize
|
||||
size = (offset + max_align - 1) // max_align * max_align if offset else 0
|
||||
self._size_cache[key] = (size,)
|
||||
return size
|
||||
|
||||
def enum_width(self, enum_type):
|
||||
node = enum_type
|
||||
if node.get("Kind") == "ApiRef":
|
||||
node, _ = self.resolve(node)
|
||||
integer_base = node.get("IntegerBase")
|
||||
if integer_base:
|
||||
return NATIVE_SIZE.get(integer_base, 4)
|
||||
return 4
|
||||
|
||||
|
||||
def attr_bits(attrs):
|
||||
"""Fold a param's Attrs list into a bitmask plus its MemorySize source, if any."""
|
||||
bits = 0
|
||||
bytes_param = None
|
||||
for a in attrs:
|
||||
if isinstance(a, dict):
|
||||
if a.get("Kind") == "MemorySize":
|
||||
bytes_param = a.get("BytesParamIndex")
|
||||
continue
|
||||
bits |= {
|
||||
"In": A_IN, "Out": A_OUT, "Optional": A_OPTIONAL, "Const": A_CONST,
|
||||
"Reserved": A_RESERVED, "NotNullTerminated": A_NOT_NUL_TERM,
|
||||
"NullNullTerminated": A_NULNUL_TERM, "ComOutPtr": A_COM_OUT_PTR,
|
||||
}.get(a, 0)
|
||||
return bits, bytes_param
|
||||
|
||||
|
||||
class Param:
|
||||
__slots__ = ("name", "type_name", "kind", "attrs", "slot",
|
||||
"aux_kind", "aux_value", "enum_idx", "pointee_size")
|
||||
|
||||
def __init__(self):
|
||||
self.name = ""
|
||||
self.type_name = ""
|
||||
self.kind = K_UNKNOWN
|
||||
self.attrs = 0
|
||||
self.slot = 0
|
||||
self.aux_kind = AUX_NONE
|
||||
self.aux_value = 0
|
||||
self.enum_idx = NO_ENUM
|
||||
self.pointee_size = 0
|
||||
|
||||
|
||||
def type_display_name(t):
|
||||
"""A short human-readable name for the report/timeline, e.g. "PWSTR", "void*"."""
|
||||
kind = t.get("Kind")
|
||||
if kind == "ApiRef":
|
||||
return t.get("Name", "")
|
||||
if kind == "Native":
|
||||
return t.get("Name", "")
|
||||
if kind == "PointerTo":
|
||||
return type_display_name(t["Child"]) + "*"
|
||||
if kind == "LPArray":
|
||||
return type_display_name(t["Child"]) + "[]"
|
||||
if kind == "FunctionPointer":
|
||||
return "fnptr"
|
||||
return kind or ""
|
||||
|
||||
|
||||
def classify_pointee(md, child):
|
||||
"""Classify what a pointer points at -> (kind, pointee_size).
|
||||
|
||||
Split out because PointerTo and LPArray share it.
|
||||
"""
|
||||
node, typedef = md.resolve(child)
|
||||
if typedef is not None and typedef["Name"] in STRING_TYPEDEFS:
|
||||
# e.g. PWSTR* -- an out-parameter that receives an allocated string
|
||||
return ({K_ANSI_STRING: K_PTR_TO_ANSI_STRING,
|
||||
K_WIDE_STRING: K_PTR_TO_WIDE_STRING}[STRING_TYPEDEFS[typedef["Name"]]], 8)
|
||||
|
||||
kind = node.get("Kind")
|
||||
if kind == "Native":
|
||||
name = node.get("Name")
|
||||
if name in ("Byte", "SByte"):
|
||||
return K_ANSI_BUFFER, 1
|
||||
if name == "Char":
|
||||
return K_WIDE_BUFFER, 2
|
||||
if name == "Void":
|
||||
return K_POINTER, 0
|
||||
if name == "Guid":
|
||||
return K_GUID, 16
|
||||
if name in NATIVE_INT:
|
||||
return K_PTR_TO_INT, NATIVE_SIZE[name]
|
||||
if name in ("Single", "Double"):
|
||||
return K_PTR_TO_INT, NATIVE_SIZE[name]
|
||||
return K_POINTER, 0
|
||||
if kind == "Enum":
|
||||
return K_PTR_TO_INT, md.enum_width(node)
|
||||
if kind in ("Struct", "Union", "Com"):
|
||||
return K_STRUCT_PTR, md.sizeof(node) or 0
|
||||
if kind == "FunctionPointer":
|
||||
return K_FUNC_PTR, 8
|
||||
if kind in ("PointerTo", "LPArray"):
|
||||
return K_PTR_TO_INT, 8 # void** / T** -- at least surface the inner pointer
|
||||
return K_POINTER, 0
|
||||
|
||||
|
||||
def classify_param(md, p, enum_ids):
|
||||
"""Map one metadata parameter onto a decode kind + buffer-length source.
|
||||
|
||||
Returns a Param with everything except `slot` filled in (slot assignment needs
|
||||
whole-signature context and happens in build_function).
|
||||
"""
|
||||
out = Param()
|
||||
out.name = p.get("Name") or ""
|
||||
t = p["Type"]
|
||||
out.type_name = type_display_name(t)
|
||||
out.attrs, bytes_param = attr_bits(p.get("Attrs") or [])
|
||||
|
||||
kind = t.get("Kind")
|
||||
|
||||
if kind == "LPArray":
|
||||
out.kind, out.pointee_size = classify_pointee(md, t["Child"])
|
||||
if out.kind == K_POINTER:
|
||||
out.kind = K_BYTE_BUFFER
|
||||
count_param = t.get("CountParamIndex", -1)
|
||||
count_const = t.get("CountConst", -1)
|
||||
if count_param is not None and count_param >= 0:
|
||||
out.aux_kind, out.aux_value = AUX_COUNT_FROM_PARAM, count_param
|
||||
elif count_const is not None and count_const >= 0:
|
||||
out.aux_kind, out.aux_value = AUX_COUNT_CONST, count_const
|
||||
|
||||
elif kind == "PointerTo":
|
||||
out.kind, out.pointee_size = classify_pointee(md, t["Child"])
|
||||
|
||||
elif kind == "ApiRef":
|
||||
node, typedef = md.resolve(t)
|
||||
if typedef is not None and typedef["Name"] in STRING_TYPEDEFS:
|
||||
out.kind = STRING_TYPEDEFS[typedef["Name"]]
|
||||
out.pointee_size = 1 if out.kind == K_ANSI_STRING else 2
|
||||
elif typedef is not None and is_handle_typedef(typedef):
|
||||
# an opaque kernel/GDI/etc handle: pointer-sized, never dereference it
|
||||
out.kind, out.pointee_size = K_HANDLE, 8
|
||||
else:
|
||||
nkind = node.get("Kind")
|
||||
if nkind == "Enum":
|
||||
out.kind = K_ENUM
|
||||
out.pointee_size = md.enum_width(node)
|
||||
enum_key = (node.get("__api__"), node["Name"])
|
||||
out.enum_idx = enum_ids.get(enum_key, NO_ENUM)
|
||||
elif nkind == "Native":
|
||||
out.kind, out.pointee_size = classify_native(node)
|
||||
elif nkind in ("Struct", "Union"):
|
||||
size = md.sizeof(node)
|
||||
if size in (1, 2, 4, 8):
|
||||
out.kind, out.pointee_size = K_INTEGER, size
|
||||
elif size is None:
|
||||
out.kind = K_UNKNOWN
|
||||
else:
|
||||
# x64: aggregates that aren't 1/2/4/8 bytes go by hidden pointer
|
||||
out.kind, out.pointee_size = K_STRUCT_PTR, size
|
||||
elif nkind == "Com":
|
||||
out.kind, out.pointee_size = K_POINTER, 8
|
||||
elif nkind == "FunctionPointer":
|
||||
out.kind, out.pointee_size = K_FUNC_PTR, 8
|
||||
elif nkind == "PointerTo":
|
||||
out.kind, out.pointee_size = classify_pointee(md, node["Child"])
|
||||
else:
|
||||
out.kind = K_UNKNOWN
|
||||
|
||||
elif kind == "Native":
|
||||
out.kind, out.pointee_size = classify_native(t)
|
||||
if out.kind == K_GUID:
|
||||
out.kind = K_STRUCT_PTR # 16 bytes by value -> hidden pointer on x64
|
||||
|
||||
elif kind == "FunctionPointer":
|
||||
out.kind, out.pointee_size = K_FUNC_PTR, 8
|
||||
|
||||
else:
|
||||
out.kind = K_UNKNOWN
|
||||
|
||||
# a MemorySize attribute always wins: it says this really is a sized buffer
|
||||
if bytes_param is not None and bytes_param >= 0:
|
||||
out.aux_kind, out.aux_value = AUX_BYTES_FROM_PARAM, bytes_param
|
||||
if out.kind in (K_POINTER, K_UNKNOWN, K_STRUCT_PTR):
|
||||
out.kind = K_BYTE_BUFFER
|
||||
out.pointee_size = 1
|
||||
|
||||
return out
|
||||
|
||||
|
||||
def is_handle_typedef(typedef):
|
||||
"""Is this NativeTypedef an opaque handle we must never dereference?
|
||||
|
||||
RAIIFree/InvalidHandleValue mark most of them (HANDLE, HKEY, ...). The rest are
|
||||
pointer-sized H-prefixed typedefs with no lifetime metadata (HWND, HDESK, ...);
|
||||
treating those as integers would be harmless but renders worse.
|
||||
"""
|
||||
if typedef.get("FreeFunc") or typedef.get("InvalidHandleValue") is not None:
|
||||
return True
|
||||
name = typedef["Name"]
|
||||
return (
|
||||
name.startswith("H")
|
||||
and typedef["Def"].get("Kind") == "Native"
|
||||
and typedef["Def"].get("Name") in ("IntPtr", "UIntPtr")
|
||||
)
|
||||
|
||||
|
||||
def classify_native(t):
|
||||
name = t.get("Name")
|
||||
if name == "Single":
|
||||
return K_FLOAT, 4
|
||||
if name == "Double":
|
||||
return K_DOUBLE, 8
|
||||
if name == "Boolean":
|
||||
return K_BOOL, 1
|
||||
if name == "Guid":
|
||||
return K_GUID, 16
|
||||
if name == "Void":
|
||||
return K_UNKNOWN, 0
|
||||
if name in NATIVE_INT:
|
||||
return K_INTEGER, NATIVE_SIZE[name]
|
||||
return K_UNKNOWN, 0
|
||||
|
||||
|
||||
def wants_x64(arches):
|
||||
"""Architectures==[] means arch-neutral; otherwise it must list X64."""
|
||||
return not arches or "X64" in arches
|
||||
|
||||
|
||||
def build_function(md, api, fn, enum_ids):
|
||||
"""Produce (params, flags) for one function, or None if it isn't x64-relevant."""
|
||||
if not wants_x64(fn.get("Architectures") or []):
|
||||
return None
|
||||
|
||||
flags = F_SET_LAST_ERROR if fn.get("SetLastError") else 0
|
||||
|
||||
# A function returning an aggregate that isn't 1/2/4/8 bytes takes a hidden
|
||||
# pointer in RCX, pushing every real parameter one slot to the right.
|
||||
ret = fn.get("ReturnType") or {}
|
||||
ret_node, _ = md.resolve(ret) if ret.get("Kind") == "ApiRef" else (ret, None)
|
||||
base_slot = 0
|
||||
if ret_node.get("Kind") in ("Struct", "Union") or (
|
||||
ret_node.get("Kind") == "Native" and ret_node.get("Name") == "Guid"
|
||||
):
|
||||
size = md.sizeof(ret_node)
|
||||
if size is None:
|
||||
flags |= F_UNSUPPORTED
|
||||
elif size not in (1, 2, 4, 8):
|
||||
flags |= F_HIDDEN_RET_PTR
|
||||
base_slot = 1
|
||||
|
||||
raw_params = fn.get("Params") or []
|
||||
if len(raw_params) + base_slot > MAX_PARAMS:
|
||||
return None
|
||||
|
||||
params = []
|
||||
for i, p in enumerate(raw_params):
|
||||
cp = classify_param(md, p, enum_ids)
|
||||
cp.slot = base_slot + i
|
||||
if cp.kind == K_UNKNOWN:
|
||||
flags |= F_UNSUPPORTED
|
||||
params.append(cp)
|
||||
|
||||
# aux_value indexes into the *parameter* list; the C++ side reads captured
|
||||
# values by parameter index too, so no slot translation is needed. Drop
|
||||
# references that point outside the list rather than trusting them at runtime.
|
||||
for cp in params:
|
||||
if cp.aux_kind in (AUX_BYTES_FROM_PARAM, AUX_COUNT_FROM_PARAM):
|
||||
if not (0 <= cp.aux_value < len(params)):
|
||||
cp.aux_kind, cp.aux_value = AUX_NONE, 0
|
||||
|
||||
return params, flags
|
||||
|
||||
|
||||
def collect_enums(md):
|
||||
"""Index only the enums a parameter can actually reference (950 of 7005)."""
|
||||
referenced = {}
|
||||
|
||||
def visit(t, depth=0):
|
||||
if depth > 16:
|
||||
return
|
||||
kind = t.get("Kind")
|
||||
if kind == "ApiRef":
|
||||
node, _ = md.resolve(t)
|
||||
if node.get("Kind") == "Enum":
|
||||
referenced.setdefault((t.get("Api"), t.get("Name")), node)
|
||||
elif kind in ("PointerTo", "LPArray"):
|
||||
visit(t["Child"], depth + 1)
|
||||
|
||||
for api, fn in md.functions:
|
||||
for p in fn.get("Params") or []:
|
||||
visit(p["Type"])
|
||||
|
||||
enums = []
|
||||
enum_ids = {}
|
||||
for (eapi, ename), node in sorted(referenced.items()):
|
||||
node = dict(node)
|
||||
node["__api__"] = eapi
|
||||
enum_ids[(eapi, ename)] = len(enums)
|
||||
enums.append((eapi, ename, node))
|
||||
return enums, enum_ids
|
||||
|
||||
|
||||
def resolve_enum_index(md, t, enum_ids, depth=0):
|
||||
"""The enum table index a param type refers to, or NO_ENUM."""
|
||||
if depth > 16:
|
||||
return NO_ENUM
|
||||
kind = t.get("Kind")
|
||||
if kind == "ApiRef":
|
||||
node, _ = md.resolve(t)
|
||||
if node.get("Kind") == "Enum":
|
||||
return enum_ids.get((t.get("Api"), t.get("Name")), NO_ENUM)
|
||||
elif kind in ("PointerTo", "LPArray"):
|
||||
return resolve_enum_index(md, t["Child"], enum_ids, depth + 1)
|
||||
return NO_ENUM
|
||||
|
||||
|
||||
def build(api_dir, out_path):
|
||||
print(f"[+] loading {api_dir} ...")
|
||||
md = Metadata(api_dir)
|
||||
print(f"[+] {len(md.functions)} functions, {len(md.types)} types")
|
||||
|
||||
enums, enum_ids = collect_enums(md)
|
||||
print(f"[+] {len(enums)} referenced enums")
|
||||
|
||||
strtab = StringTable()
|
||||
|
||||
# enum tables
|
||||
enum_recs = []
|
||||
enumval_recs = []
|
||||
for eapi, ename, node in enums:
|
||||
values = node.get("Values") or []
|
||||
val_off = len(enumval_recs)
|
||||
for v in values:
|
||||
enumval_recs.append((strtab.add(v["Name"]), int(v["Value"])))
|
||||
enum_recs.append((
|
||||
strtab.add(ename), val_off, len(values),
|
||||
1 if node.get("Flags") else 0, md.enum_width(node),
|
||||
))
|
||||
|
||||
# functions, keyed by bare export name (module is a weak hint only -- API sets
|
||||
# mean kernel32!CreateFileW surfaces as KERNELBASE!CreateFileW in a trace)
|
||||
func_recs = []
|
||||
param_recs = []
|
||||
seen_names = {}
|
||||
skipped_arch = 0
|
||||
unsupported = 0
|
||||
collisions = 0
|
||||
|
||||
for api, fn in sorted(md.functions, key=lambda x: x[1]["Name"]):
|
||||
name = fn["Name"]
|
||||
built = build_function(md, api, fn, enum_ids)
|
||||
if built is None:
|
||||
skipped_arch += 1
|
||||
continue
|
||||
params, flags = built
|
||||
|
||||
if name in seen_names:
|
||||
prior = seen_names[name]
|
||||
if prior != tuple((p.kind, p.slot) for p in params):
|
||||
collisions += 1
|
||||
continue
|
||||
seen_names[name] = tuple((p.kind, p.slot) for p in params)
|
||||
|
||||
for i, p in enumerate(params):
|
||||
enum_idx = resolve_enum_index(md, fn["Params"][i]["Type"], enum_ids)
|
||||
param_recs.append((
|
||||
strtab.add(p.name), strtab.add(p.type_name), p.kind, p.attrs,
|
||||
p.slot, p.aux_kind, p.aux_value, enum_idx, p.pointee_size,
|
||||
))
|
||||
if flags & F_UNSUPPORTED:
|
||||
unsupported += 1
|
||||
func_recs.append((
|
||||
strtab.add(name), strtab.add(fn.get("DllImport") or ""),
|
||||
len(param_recs) - len(params), len(params), flags,
|
||||
))
|
||||
|
||||
print(f"[+] indexed {len(func_recs)} functions "
|
||||
f"({skipped_arch} not x64, {unsupported} flagged UNSUPPORTED, "
|
||||
f"{collisions} name collisions with differing shapes)")
|
||||
print(f"[+] {len(param_recs)} params, {len(enumval_recs)} enum values, "
|
||||
f"{len(strtab.buf)} bytes of strings")
|
||||
|
||||
blob = io.BytesIO()
|
||||
blob.write(MAGIC)
|
||||
blob.write(struct.pack(
|
||||
"<IIIIII", FORMAT_VERSION, len(func_recs), len(param_recs),
|
||||
len(enum_recs), len(enumval_recs), len(strtab.buf),
|
||||
))
|
||||
for name_off, dll_off, param_off, count, flags in func_recs:
|
||||
blob.write(struct.pack("<IIIBBH", name_off, dll_off, param_off, count, flags, 0))
|
||||
for rec in param_recs:
|
||||
name_off, type_off, kind, attrs, slot, aux_kind, aux_value, enum_idx, pointee = rec
|
||||
blob.write(struct.pack("<IIBBBBiIH2x", name_off, type_off, kind, attrs,
|
||||
slot, aux_kind, aux_value, enum_idx, min(pointee, 0xFFFF)))
|
||||
for name_off, val_off, val_count, is_flags, width in enum_recs:
|
||||
blob.write(struct.pack("<IIIBB2x", name_off, val_off, val_count, is_flags, width))
|
||||
for name_off, value in enumval_recs:
|
||||
blob.write(struct.pack("<I4xq", name_off, value))
|
||||
blob.write(strtab.buf)
|
||||
|
||||
out_path = Path(out_path)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_bytes(blob.getvalue())
|
||||
print(f"[+] wrote {out_path} ({out_path.stat().st_size / 1024 / 1024:.2f} MB)")
|
||||
return 0
|
||||
|
||||
|
||||
def main(argv):
|
||||
here = Path(__file__).resolve().parent.parent
|
||||
ap = argparse.ArgumentParser(description=__doc__.splitlines()[1])
|
||||
ap.add_argument("api_dir", nargs="?", default=str(here / "win32json" / "api"),
|
||||
help="path to win32json/api")
|
||||
ap.add_argument("-o", "--output", default=str(here / "ttd" / "data" / "win32-index.bin"),
|
||||
help="output blob path")
|
||||
args = ap.parse_args(argv)
|
||||
return build(args.api_dir, args.output)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv[1:]))
|
||||
+23
-19
@@ -1,16 +1,3 @@
|
||||
#!/usr/bin/env python3
|
||||
# Copyright 2024 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
|
||||
"""
|
||||
One-command CAPA-over-TTD wrapper.
|
||||
|
||||
@@ -46,10 +33,8 @@ def find_extractor(explicit: str | None) -> Path:
|
||||
# search common build output locations relative to this script
|
||||
candidates = [
|
||||
HERE / "ttdcapa-extract.exe",
|
||||
HERE / "build" / "ttdcapa-extract.exe",
|
||||
HERE / "ttd" / "bin" / "x64" / "Release" / "ttdcapa-extract.exe",
|
||||
HERE / "ttd" / "bin" / "x64" / "Debug" / "ttdcapa-extract.exe",
|
||||
HERE / "ttd" / "ttdcapa-extract.exe",
|
||||
HERE / "ttd" / "bin" / "x64" / "Debug" / "ttdcapa-extract.exe"
|
||||
]
|
||||
for c in candidates:
|
||||
if c.is_file():
|
||||
@@ -75,7 +60,19 @@ def main(argv: list[str]) -> int:
|
||||
parser.add_argument("--sample", help="optional on-disk sample for accurate hashes")
|
||||
parser.add_argument("--extractor", help="path to ttdcapa-extract.exe")
|
||||
parser.add_argument("--max-calls", type=int, help="cap recorded API calls (for huge traces)")
|
||||
parser.add_argument("--with-stack-args", action="store_true", help="also capture stack args 5+")
|
||||
parser.add_argument(
|
||||
"--with-stack-args",
|
||||
action="store_true",
|
||||
help="for calls with no Win32 metadata, also grab four stack slots past the "
|
||||
"register args (calls we have a signature for always capture their true arity)",
|
||||
)
|
||||
parser.add_argument("--win32-index", help="path to win32-index.bin (default: next to the extractor)")
|
||||
parser.add_argument(
|
||||
"--no-metadata",
|
||||
action="store_true",
|
||||
help="disable metadata-driven argument decoding entirely",
|
||||
)
|
||||
parser.add_argument("--max-buffer", type=int, help="bytes kept from any one captured buffer (default 256)")
|
||||
parser.add_argument("--keep-json", action="store_true", help="keep the intermediate TTD report")
|
||||
parser.add_argument("--python", default=sys.executable, help="python interpreter to run capa")
|
||||
args, capa_extra = parser.parse_known_args(argv)
|
||||
@@ -86,9 +83,10 @@ def main(argv: list[str]) -> int:
|
||||
trace = Path(args.trace)
|
||||
if not trace.is_file():
|
||||
sys.exit(f"trace not found: {trace}")
|
||||
rules = Path(args.rules)
|
||||
# absolute, because capa is invoked with cwd set to the report's directory
|
||||
rules = Path(args.rules).resolve()
|
||||
if not rules.exists():
|
||||
sys.exit(f"rules path not found: {rules}")
|
||||
sys.exit(f"rules path not found: {args.rules}")
|
||||
|
||||
extractor = find_extractor(args.extractor)
|
||||
|
||||
@@ -104,6 +102,12 @@ def main(argv: list[str]) -> int:
|
||||
extract_cmd += ["--max-calls", str(args.max_calls)]
|
||||
if args.with_stack_args:
|
||||
extract_cmd += ["--with-stack-args"]
|
||||
if args.win32_index:
|
||||
extract_cmd += ["--win32-index", args.win32_index]
|
||||
if args.no_metadata:
|
||||
extract_cmd += ["--no-metadata"]
|
||||
if args.max_buffer:
|
||||
extract_cmd += ["--max-buffer", str(args.max_buffer)]
|
||||
|
||||
print(f"[ttd-capa] extracting: {' '.join(extract_cmd)}", file=sys.stderr)
|
||||
rc = subprocess.call(extract_cmd)
|
||||
|
||||
+39
-14
@@ -1,16 +1,3 @@
|
||||
#!/usr/bin/env python3
|
||||
# Copyright 2024 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
|
||||
"""
|
||||
Print a chronological "timeline" of what the TTD extractor uncovered, ordered by
|
||||
TTD trace position so you can step through it in WinDbg.
|
||||
@@ -41,6 +28,20 @@ from typing import Optional
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
|
||||
# Recovered guest strings are arbitrary text -- non-Latin paths, HTTP bodies, the
|
||||
# occasional lone surrogate from a half-written UTF-16 buffer. When stdout is a
|
||||
# console Python uses UTF-8, but when it is redirected to a file it falls back to
|
||||
# the locale encoding (cp1252 here), which cannot encode any of that and aborts the
|
||||
# whole run. Pin UTF-8 and degrade unencodable characters instead of dying.
|
||||
for _stream in (sys.stdout, sys.stderr):
|
||||
if hasattr(_stream, "reconfigure"):
|
||||
_stream.reconfigure(encoding="utf-8", errors="replace")
|
||||
|
||||
# Python puts this script's directory first on sys.path, and the capa fork lives in
|
||||
# a `capa/` subdirectory of it -- which shadows the installed `capa` package with a
|
||||
# namespace package that has no submodules. Drop it; nothing here is imported locally.
|
||||
sys.path[:] = [p for p in sys.path if p and Path(p).resolve() != HERE]
|
||||
|
||||
from capa.features.address import ProcessAddress, ThreadAddress, DynamicCallAddress # noqa: E402
|
||||
from capa.features.extractors.ttd.models import TtdCall # noqa: E402
|
||||
from capa.features.extractors.ttd.extractor import TtdExtractor # noqa: E402
|
||||
@@ -79,9 +80,33 @@ def fmt_arg(a) -> str:
|
||||
return repr(a)
|
||||
|
||||
|
||||
def fmt_param(p) -> str:
|
||||
"""One metadata-decoded parameter as `name=value`.
|
||||
|
||||
Prefers the most informative rendering the extractor managed: decoded text,
|
||||
then symbolic flag names, then a dereferenced pointee, then the raw value.
|
||||
"""
|
||||
if p.str_ is not None:
|
||||
value = repr(p.str_)
|
||||
elif p.flags:
|
||||
value = "|".join(p.flags)
|
||||
elif p.float_ is not None:
|
||||
value = repr(p.float_)
|
||||
else:
|
||||
value = fmt_arg(p.value)
|
||||
if p.deref is not None:
|
||||
value += f"->{fmt_arg(p.deref)}"
|
||||
if p.at_return:
|
||||
value += "@ret"
|
||||
return f"{p.name}={value}" if p.name else value
|
||||
|
||||
|
||||
def format_call(call: TtdCall) -> str:
|
||||
api = f"{call.module}.{call.api}" if call.module else call.api
|
||||
args = ", ".join(fmt_arg(a) for a in call.args)
|
||||
if call.params:
|
||||
args = ", ".join(fmt_param(p) for p in call.params)
|
||||
else:
|
||||
args = ", ".join(fmt_arg(a) for a in call.args)
|
||||
ret = "" if call.ret is None else f" -> 0x{call.ret & 0xFFFFFFFFFFFFFFFF:x}"
|
||||
return f"{api}({args}){ret}"
|
||||
|
||||
|
||||
Binary file not shown.
+613
@@ -0,0 +1,613 @@
|
||||
#include "abi.hpp"
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
|
||||
using ttdcapa::win32meta::ArgKind;
|
||||
using ttdcapa::win32meta::AuxKind;
|
||||
|
||||
namespace ttdcapa {
|
||||
namespace {
|
||||
|
||||
// A dereference is only worth attempting above the first 64 KiB; the null
|
||||
// page and its neighbours are never mapped in user mode.
|
||||
constexpr uint64_t kMinDerefAddr = 0x10000;
|
||||
|
||||
// Guards against a mis-typed count parameter turning into a huge read.
|
||||
constexpr uint64_t kMaxCountElements = 1u << 20;
|
||||
|
||||
// Every x86 stack argument occupies a whole number of 4-byte words.
|
||||
constexpr uint16_t kX86StackAlign = 4;
|
||||
|
||||
// Matches win32meta's MAX_PARAMS; the index never emits more.
|
||||
constexpr size_t kMaxParams = 32;
|
||||
|
||||
bool readGuest(TTD::Replay::IThreadView const* thread, uint64_t addr, void* dst, size_t size) {
|
||||
if (addr < kMinDerefAddr || size == 0) {
|
||||
return false;
|
||||
}
|
||||
auto result = thread->QueryMemoryBuffer(TTD::GuestAddress{ addr }, TTD::BufferView{ dst, size });
|
||||
return result.Memory.Size == size;
|
||||
}
|
||||
|
||||
// True for the kinds that are a pointer in the guest, whatever they point at.
|
||||
bool isPointerKind(ArgKind kind) {
|
||||
switch (kind) {
|
||||
case ArgKind::AnsiString:
|
||||
case ArgKind::WideString:
|
||||
case ArgKind::AnsiBuffer:
|
||||
case ArgKind::WideBuffer:
|
||||
case ArgKind::ByteBuffer:
|
||||
case ArgKind::PtrToInt:
|
||||
case ArgKind::StructPtr:
|
||||
case ArgKind::FuncPtr:
|
||||
case ArgKind::Guid:
|
||||
case ArgKind::Pointer:
|
||||
case ArgKind::PtrToAnsiString:
|
||||
case ArgKind::PtrToWideString:
|
||||
case ArgKind::Handle: // opaque but pointer-sized
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// An aggregate the x64 ABI passes by hidden pointer but x86 pushes by value.
|
||||
// The index cannot tell us its x86 footprint -- it records the x64 size, which
|
||||
// is wrong for any struct holding a pointer -- so its presence makes the whole
|
||||
// signature unlayoutable on x86. The display type is the discriminator: a real
|
||||
// pointer parameter renders as "OVERLAPPED*", a by-value one as "VARIANT".
|
||||
bool isByValueAggregate(const win32meta::ParamSig& p) {
|
||||
if (p.kind != ArgKind::StructPtr) {
|
||||
return false;
|
||||
}
|
||||
size_t len = p.type != nullptr ? std::strlen(p.type) : 0;
|
||||
return len == 0 || p.type[len - 1] != '*';
|
||||
}
|
||||
|
||||
// Bytes one parameter occupies on the x86 argument stack, or 0 if unknowable.
|
||||
uint16_t x86StackFootprint(const win32meta::ParamSig& p) {
|
||||
if (isByValueAggregate(p)) {
|
||||
return 0;
|
||||
}
|
||||
uint16_t size = 4;
|
||||
switch (p.kind) {
|
||||
case ArgKind::Float:
|
||||
size = 4;
|
||||
break;
|
||||
case ArgKind::Double:
|
||||
size = 8;
|
||||
break;
|
||||
case ArgKind::Integer:
|
||||
case ArgKind::Enum:
|
||||
case ArgKind::Bool:
|
||||
// For scalars the index stores the value's own width here -- but
|
||||
// computed for x64, where a pointer-sized scalar (UIntPtr, and the
|
||||
// SIZE_T/WPARAM/LPARAM typedefs over it) is 8 bytes and on x86 is
|
||||
// 4. An 8 here is therefore ambiguous: Int64 really does push 8
|
||||
// bytes on x86, UIntPtr pushes 4, and the index cannot tell us
|
||||
// which. Guessing would silently shift every later parameter, so
|
||||
// decline and let the caller fall back to the heuristic.
|
||||
//
|
||||
// Resolving this properly means having the builder emit the x86
|
||||
// footprint alongside the x64 one; there is a spare pad byte in the
|
||||
// parameter record for it.
|
||||
if (p.pointeeSize == 8) {
|
||||
return 0;
|
||||
}
|
||||
size = (p.pointeeSize >= 1 && p.pointeeSize < 8) ? p.pointeeSize : 4;
|
||||
break;
|
||||
default:
|
||||
size = 4; // pointers, handles, and anything unclassified
|
||||
break;
|
||||
}
|
||||
return static_cast<uint16_t>((size + kX86StackAlign - 1) / kX86StackAlign * kX86StackAlign);
|
||||
}
|
||||
|
||||
// Byte offset of each parameter from the first argument on the x86 stack.
|
||||
// Returns false when any parameter's footprint is unknown, in which case every
|
||||
// offset after it would be wrong and the signature must not be used.
|
||||
bool computeStackOffsets(const win32meta::FuncSig& sig, uint16_t (&offsets)[kMaxParams]) {
|
||||
// `slot` is the positional ABI index and is already shifted for a hidden
|
||||
// return pointer, which on x86 is simply the first pushed argument. Walk in
|
||||
// slot order so a shifted signature still lays out correctly.
|
||||
uint16_t running[kMaxParams] = {};
|
||||
uint16_t footprint[kMaxParams] = {};
|
||||
uint8_t maxSlot = 0;
|
||||
|
||||
for (uint8_t i = 0; i < sig.paramCount && i < kMaxParams; ++i) {
|
||||
const win32meta::ParamSig& p = sig.params[i];
|
||||
if (p.slot >= kMaxParams) {
|
||||
return false;
|
||||
}
|
||||
uint16_t bytes = x86StackFootprint(p);
|
||||
if (bytes == 0) {
|
||||
return false;
|
||||
}
|
||||
footprint[p.slot] = bytes;
|
||||
maxSlot = p.slot > maxSlot ? p.slot : maxSlot;
|
||||
}
|
||||
|
||||
// A hidden return pointer occupies slot 0 without appearing in the
|
||||
// parameter list, so fill any gap with a pointer-sized push.
|
||||
uint16_t offset = 0;
|
||||
for (uint8_t s = 0; s <= maxSlot && s < kMaxParams; ++s) {
|
||||
running[s] = offset;
|
||||
offset += footprint[s] != 0 ? footprint[s] : kX86StackAlign;
|
||||
}
|
||||
|
||||
for (uint8_t i = 0; i < sig.paramCount && i < kMaxParams; ++i) {
|
||||
offsets[i] = running[sig.params[i].slot];
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// The value of parameter `index`, from wherever its architecture puts it.
|
||||
uint64_t fetchArg(const CallFrame& frame, const win32meta::FuncSig& sig, uint8_t index,
|
||||
const uint16_t (&x86Offsets)[kMaxParams], bool& ok) {
|
||||
const win32meta::ParamSig& p = sig.params[index];
|
||||
ok = true;
|
||||
|
||||
if (frame.arch == GuestArch::X86) {
|
||||
// At the callee's first instruction ESP points at the return address,
|
||||
// so the arguments begin one word above it.
|
||||
uint64_t addr = static_cast<uint64_t>(frame.x86->Esp) + kX86StackAlign + x86Offsets[index];
|
||||
uint64_t v = 0;
|
||||
size_t width = p.kind == ArgKind::Double ? 8 : 4;
|
||||
ok = readGuest(frame.thread, addr, &v, width);
|
||||
return v;
|
||||
}
|
||||
|
||||
const AMD64_CONTEXT& ctx = *frame.x64;
|
||||
if (p.slot < 4) {
|
||||
if (p.isFloat()) {
|
||||
const M128BIT* xmm[4] = { &ctx.Xmm0, &ctx.Xmm1, &ctx.Xmm2, &ctx.Xmm3 };
|
||||
return xmm[p.slot]->Low;
|
||||
}
|
||||
const uint64_t gpr[4] = { ctx.Rcx, ctx.Rdx, ctx.R8, ctx.R9 };
|
||||
return gpr[p.slot];
|
||||
}
|
||||
// Above the shadow space the caller reserved for RCX/RDX/R8/R9.
|
||||
uint64_t addr = ctx.Rsp + 0x28 + static_cast<uint64_t>(p.slot - 4) * 8;
|
||||
uint64_t v = 0;
|
||||
ok = readGuest(frame.thread, addr, &v, sizeof(v));
|
||||
return v;
|
||||
}
|
||||
|
||||
double floatValue(const win32meta::ParamSig& p, uint64_t bits) {
|
||||
if (p.kind == ArgKind::Float) {
|
||||
float f = 0.0f;
|
||||
uint32_t lo = static_cast<uint32_t>(bits);
|
||||
std::memcpy(&f, &lo, sizeof(f));
|
||||
return static_cast<double>(f);
|
||||
}
|
||||
double d = 0.0;
|
||||
std::memcpy(&d, &bits, sizeof(d));
|
||||
return d;
|
||||
}
|
||||
|
||||
// Widen a pointee of `size` bytes to 64 bits.
|
||||
uint64_t narrowRead(const uint8_t* raw, uint16_t size) {
|
||||
uint64_t v = 0;
|
||||
std::memcpy(&v, raw, size > 8 ? 8 : size);
|
||||
return v;
|
||||
}
|
||||
|
||||
// `fallbackWidth` is what to read when the metadata did not pin the pointee
|
||||
// down -- the guest's pointer width, since an untyped pointee is usually one.
|
||||
bool derefScalar(TTD::Replay::IThreadView const* thread, uint64_t ptr, uint16_t size,
|
||||
uint16_t fallbackWidth, uint64_t& out) {
|
||||
uint16_t width = size == 0 ? fallbackWidth : (size > 8 ? 8 : size);
|
||||
uint8_t buf[8] = {};
|
||||
if (!readGuest(thread, ptr, buf, width)) {
|
||||
return false;
|
||||
}
|
||||
out = narrowRead(buf, width);
|
||||
return true;
|
||||
}
|
||||
|
||||
std::string formatGuid(const uint8_t* b) {
|
||||
char buf[40];
|
||||
std::snprintf(buf, sizeof(buf),
|
||||
"{%08lX-%04X-%04X-%02X%02X-%02X%02X%02X%02X%02X%02X}",
|
||||
static_cast<unsigned long>(narrowRead(b, 4)),
|
||||
static_cast<unsigned>(narrowRead(b + 4, 2)),
|
||||
static_cast<unsigned>(narrowRead(b + 6, 2)),
|
||||
b[8], b[9], b[10], b[11], b[12], b[13], b[14], b[15]);
|
||||
return buf;
|
||||
}
|
||||
|
||||
bool isBufferKind(ArgKind kind) {
|
||||
return kind == ArgKind::AnsiBuffer || kind == ArgKind::WideBuffer || kind == ArgKind::ByteBuffer;
|
||||
}
|
||||
|
||||
// How many bytes a counted buffer parameter spans, or 0 when we can't tell.
|
||||
// `args` supplies the sibling parameter the count lives in -- after a return
|
||||
// pass those may themselves have been filled in, which is exactly what makes
|
||||
// ReadFile's lpBuffer renderable at its *actual* length.
|
||||
uint64_t resolveByteCount(const win32meta::ParamSig& p, const std::vector<DecodedArg>& args) {
|
||||
uint64_t count = 0;
|
||||
switch (p.auxKind) {
|
||||
case AuxKind::CountConst:
|
||||
count = static_cast<uint64_t>(p.auxValue);
|
||||
break;
|
||||
case AuxKind::BytesFromParam:
|
||||
case AuxKind::CountFromParam: {
|
||||
if (p.auxValue < 0 || static_cast<size_t>(p.auxValue) >= args.size()) {
|
||||
return 0;
|
||||
}
|
||||
const DecodedArg& src = args[static_cast<size_t>(p.auxValue)];
|
||||
count = src.has_deref ? src.deref : src.raw;
|
||||
break;
|
||||
}
|
||||
case AuxKind::None:
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
if (count == 0 || count > kMaxCountElements) {
|
||||
return 0;
|
||||
}
|
||||
if (p.auxKind == AuxKind::BytesFromParam) {
|
||||
return count;
|
||||
}
|
||||
uint16_t elem = p.pointeeSize ? p.pointeeSize : 1;
|
||||
return count * elem;
|
||||
}
|
||||
|
||||
// Length of `buf` up to and including its first `charWidth`-wide NUL. A
|
||||
// character buffer's count parameter is the caller's *capacity*, so without
|
||||
// this the report would carry a few hundred bytes of unrelated stack memory
|
||||
// after every out-string.
|
||||
size_t terminatorEnd(const std::vector<uint8_t>& buf, size_t charWidth) {
|
||||
for (size_t i = 0; i + charWidth <= buf.size(); i += charWidth) {
|
||||
bool nul = true;
|
||||
for (size_t k = 0; k < charWidth; ++k) {
|
||||
if (buf[i + k] != 0) {
|
||||
nul = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (nul) {
|
||||
return i + charWidth;
|
||||
}
|
||||
}
|
||||
return buf.size();
|
||||
}
|
||||
|
||||
// Read a counted buffer into `arg`, capped at opt.max_buffer. Character
|
||||
// buffers additionally get a textual rendering, since that's what a rule or
|
||||
// an analyst actually wants to see.
|
||||
void captureBuffer(TTD::Replay::IThreadView const* thread, const DecodeOptions& opt,
|
||||
ArgKind kind, uint64_t ptr, uint64_t byteCount, DecodedArg& arg) {
|
||||
if (ptr < kMinDerefAddr || byteCount == 0) {
|
||||
return;
|
||||
}
|
||||
size_t want = static_cast<size_t>(byteCount < opt.max_buffer ? byteCount : opt.max_buffer);
|
||||
// Record what the buffer really spans, so a consumer can tell a short
|
||||
// buffer from a long one we only kept the head of.
|
||||
arg.bytes_total = byteCount;
|
||||
arg.bytes_capped = want < byteCount;
|
||||
std::vector<uint8_t> buf(want);
|
||||
auto result = thread->QueryMemoryBuffer(TTD::GuestAddress{ ptr }, TTD::BufferView{ buf.data(), want });
|
||||
size_t got = result.Memory.Size;
|
||||
if (got == 0) {
|
||||
return;
|
||||
}
|
||||
buf.resize(got);
|
||||
|
||||
// A character buffer's count is the caller's capacity, so trimming at the
|
||||
// terminator is not truncation -- the string really did end there, and
|
||||
// saying otherwise would send someone hunting for data that never existed.
|
||||
// Only a buffer still running at the cap is genuinely cut short.
|
||||
if (kind == ArgKind::AnsiBuffer || kind == ArgKind::WideBuffer) {
|
||||
size_t charWidth = kind == ArgKind::AnsiBuffer ? 1 : 2;
|
||||
if (auto s = charWidth == 1 ? readAnsiString(thread, ptr, got)
|
||||
: readWideString(thread, ptr, got / sizeof(wchar_t))) {
|
||||
arg.str = std::move(*s);
|
||||
arg.has_str = !arg.str.empty();
|
||||
}
|
||||
size_t end = terminatorEnd(buf, charWidth);
|
||||
if (end < buf.size()) {
|
||||
arg.bytes_capped = false;
|
||||
}
|
||||
buf.resize(end);
|
||||
}
|
||||
arg.bytes = std::move(buf);
|
||||
}
|
||||
|
||||
// Everything that needs a pointer followed. Shared by the entry pass (for
|
||||
// [In] parameters, which are already valid) and the return pass.
|
||||
void dereference(TTD::Replay::IThreadView const* thread, const DecodeOptions& opt,
|
||||
ArgKind kind, uint64_t ptr, uint16_t pointeeSize, uint16_t pointerSize,
|
||||
uint64_t byteCount, DecodedArg& arg) {
|
||||
switch (kind) {
|
||||
case ArgKind::AnsiString:
|
||||
if (auto s = readAnsiString(thread, ptr, opt.max_string)) {
|
||||
arg.str = std::move(*s);
|
||||
arg.has_str = true;
|
||||
}
|
||||
break;
|
||||
case ArgKind::WideString:
|
||||
if (auto s = readWideString(thread, ptr, opt.max_string)) {
|
||||
arg.str = std::move(*s);
|
||||
arg.has_str = true;
|
||||
}
|
||||
break;
|
||||
case ArgKind::PtrToInt: {
|
||||
// An 8-byte pointee on a 32-bit guest is almost always a
|
||||
// pointer-sized type the index measured at x64 width -- HANDLE*,
|
||||
// SIZE_T*, ULONG_PTR* and friends. Reading 8 bytes there splices
|
||||
// the following dword into the value, so trust the guest's width
|
||||
// instead. A genuine 64-bit pointee (LONGLONG*) loses its high
|
||||
// half, which is still better than a value mixed with unrelated
|
||||
// memory. Telling the two apart needs the index to carry 32-bit
|
||||
// sizes; see the README.
|
||||
uint16_t eff = (pointerSize == 4 && pointeeSize == 8) ? 4 : pointeeSize;
|
||||
uint64_t v = 0;
|
||||
if (derefScalar(thread, ptr, eff, pointerSize, v)) {
|
||||
arg.deref = v;
|
||||
arg.has_deref = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case ArgKind::PtrToAnsiString:
|
||||
case ArgKind::PtrToWideString: {
|
||||
// The pointee here is itself a pointer, so its width is the guest's.
|
||||
uint64_t inner = 0;
|
||||
if (!derefScalar(thread, ptr, pointerSize, pointerSize, inner)) {
|
||||
break;
|
||||
}
|
||||
arg.deref = inner;
|
||||
arg.has_deref = true;
|
||||
auto s = (kind == ArgKind::PtrToAnsiString)
|
||||
? readAnsiString(thread, inner, opt.max_string)
|
||||
: readWideString(thread, inner, opt.max_string);
|
||||
if (s) {
|
||||
arg.str = std::move(*s);
|
||||
arg.has_str = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case ArgKind::Guid: {
|
||||
uint8_t g[16] = {};
|
||||
if (readGuest(thread, ptr, g, sizeof(g))) {
|
||||
arg.str = formatGuid(g);
|
||||
arg.has_str = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case ArgKind::AnsiBuffer:
|
||||
case ArgKind::WideBuffer:
|
||||
case ArgKind::ByteBuffer:
|
||||
captureBuffer(thread, opt, kind, ptr, byteCount, arg);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Some parameters are pointers whose pointee the metadata can't pin down:
|
||||
// opaque void*, and the raw UInt16*/UIntPtr* that RPC uses for RPC_WSTR.
|
||||
// For those the old guess-if-it-looks-like-text heuristic is still the best
|
||||
// information available -- and unlike before, we now only apply it to values
|
||||
// we know really are parameters and really are pointers.
|
||||
bool mayHoldUntypedString(ArgKind kind) {
|
||||
return kind == ArgKind::Pointer || kind == ArgKind::Unknown || kind == ArgKind::PtrToInt;
|
||||
}
|
||||
|
||||
void tryUntypedString(TTD::Replay::IThreadView const* thread, DecodedArg& arg) {
|
||||
if (arg.has_str || !mayHoldUntypedString(arg.kind)) {
|
||||
return;
|
||||
}
|
||||
if (auto s = tryReadString(thread, arg.raw)) {
|
||||
arg.str = std::move(*s);
|
||||
arg.has_str = true;
|
||||
return;
|
||||
}
|
||||
// A T** out-parameter: the string lives one more hop away.
|
||||
if (arg.has_deref) {
|
||||
if (auto s = tryReadString(thread, arg.deref)) {
|
||||
arg.str = std::move(*s);
|
||||
arg.has_str = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool needsDeref(ArgKind kind) {
|
||||
switch (kind) {
|
||||
case ArgKind::AnsiString:
|
||||
case ArgKind::WideString:
|
||||
case ArgKind::PtrToInt:
|
||||
case ArgKind::PtrToAnsiString:
|
||||
case ArgKind::PtrToWideString:
|
||||
case ArgKind::Guid:
|
||||
case ArgKind::AnsiBuffer:
|
||||
case ArgKind::WideBuffer:
|
||||
case ArgKind::ByteBuffer:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool x86StackLayout(const win32meta::FuncSig& sig, std::vector<uint16_t>& offsets) {
|
||||
offsets.clear();
|
||||
if (sig.paramCount > kMaxParams) {
|
||||
return false;
|
||||
}
|
||||
uint16_t computed[kMaxParams] = {};
|
||||
if (!computeStackOffsets(sig, computed)) {
|
||||
return false;
|
||||
}
|
||||
offsets.assign(computed, computed + sig.paramCount);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool decodeArgs(const win32meta::FuncSig& sig,
|
||||
const CallFrame& frame,
|
||||
const DecodeOptions& opt,
|
||||
std::vector<DecodedArg>& out,
|
||||
std::vector<PendingOut>& deferred) {
|
||||
if (sig.paramCount > kMaxParams) {
|
||||
return false;
|
||||
}
|
||||
if ((frame.arch == GuestArch::X64 && frame.x64 == nullptr)
|
||||
|| (frame.arch == GuestArch::X86 && frame.x86 == nullptr)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
uint16_t x86Offsets[kMaxParams] = {};
|
||||
if (frame.arch == GuestArch::X86 && !computeStackOffsets(sig, x86Offsets)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const uint16_t pointerSize = frame.pointerSize();
|
||||
TTD::Replay::IThreadView const* thread = frame.thread;
|
||||
|
||||
out.clear();
|
||||
out.resize(sig.paramCount);
|
||||
|
||||
// Pass 1: capture every raw slot first. A buffer's length can live in a
|
||||
// parameter that comes *after* it (ReadFile's lpBuffer refers forward to
|
||||
// nNumberOfBytesToRead), so no dereferencing until all the scalars are in.
|
||||
for (uint8_t i = 0; i < sig.paramCount; ++i) {
|
||||
const win32meta::ParamSig& p = sig.params[i];
|
||||
DecodedArg& arg = out[i];
|
||||
arg.name = p.name;
|
||||
arg.type = p.type;
|
||||
arg.kind = p.kind;
|
||||
arg.enum_index = p.enumIndex;
|
||||
arg.is_out = p.isOut();
|
||||
|
||||
bool ok = false;
|
||||
arg.raw = fetchArg(frame, sig, i, x86Offsets, ok);
|
||||
if (!ok) {
|
||||
// Stack slot we couldn't read: leave it zero rather than invent one.
|
||||
arg.raw = 0;
|
||||
}
|
||||
// A 32-bit guest's pointers and handles are 4 bytes; anything above that
|
||||
// in the word we read is not part of the value.
|
||||
if (pointerSize == 4 && isPointerKind(p.kind)) {
|
||||
arg.raw &= 0xFFFFFFFFull;
|
||||
}
|
||||
if (p.isFloat()) {
|
||||
arg.fval = floatValue(p, arg.raw);
|
||||
arg.has_fval = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2: dereference. Scalars and handles are deliberately untouched --
|
||||
// that alone removes most of the bogus String features the old heuristic
|
||||
// produced from flag values that happened to look like addresses.
|
||||
for (uint8_t i = 0; i < sig.paramCount; ++i) {
|
||||
const win32meta::ParamSig& p = sig.params[i];
|
||||
DecodedArg& arg = out[i];
|
||||
if (!needsDeref(p.kind)) {
|
||||
tryUntypedString(thread, arg);
|
||||
continue;
|
||||
}
|
||||
|
||||
uint64_t byteCount = isBufferKind(p.kind) ? resolveByteCount(p, out) : 0;
|
||||
|
||||
// [In] contents are already valid here. [In,Out] gets read twice: once
|
||||
// now for what the caller passed, then again at the return.
|
||||
if (p.isIn()) {
|
||||
dereference(thread, opt, p.kind, arg.raw, p.pointeeSize, pointerSize, byteCount, arg);
|
||||
tryUntypedString(thread, arg);
|
||||
}
|
||||
if (p.isOut() && arg.raw >= kMinDerefAddr) {
|
||||
PendingOut pending;
|
||||
pending.param_index = i;
|
||||
pending.kind = p.kind;
|
||||
pending.ptr = arg.raw;
|
||||
pending.pointee_size = p.pointeeSize;
|
||||
pending.aux_kind = p.auxKind;
|
||||
pending.aux_value = p.auxValue;
|
||||
pending.in_cap = byteCount;
|
||||
deferred.push_back(pending);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void resolvePendingOuts(const std::vector<PendingOut>& pending,
|
||||
const CallFrame& frame,
|
||||
const DecodeOptions& opt,
|
||||
std::vector<DecodedArg>& args) {
|
||||
const uint16_t pointerSize = frame.pointerSize();
|
||||
TTD::Replay::IThreadView const* thread = frame.thread;
|
||||
|
||||
// Scalars first: a buffer's real length is usually itself an [Out] scalar
|
||||
// (ReadFile's lpNumberOfBytesRead), so it has to be resolved before the
|
||||
// buffer that depends on it.
|
||||
for (const PendingOut& po : pending) {
|
||||
if (po.param_index >= args.size() || isBufferKind(po.kind)) {
|
||||
continue;
|
||||
}
|
||||
DecodedArg& arg = args[po.param_index];
|
||||
DecodedArg fresh;
|
||||
dereference(thread, opt, po.kind, po.ptr, po.pointee_size, pointerSize, 0, fresh);
|
||||
if (fresh.has_deref || fresh.has_str) {
|
||||
fresh.name = arg.name;
|
||||
fresh.type = arg.type;
|
||||
fresh.kind = arg.kind;
|
||||
fresh.enum_index = arg.enum_index;
|
||||
fresh.raw = arg.raw;
|
||||
fresh.is_out = true;
|
||||
fresh.from_return = true;
|
||||
arg = std::move(fresh);
|
||||
}
|
||||
tryUntypedString(thread, arg);
|
||||
}
|
||||
|
||||
for (const PendingOut& po : pending) {
|
||||
if (po.param_index >= args.size() || !isBufferKind(po.kind)) {
|
||||
continue;
|
||||
}
|
||||
DecodedArg& arg = args[po.param_index];
|
||||
|
||||
// Prefer the length the callee reported; fall back to what the caller
|
||||
// offered, and never exceed it.
|
||||
win32meta::ParamSig probe;
|
||||
probe.auxKind = po.aux_kind;
|
||||
probe.auxValue = po.aux_value;
|
||||
probe.pointeeSize = po.pointee_size;
|
||||
uint64_t byteCount = resolveByteCount(probe, args);
|
||||
if (byteCount == 0) {
|
||||
byteCount = po.in_cap;
|
||||
} else if (po.in_cap != 0 && byteCount > po.in_cap) {
|
||||
byteCount = po.in_cap;
|
||||
}
|
||||
if (byteCount == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
DecodedArg fresh;
|
||||
captureBuffer(thread, opt, po.kind, po.ptr, byteCount, fresh);
|
||||
if (!fresh.bytes.empty() || fresh.has_str) {
|
||||
arg.bytes = std::move(fresh.bytes);
|
||||
arg.str = std::move(fresh.str);
|
||||
arg.has_str = fresh.has_str;
|
||||
arg.bytes_total = fresh.bytes_total;
|
||||
arg.bytes_capped = fresh.bytes_capped;
|
||||
arg.from_return = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<ArgValue> toCapaArgs(const std::vector<DecodedArg>& args) {
|
||||
std::vector<ArgValue> out;
|
||||
out.reserve(args.size());
|
||||
for (const DecodedArg& a : args) {
|
||||
if (a.has_str && !a.str.empty()) {
|
||||
out.push_back(a.str);
|
||||
} else {
|
||||
out.push_back(static_cast<int64_t>(a.raw));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
} // namespace ttdcapa
|
||||
@@ -0,0 +1,96 @@
|
||||
#ifndef ABI_HPP
|
||||
#define ABI_HPP
|
||||
|
||||
// The calling-convention decoders.
|
||||
//
|
||||
// The Win32 metadata gives us types and semantics; mapping those onto registers
|
||||
// and stack slots is ours to write. That's this file: given a signature and a
|
||||
// thread's register state at a CALL, produce one DecodedArg per real parameter --
|
||||
// no more, no less.
|
||||
//
|
||||
// Microsoft x64 in one paragraph: the first four parameters go in RCX/RDX/R8/R9,
|
||||
// or XMM0-3 if they're floating point. Crucially the slot index is *shared* between
|
||||
// the two register files, so a float in position 2 lives in XMM2, not XMM0.
|
||||
// Parameters five and up sit at [RSP+0x28] onwards, above the 32-byte shadow space
|
||||
// the caller must reserve. Aggregates that aren't exactly 1/2/4/8 bytes are passed
|
||||
// by hidden pointer, and a function returning one takes an extra hidden first
|
||||
// parameter -- the indexer pre-computes both, so `slot` is always final.
|
||||
//
|
||||
// x86 is simpler and harder at once. Every parameter -- integers, pointers and
|
||||
// floats alike -- is pushed on the stack, so there are no registers to read; but
|
||||
// because it is a byte offset rather than a slot index, each parameter's position
|
||||
// depends on the *sizes* of the ones before it. computeStackOffsets() below walks
|
||||
// the signature to recover those offsets. Note that __stdcall and __cdecl differ
|
||||
// only in who pops the arguments, which is invisible from the callee's entry, so
|
||||
// one layout serves both -- and that covers essentially the whole Win32 surface.
|
||||
//
|
||||
// A WoW64 trace holds 32-bit and 64-bit code at the same time, so the choice is
|
||||
// made per call, from the bitness of the module owning the call target.
|
||||
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#include <TTD/IReplayEngineStl.h>
|
||||
#include <TTD/IReplayEngineRegisters.h>
|
||||
|
||||
#include "ttdutils.hpp"
|
||||
#include "win32meta.hpp"
|
||||
|
||||
namespace ttdcapa {
|
||||
|
||||
struct DecodeOptions {
|
||||
size_t max_buffer = 65536; // bytes of any one counted buffer to keep
|
||||
size_t max_string = 512; // characters of any one string to keep
|
||||
};
|
||||
|
||||
enum class GuestArch : uint8_t {
|
||||
X64,
|
||||
X86,
|
||||
};
|
||||
|
||||
// A call's argument-passing state, whichever architecture it belongs to. Exactly
|
||||
// one of the context pointers is set, matching `arch`.
|
||||
struct CallFrame {
|
||||
GuestArch arch = GuestArch::X64;
|
||||
const AMD64_CONTEXT* x64 = nullptr;
|
||||
const X86_NT5_CONTEXT* x86 = nullptr;
|
||||
TTD::Replay::IThreadView const* thread = nullptr;
|
||||
|
||||
uint16_t pointerSize() const { return arch == GuestArch::X64 ? 8 : 4; }
|
||||
};
|
||||
|
||||
// Decode every parameter of `sig` from the argument state at the call.
|
||||
// Dereferences that only make sense once the callee has run are appended to
|
||||
// `deferred` instead; feed those to resolvePendingOuts at the return.
|
||||
//
|
||||
// Returns false when this signature cannot be laid out for `frame.arch` -- see
|
||||
// computeStackOffsets -- which is the caller's cue to fall back to the heuristic
|
||||
// capture rather than report parameters read from the wrong offsets.
|
||||
bool decodeArgs(const win32meta::FuncSig& sig,
|
||||
const CallFrame& frame,
|
||||
const DecodeOptions& opt,
|
||||
std::vector<DecodedArg>& out,
|
||||
std::vector<PendingOut>& deferred);
|
||||
|
||||
// Re-read the deferred dereferences at the return position and patch them into
|
||||
// `args` (which must be the vector decodeArgs filled in for this same call).
|
||||
void resolvePendingOuts(const std::vector<PendingOut>& pending,
|
||||
const CallFrame& frame,
|
||||
const DecodeOptions& opt,
|
||||
std::vector<DecodedArg>& args);
|
||||
|
||||
// Byte offset of each parameter from the first argument on the x86 stack, in
|
||||
// parameter order. False when the signature has no knowable x86 layout, which is
|
||||
// exactly when decodeArgs would decline it. Exposed so --dump-sig can show the
|
||||
// layout without needing a trace to replay.
|
||||
bool x86StackLayout(const win32meta::FuncSig& sig, std::vector<uint16_t>& offsets);
|
||||
|
||||
// Flatten decoded parameters into the int/string list capa matches rules
|
||||
// against. Strings recovered from [Out] parameters are included -- they are
|
||||
// genuine new evidence; symbolic enum names are not, so rules that match flags
|
||||
// numerically keep working.
|
||||
std::vector<ArgValue> toCapaArgs(const std::vector<DecodedArg>& args);
|
||||
|
||||
} // namespace ttdcapa
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,352 @@
|
||||
#include "binreport.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
#include "win32meta.hpp"
|
||||
|
||||
namespace ttdcapa::binreport {
|
||||
namespace {
|
||||
|
||||
// Append-only growable buffer with little-endian writers. Everything is built in
|
||||
// memory and written once: the whole point is to avoid per-record formatting, and
|
||||
// a few hundred MB of contiguous bytes is one sequential write.
|
||||
class ByteSink {
|
||||
public:
|
||||
void reserve(size_t bytes) { buf_.reserve(bytes); }
|
||||
size_t size() const { return buf_.size(); }
|
||||
const uint8_t* data() const { return buf_.data(); }
|
||||
|
||||
void u8(uint8_t v) { buf_.push_back(v); }
|
||||
void u16(uint16_t v) { raw(&v, sizeof(v)); }
|
||||
void u32(uint32_t v) { raw(&v, sizeof(v)); }
|
||||
void u64(uint64_t v) { raw(&v, sizeof(v)); }
|
||||
|
||||
void raw(const void* p, size_t n) {
|
||||
const auto* b = static_cast<const uint8_t*>(p);
|
||||
buf_.insert(buf_.end(), b, b + n);
|
||||
}
|
||||
|
||||
void pad(size_t to) {
|
||||
while (buf_.size() < to) {
|
||||
buf_.push_back(0);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
std::vector<uint8_t> buf_;
|
||||
};
|
||||
|
||||
// Deduplicating string table. Module and API names repeat across millions of
|
||||
// calls but number only in the thousands, so this is where most of the text goes.
|
||||
class StringTable {
|
||||
public:
|
||||
StringTable() { sink_.u8(0); } // offset 0 is the empty string
|
||||
|
||||
uint32_t add(const std::string& s) {
|
||||
if (s.empty()) {
|
||||
return 0;
|
||||
}
|
||||
auto it = seen_.find(s);
|
||||
if (it != seen_.end()) {
|
||||
return it->second;
|
||||
}
|
||||
uint32_t off = static_cast<uint32_t>(sink_.size());
|
||||
sink_.raw(s.data(), s.size());
|
||||
sink_.u8(0);
|
||||
seen_.emplace(s, off);
|
||||
return off;
|
||||
}
|
||||
|
||||
const ByteSink& bytes() const { return sink_; }
|
||||
|
||||
private:
|
||||
ByteSink sink_;
|
||||
std::unordered_map<std::string, uint32_t> seen_;
|
||||
};
|
||||
|
||||
std::string toLower(std::string s) {
|
||||
std::transform(s.begin(), s.end(), s.begin(),
|
||||
[](unsigned char c) { return static_cast<char>(std::tolower(c)); });
|
||||
return s;
|
||||
}
|
||||
|
||||
std::string toHexValue(uint64_t v) {
|
||||
char buf[32];
|
||||
std::snprintf(buf, sizeof(buf), "0x%llx", static_cast<unsigned long long>(v));
|
||||
return buf;
|
||||
}
|
||||
|
||||
// Printable runs from a captured buffer, so that searching for text a program
|
||||
// wrote or read finds the call that carried it. Without this a buffer parameter
|
||||
// contributes only its pointer to the searchable text, which is useless. Bounded
|
||||
// because a 64 KiB buffer has no business inflating every haystack.
|
||||
std::string printableRuns(const std::vector<uint8_t>& bytes, size_t limit = 96) {
|
||||
std::string out;
|
||||
size_t run = 0;
|
||||
for (uint8_t b : bytes) {
|
||||
if (b >= 0x20 && b < 0x7f) {
|
||||
if (out.size() >= limit) {
|
||||
break;
|
||||
}
|
||||
out.push_back(static_cast<char>(b));
|
||||
++run;
|
||||
} else {
|
||||
// Keep runs separated so two adjacent fields cannot look like one word.
|
||||
if (run != 0 && !out.empty() && out.back() != ' ') {
|
||||
out.push_back(' ');
|
||||
}
|
||||
run = 0;
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// The same text a viewer would search: "module!api" plus each parameter rendered
|
||||
// the way it is displayed. Built here so a filter never has to reconstruct it.
|
||||
std::string buildHaystack(const CallRecord& call, const std::string& moduleName) {
|
||||
std::string out;
|
||||
out.reserve(96);
|
||||
out += moduleName;
|
||||
out += '!';
|
||||
out += call.api;
|
||||
|
||||
if (call.metadata) {
|
||||
for (const DecodedArg& p : call.params) {
|
||||
out += ' ';
|
||||
if (p.name != nullptr) {
|
||||
out += p.name;
|
||||
out += '=';
|
||||
}
|
||||
if (p.has_str && !p.str.empty()) {
|
||||
out += '"';
|
||||
out += p.str;
|
||||
out += '"';
|
||||
} else if (p.enum_index != 0xFFFFFFFFu) {
|
||||
auto names = win32meta::index().decodeEnum(p.enum_index, p.raw);
|
||||
if (!names.empty()) {
|
||||
for (size_t i = 0; i < names.size(); ++i) {
|
||||
if (i != 0) {
|
||||
out += '|';
|
||||
}
|
||||
out += names[i];
|
||||
}
|
||||
} else {
|
||||
out += toHexValue(p.raw);
|
||||
}
|
||||
} else {
|
||||
out += toHexValue(p.raw);
|
||||
}
|
||||
if (!p.bytes.empty()) {
|
||||
std::string text = printableRuns(p.bytes);
|
||||
if (!text.empty()) {
|
||||
out += ' ';
|
||||
out += text;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (const ArgValue& a : call.args) {
|
||||
out += ' ';
|
||||
if (std::holds_alternative<std::string>(a)) {
|
||||
out += '"';
|
||||
out += std::get<std::string>(a);
|
||||
out += '"';
|
||||
} else {
|
||||
out += toHexValue(static_cast<uint64_t>(std::get<int64_t>(a)));
|
||||
}
|
||||
}
|
||||
}
|
||||
return toLower(std::move(out));
|
||||
}
|
||||
|
||||
// "Sequence:Steps" is stored as two integers; parse the string form the sweep
|
||||
// already produced rather than plumbing the raw Position through.
|
||||
void parsePosition(const std::string& text, uint32_t& sequence, uint32_t& steps) {
|
||||
sequence = 0;
|
||||
steps = 0;
|
||||
size_t colon = text.find(':');
|
||||
if (colon == std::string::npos) {
|
||||
return;
|
||||
}
|
||||
sequence = static_cast<uint32_t>(std::strtoull(text.c_str(), nullptr, 16));
|
||||
steps = static_cast<uint32_t>(std::strtoull(text.c_str() + colon + 1, nullptr, 16));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool write(const std::filesystem::path& path, const Report& report, std::string& error) {
|
||||
StringTable strings;
|
||||
ByteSink calls;
|
||||
ByteSink params;
|
||||
ByteSink blob;
|
||||
|
||||
const size_t callCount = report.process.calls.size();
|
||||
calls.reserve(callCount * kCallRecordSize);
|
||||
blob.reserve(callCount * 96);
|
||||
params.reserve(callCount * 24);
|
||||
|
||||
const uint32_t tracePathStr = strings.add(report.trace_path.string());
|
||||
const uint32_t sampleNameStr = strings.add(report.sample_name);
|
||||
|
||||
uint64_t decodedCount = 0;
|
||||
uint64_t maxSeq = 0;
|
||||
uint32_t maxPositionChars = 0;
|
||||
|
||||
for (const CallRecord& call : report.process.calls) {
|
||||
std::string moduleName(call.module.begin(), call.module.end());
|
||||
|
||||
// The haystack and any parameter payloads go in the blob; the call record
|
||||
// only ever holds offsets and lengths.
|
||||
std::string haystack = buildHaystack(call, moduleName);
|
||||
uint32_t searchOff = static_cast<uint32_t>(blob.size());
|
||||
blob.raw(haystack.data(), haystack.size());
|
||||
|
||||
uint32_t paramOff = 0;
|
||||
uint16_t paramCount = 0;
|
||||
if (call.metadata) {
|
||||
++decodedCount;
|
||||
paramCount = static_cast<uint16_t>(call.params.size() & 0x7FFF) | kDecodedFlag;
|
||||
paramOff = static_cast<uint32_t>(params.size());
|
||||
|
||||
for (const DecodedArg& p : call.params) {
|
||||
uint8_t bits = 0;
|
||||
if (p.is_out) bits |= ParamOut;
|
||||
if (p.from_return) bits |= ParamAtReturn;
|
||||
if (p.has_deref) bits |= ParamHasDeref;
|
||||
if (p.has_str && !p.str.empty()) bits |= ParamHasStr;
|
||||
if (!p.bytes.empty()) bits |= ParamHasBytes;
|
||||
|
||||
std::string flagText;
|
||||
if (p.enum_index != 0xFFFFFFFFu) {
|
||||
auto names = win32meta::index().decodeEnum(p.enum_index, p.raw);
|
||||
for (size_t i = 0; i < names.size(); ++i) {
|
||||
if (i != 0) {
|
||||
flagText += '|';
|
||||
}
|
||||
flagText += names[i];
|
||||
}
|
||||
}
|
||||
if (!flagText.empty()) bits |= ParamHasFlags;
|
||||
|
||||
params.u8(static_cast<uint8_t>(p.kind));
|
||||
params.u8(bits);
|
||||
params.u32(strings.add(p.name != nullptr ? p.name : ""));
|
||||
params.u32(strings.add(p.type != nullptr ? p.type : ""));
|
||||
params.u64(p.raw);
|
||||
|
||||
if (bits & ParamHasDeref) {
|
||||
params.u64(p.deref);
|
||||
}
|
||||
if (bits & ParamHasStr) {
|
||||
params.u32(static_cast<uint32_t>(blob.size()));
|
||||
params.u32(static_cast<uint32_t>(p.str.size()));
|
||||
blob.raw(p.str.data(), p.str.size());
|
||||
}
|
||||
if (bits & ParamHasBytes) {
|
||||
params.u32(static_cast<uint32_t>(blob.size()));
|
||||
params.u32(static_cast<uint32_t>(p.bytes.size()));
|
||||
// Raw, not hex: half the size and nothing to decode on load.
|
||||
params.u64(p.bytes_capped ? p.bytes_total : p.bytes.size());
|
||||
blob.raw(p.bytes.data(), p.bytes.size());
|
||||
}
|
||||
if (bits & ParamHasFlags) {
|
||||
params.u32(static_cast<uint32_t>(blob.size()));
|
||||
params.u32(static_cast<uint32_t>(flagText.size()));
|
||||
blob.raw(flagText.data(), flagText.size());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Heuristic capture has no names or types, but the values still need to
|
||||
// be displayable. Encode them as nameless parameters so the reader has
|
||||
// one representation to handle; the missing kDecodedFlag is what tells it
|
||||
// to render them positionally.
|
||||
paramCount = static_cast<uint16_t>(call.args.size() & 0x7FFF);
|
||||
paramOff = static_cast<uint32_t>(params.size());
|
||||
|
||||
for (const ArgValue& a : call.args) {
|
||||
bool isString = std::holds_alternative<std::string>(a);
|
||||
params.u8(static_cast<uint8_t>(win32meta::ArgKind::Unknown));
|
||||
params.u8(isString ? static_cast<uint8_t>(ParamHasStr) : uint8_t{0});
|
||||
params.u32(0); // no name
|
||||
params.u32(0); // no type
|
||||
if (isString) {
|
||||
const std::string& s = std::get<std::string>(a);
|
||||
params.u64(0);
|
||||
params.u32(static_cast<uint32_t>(blob.size()));
|
||||
params.u32(static_cast<uint32_t>(s.size()));
|
||||
blob.raw(s.data(), s.size());
|
||||
} else {
|
||||
params.u64(static_cast<uint64_t>(std::get<int64_t>(a)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
uint32_t posSequence = 0;
|
||||
uint32_t posSteps = 0;
|
||||
parsePosition(call.position, posSequence, posSteps);
|
||||
maxPositionChars = std::max(maxPositionChars, static_cast<uint32_t>(call.position.size()));
|
||||
maxSeq = std::max(maxSeq, call.seq);
|
||||
|
||||
calls.u64(call.has_ret ? call.ret : 0);
|
||||
calls.u32(static_cast<uint32_t>(call.tid));
|
||||
calls.u32(posSequence);
|
||||
calls.u32(posSteps);
|
||||
calls.u32(strings.add(moduleName));
|
||||
calls.u32(strings.add(call.api));
|
||||
calls.u32(paramOff);
|
||||
calls.u32(searchOff);
|
||||
calls.u16(paramCount);
|
||||
calls.u16(static_cast<uint16_t>(std::min<size_t>(haystack.size(), 0xFFFF)));
|
||||
calls.u64(call.return_address);
|
||||
}
|
||||
|
||||
// Assemble: header, then each region in the order the header declares.
|
||||
const uint64_t callsOff = kHeaderSize;
|
||||
const uint64_t paramsOff = callsOff + calls.size();
|
||||
const uint64_t stringsOff = paramsOff + params.size();
|
||||
const uint64_t blobOff = stringsOff + strings.bytes().size();
|
||||
|
||||
ByteSink header;
|
||||
header.raw(kMagic, sizeof(kMagic));
|
||||
header.u32(kVersion);
|
||||
header.u32(static_cast<uint32_t>(report.arch == "x86" ? Arch::X86 : Arch::X64));
|
||||
header.u64(callCount);
|
||||
header.u64(params.size());
|
||||
header.u64(report.process.pid);
|
||||
header.u64(callsOff);
|
||||
header.u64(paramsOff);
|
||||
header.u64(stringsOff);
|
||||
header.u64(strings.bytes().size());
|
||||
header.u64(blobOff);
|
||||
header.u64(blob.size());
|
||||
header.u64(decodedCount);
|
||||
header.u64(maxSeq);
|
||||
header.u32(tracePathStr);
|
||||
header.u32(sampleNameStr);
|
||||
header.u32(maxPositionChars);
|
||||
header.pad(kHeaderSize);
|
||||
|
||||
std::ofstream out(path, std::ios::binary | std::ios::trunc);
|
||||
if (!out.is_open()) {
|
||||
error = "cannot open " + path.string();
|
||||
return false;
|
||||
}
|
||||
out.write(reinterpret_cast<const char*>(header.data()), static_cast<std::streamsize>(header.size()));
|
||||
out.write(reinterpret_cast<const char*>(calls.data()), static_cast<std::streamsize>(calls.size()));
|
||||
out.write(reinterpret_cast<const char*>(params.data()), static_cast<std::streamsize>(params.size()));
|
||||
out.write(reinterpret_cast<const char*>(strings.bytes().data()),
|
||||
static_cast<std::streamsize>(strings.bytes().size()));
|
||||
out.write(reinterpret_cast<const char*>(blob.data()), static_cast<std::streamsize>(blob.size()));
|
||||
if (!out.good()) {
|
||||
error = "write failed for " + path.string();
|
||||
return false;
|
||||
}
|
||||
out.close();
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace ttdcapa::binreport
|
||||
@@ -0,0 +1,125 @@
|
||||
#ifndef BINREPORT_HPP
|
||||
#define BINREPORT_HPP
|
||||
|
||||
// A binary report format for consumers that load the whole thing and browse it, as
|
||||
// opposed to capa, which wants the JSON.
|
||||
//
|
||||
// Why: on a 3.4M-call trace the JSON report costs about 50s to serialise and 16.6s to
|
||||
// load back (8.9s of QJsonDocument parse plus 7.6s building per-call objects), against a
|
||||
// 30s sweep and 138ms of actual disk I/O. Two thirds of the wall clock was the format.
|
||||
// Nothing about that is inherent -- it is text formatting, hex encoding, and allocating
|
||||
// several million small objects.
|
||||
//
|
||||
// This layout is designed to be memory-mapped and read in place: fixed-size call records
|
||||
// indexable by row, a deduplicated string table for the names that repeat (a trace has
|
||||
// millions of calls but only thousands of distinct module!api pairs), raw bytes instead
|
||||
// of hex, positions as two integers instead of "45905:15A5", and a per-call lowercased
|
||||
// search haystack so a filter is a scan over mapped pages rather than something that has
|
||||
// to be built first. Loading becomes a header validation; nothing is parsed and nothing
|
||||
// is allocated per call.
|
||||
//
|
||||
// All integers are little-endian. All offsets are byte offsets from the start of the
|
||||
// file, so a mapped view needs no fixups.
|
||||
|
||||
#include <cstdint>
|
||||
#include <filesystem>
|
||||
#include <string>
|
||||
|
||||
#include "ttdutils.hpp"
|
||||
#include "utils.hpp" // Report
|
||||
|
||||
namespace ttdcapa {
|
||||
|
||||
namespace binreport {
|
||||
|
||||
constexpr char kMagic[8] = { 'T', 'T', 'D', 'B', 'E', 'H', 'V', '1' };
|
||||
// 2 added returnAddress to the call record. A reader that checks the version will
|
||||
// ask for a re-extract rather than misread the older layout.
|
||||
constexpr uint32_t kVersion = 2;
|
||||
constexpr size_t kHeaderSize = 128;
|
||||
constexpr size_t kCallRecordSize = 48;
|
||||
|
||||
enum class Arch : uint32_t {
|
||||
X64 = 0,
|
||||
X86 = 1,
|
||||
};
|
||||
|
||||
// Presence bits in a parameter record's `bits` byte.
|
||||
enum ParamBits : uint8_t {
|
||||
ParamOut = 0x01,
|
||||
ParamAtReturn = 0x02,
|
||||
ParamHasDeref = 0x04,
|
||||
ParamHasStr = 0x08,
|
||||
ParamHasBytes = 0x10,
|
||||
ParamHasFlags = 0x20,
|
||||
};
|
||||
|
||||
// Set in a call record's paramCount field when the call was decoded from a real
|
||||
// signature. Needed separately because a decoded call can legitimately have zero
|
||||
// parameters.
|
||||
constexpr uint16_t kDecodedFlag = 0x8000;
|
||||
|
||||
// File layout:
|
||||
//
|
||||
// [0, 128) header
|
||||
// [callsOff, ...) callCount * 40-byte records, in sweep order
|
||||
// [paramsOff, ...) variable-length parameter records, referenced by call
|
||||
// [stringsOff, ...) NUL-terminated UTF-8, deduplicated
|
||||
// [blobOff, ...) search haystacks, parameter strings, buffer bytes
|
||||
//
|
||||
// Header (offset: type name):
|
||||
// 0: char[8] magic
|
||||
// 8: u32 version
|
||||
// 12: u32 arch
|
||||
// 16: u64 callCount
|
||||
// 24: u64 paramBytes (size of the parameter region)
|
||||
// 32: u64 pid
|
||||
// 40: u64 callsOff
|
||||
// 48: u64 paramsOff
|
||||
// 56: u64 stringsOff
|
||||
// 64: u64 stringsSize
|
||||
// 72: u64 blobOff
|
||||
// 80: u64 blobSize
|
||||
// 88: u64 decodedCount
|
||||
// 96: u64 maxSeq
|
||||
// 104: u32 tracePathStr (offset into the string table)
|
||||
// 108: u32 sampleNameStr
|
||||
// 112: u32 maxPositionChars
|
||||
// 116: u32 reserved
|
||||
// 120: u64 reserved
|
||||
//
|
||||
// Call record (offset: type name):
|
||||
// 0: u64 ret
|
||||
// 8: u32 tid
|
||||
// 12: u32 positionSequence
|
||||
// 16: u32 positionSteps
|
||||
// 20: u32 moduleStr
|
||||
// 24: u32 apiStr
|
||||
// 28: u32 paramOff (offset into the parameter region, 0 if none)
|
||||
// 32: u32 searchOff (offset into the blob)
|
||||
// 36: u16 paramCount (low 15 bits; kDecodedFlag set when decoded)
|
||||
// 38: u16 searchLen
|
||||
// 40: u64 returnAddress (the instruction after the CALL, i.e. the call site)
|
||||
//
|
||||
// Parameter record, variable length:
|
||||
// u8 kind
|
||||
// u8 bits
|
||||
// u32 nameStr
|
||||
// u32 typeStr
|
||||
// u64 value
|
||||
// u64 deref if ParamHasDeref
|
||||
// u32 strOff, u32 strLen if ParamHasStr
|
||||
// u32 bytesOff, u32 bytesLen, u64 bytesTotal if ParamHasBytes
|
||||
// u32 flagsOff, u32 flagsLen if ParamHasFlags
|
||||
//
|
||||
// `seq` is not stored: recorded calls are numbered densely from zero in sweep
|
||||
// order, so a record's index is its sequence number.
|
||||
|
||||
// Write `g_report` to `path` in the format above. Returns false with `error` set.
|
||||
bool write(const std::filesystem::path& path, const Report& report, std::string& error);
|
||||
|
||||
} // namespace binreport
|
||||
|
||||
} // namespace ttdcapa
|
||||
|
||||
#endif
|
||||
+319
-36
@@ -1,9 +1,19 @@
|
||||
// ttdcapa-extract: sweep a Time Travel Debugging (.run) trace and emit a neutral
|
||||
// "TTD report" JSON consumed by capa's TTD dynamic backend
|
||||
// (capa/features/extractors/ttd/). x64 traces only in v1.
|
||||
// (capa/features/extractors/ttd/). x64 and x86 traces, including WoW64.
|
||||
//
|
||||
// ttdcapa-extract <trace.run> [--sample <sample.exe>] [-o <out.json>]
|
||||
// [--max-calls N] [--with-stack-args]
|
||||
// [--max-calls N] [--with-stack-args] [--win32-index <path>]
|
||||
// [--no-metadata] [--max-buffer N]
|
||||
// ttdcapa-extract --dump-sig <ApiName>
|
||||
//
|
||||
// Arguments are decoded against the pre-baked Win32 metadata index whenever the
|
||||
// resolved export is in it (see win32meta.hpp / abi.hpp); calls we have no
|
||||
// signature for fall back to the original heuristic capture.
|
||||
//
|
||||
// Bitness is decided per call from the module owning the target, not once for the
|
||||
// trace: SystemInfo reports the *recording machine's* architecture, which says
|
||||
// nothing about the guest, and a WoW64 process runs both widths at once.
|
||||
|
||||
#include <cassert>
|
||||
#define DBG_ASSERT(cond) assert(cond)
|
||||
@@ -17,6 +27,7 @@
|
||||
#include <TTD/IReplayEngineRegisters.h>
|
||||
#include <TTD/ErrorReporting.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <filesystem>
|
||||
@@ -24,24 +35,119 @@
|
||||
#include <iostream>
|
||||
#include <optional>
|
||||
#include <string>
|
||||
#include <thread>
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
#include "ttdutils.hpp"
|
||||
#include "utils.hpp"
|
||||
#include "ttd_pe_utils.hpp"
|
||||
#include "win32meta.hpp"
|
||||
#include "abi.hpp"
|
||||
#include "binreport.hpp"
|
||||
|
||||
using namespace ttdcapa;
|
||||
|
||||
Report g_report;
|
||||
|
||||
// The cursor currently replaying, for the stdin watcher thread to interrupt. Only
|
||||
// non-null while ReplayForward is running; InterruptReplay is documented as safe to
|
||||
// call from any thread.
|
||||
static std::atomic<TTD::Replay::ICursor*> g_replayingCursor{ nullptr };
|
||||
|
||||
// Print one function's decoded signature and exit. Lets the metadata index and the
|
||||
// ABI classification be checked without waiting on a trace replay.
|
||||
static int dumpSignature(const std::string& api) {
|
||||
const win32meta::FuncSig* sig = win32meta::index().lookup(api);
|
||||
if (sig == nullptr) {
|
||||
std::cerr << "[-] '" << api << "' is not in the metadata index\n";
|
||||
return 3;
|
||||
}
|
||||
std::vector<uint16_t> x86Offsets;
|
||||
bool x86Ok = x86StackLayout(*sig, x86Offsets);
|
||||
|
||||
std::cout << sig->name << " dll=" << sig->dll
|
||||
<< " params=" << static_cast<int>(sig->paramCount)
|
||||
<< (sig->hiddenRetPtr() ? " [hidden-return-pointer]" : "")
|
||||
<< (sig->unsupported() ? " [unsupported]" : "")
|
||||
<< (x86Ok ? "" : " [no x86 layout]") << "\n";
|
||||
for (uint8_t i = 0; i < sig->paramCount; ++i) {
|
||||
const win32meta::ParamSig& p = sig->params[i];
|
||||
std::cout << " slot " << static_cast<int>(p.slot);
|
||||
if (x86Ok) {
|
||||
std::cout << " x86@esp+" << (4 + x86Offsets[i]);
|
||||
}
|
||||
std::cout << " " << p.name
|
||||
<< " : " << p.type << " (" << win32meta::kindName(p.kind) << ")";
|
||||
if (p.isIn()) std::cout << " in";
|
||||
if (p.isOut()) std::cout << " out";
|
||||
if (p.attrs & win32meta::AttrOptional) std::cout << " optional";
|
||||
if (p.auxKind == win32meta::AuxKind::BytesFromParam) std::cout << " bytes=param[" << p.auxValue << "]";
|
||||
if (p.auxKind == win32meta::AuxKind::CountFromParam) std::cout << " count=param[" << p.auxValue << "]";
|
||||
if (p.auxKind == win32meta::AuxKind::CountConst) std::cout << " count=" << p.auxValue;
|
||||
if (p.hasEnum()) std::cout << " enum=" << win32meta::index().enumName(p.enumIndex);
|
||||
std::cout << "\n";
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int wmain(int argc, wchar_t** argv) {
|
||||
Options opt;
|
||||
if (!parse_args(argc, argv, opt)) {
|
||||
std::cerr << "Usage: ttdcapa-extract <trace.run> [--sample <sample.exe>] [-o <out.json>] [--max-calls N] [--with-stack-args]\n";
|
||||
std::cerr << "Usage: ttdcapa-extract <trace.run> [-o <out.json>] [-b <out.ttdb>]\n"
|
||||
" [--sample <sample.exe>] [--max-calls N] [--with-stack-args]\n"
|
||||
" [--win32-index <path>] [--no-metadata] [--max-buffer N]\n"
|
||||
" [--ttd-dlls <dir>] [--progress] [--cancel-on-stdin]\n"
|
||||
" ttdcapa-extract --dump-sig <ApiName>\n"
|
||||
"\n"
|
||||
" --ttd-dlls <dir> where TTDReplay.dll and TTDReplayCPU.dll live, normally\n"
|
||||
" WinDbg's amd64\\\\ttd folder. Defaults to the usual DLL\n"
|
||||
" search order, which finds copies beside this executable.\n";
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
DecodeOptions decode_opt;
|
||||
decode_opt.max_buffer = opt.max_buffer;
|
||||
if (!opt.no_metadata) {
|
||||
std::string err;
|
||||
if (win32meta::loadIndex(opt.win32_index, err)) {
|
||||
std::cerr << "[+] Loaded Win32 metadata for " << win32meta::index().functionCount()
|
||||
<< " functions\n";
|
||||
} else {
|
||||
std::cerr << "[!] " << err << "; falling back to heuristic argument capture\n";
|
||||
}
|
||||
}
|
||||
|
||||
if (!opt.dump_sig.empty()) {
|
||||
return dumpSignature(opt.dump_sig);
|
||||
}
|
||||
|
||||
// Locate the replay engine before anything touches the TTD API. TTDReplay.dll is
|
||||
// delay-loaded, so if --ttd-dlls was given we can pre-load it from there; otherwise
|
||||
// the delay-load resolver falls back to the normal search order, which finds a copy
|
||||
// sitting next to this executable exactly as before.
|
||||
//
|
||||
// The directory goes on the search path as well as the one module being pre-loaded,
|
||||
// because TTDReplay.dll pulls in TTDReplayCPU.dll itself.
|
||||
// Loaded explicitly either way so a missing engine is a clear message rather than
|
||||
// the structured exception a failed delay-load would otherwise raise.
|
||||
{
|
||||
std::wstring replay = L"TTDReplay.dll";
|
||||
if (!opt.ttd_dlls.empty()) {
|
||||
::SetDllDirectoryW(opt.ttd_dlls.c_str());
|
||||
replay = (opt.ttd_dlls / L"TTDReplay.dll").native();
|
||||
}
|
||||
if (::LoadLibraryW(replay.c_str()) == nullptr) {
|
||||
DWORD err = ::GetLastError();
|
||||
std::wcerr << L"[-] Cannot load " << replay << L" (error " << err << L")\n";
|
||||
if (opt.ttd_dlls.empty()) {
|
||||
std::cerr << "[-] Pass --ttd-dlls <dir> pointing at WinDbg's amd64\\ttd folder, or\n"
|
||||
" put TTDReplay.dll and TTDReplayCPU.dll next to this executable.\n";
|
||||
}
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
|
||||
auto [engine, hr] = TTD::Replay::MakeReplayEngine();
|
||||
if (hr != S_OK || !engine) {
|
||||
std::cerr << "[-] Failed to create replay engine: 0x" << std::hex << hr << "\n";
|
||||
@@ -51,28 +157,35 @@ int wmain(int argc, wchar_t** argv) {
|
||||
// Ensure TTD engine is initialized before doing any actual work
|
||||
if (!initializeTTDEngine(engine, opt.trace) || engine == NULL) {
|
||||
std::wcerr << "[-] Exiting...\n";
|
||||
return 5;
|
||||
}
|
||||
|
||||
// Begin building report
|
||||
g_report.trace_path = opt.trace;
|
||||
// Deliberately not gated on sys.System.ProcessorArchitecture: that field comes
|
||||
// from GetSystemInfo() on the recording machine, so a 32-bit process recorded on
|
||||
// an x64 box reports AMD64. The guest's real width comes from each module's PE
|
||||
// header while resolving exports, below.
|
||||
TTD::SystemInfo const& sys = engine->GetSystemInfo();
|
||||
uint16_t pa = static_cast<uint16_t>(sys.System.ProcessorArchitecture);
|
||||
if (pa != PROCESSOR_ARCHITECTURE_AMD64) {
|
||||
std::cerr << "Only x64 traces are supported in this version (arch=" << static_cast<int>(pa) << ").\n";
|
||||
return 2;
|
||||
}
|
||||
g_report.process.pid = static_cast<uint64_t>(sys.ProcessId);
|
||||
|
||||
|
||||
TTD::Replay::UniqueCursor inspection_cursor{ engine->NewCursor() };
|
||||
|
||||
// Get basic report data like strings, imports, sections, etc
|
||||
initializeReport(engine, inspection_cursor, opt.sample);
|
||||
|
||||
// Get list of function VAs used for call sweep
|
||||
std::unordered_map<uint64_t, std::pair<std::wstring, std::string>> resolvedTraceModuleExports;
|
||||
std::unordered_map<uint64_t, ResolvedExport> resolvedTraceModuleExports;
|
||||
resolvedTraceModuleExports = resolveTraceModuleExports(engine, inspection_cursor);
|
||||
std::cerr << "[+] Resolved all exported module functions across execution\n";
|
||||
size_t exports32 = 0;
|
||||
for (const auto& entry : resolvedTraceModuleExports) {
|
||||
if (!entry.second.is64) {
|
||||
++exports32;
|
||||
}
|
||||
}
|
||||
std::cerr << "[+] Resolved " << resolvedTraceModuleExports.size()
|
||||
<< " exported module functions across execution ("
|
||||
<< exports32 << " in 32-bit modules)\n";
|
||||
|
||||
size_t thread_count = engine->GetThreadCount();
|
||||
TTD::Replay::ThreadInfo const* thread_list = engine->GetThreadList();
|
||||
@@ -80,17 +193,61 @@ int wmain(int argc, wchar_t** argv) {
|
||||
g_report.process.threads.push_back(static_cast<uint64_t>(thread_list[i].UniqueId));
|
||||
}
|
||||
|
||||
// per-thread stack of in-flight recorded calls: (expected return addr, call index)
|
||||
std::unordered_map<uint64_t, std::vector<std::pair<uint64_t, size_t>>> in_flight;
|
||||
// per-thread stack of recorded calls we haven't seen return yet, each carrying
|
||||
// the [Out] dereferences we deliberately postponed until the callee has run
|
||||
struct InFlight {
|
||||
uint64_t ret_addr = 0;
|
||||
size_t call_index = 0;
|
||||
GuestArch arch = GuestArch::X64; // must match the entry pass to re-read correctly
|
||||
std::vector<PendingOut> pending;
|
||||
};
|
||||
std::unordered_map<uint64_t, std::vector<InFlight>> in_flight;
|
||||
uint64_t seq = 0;
|
||||
uint64_t events_seen = 0;
|
||||
bool limit_hit = false;
|
||||
|
||||
TTD::Replay::UniqueCursor sweep{ engine->NewCursor() };
|
||||
|
||||
// Progress is measured against the trace's last sequence number. The call callback
|
||||
// is the only place we are given a position, but call/return events are dense
|
||||
// enough (tens of millions in a large trace) for that to be smooth.
|
||||
const uint64_t final_sequence = static_cast<uint64_t>(engine->GetLastPosition().Sequence);
|
||||
uint64_t next_progress_ms = 0;
|
||||
double highest_pct = 0.0;
|
||||
|
||||
auto on_call_return = [&](TTD::GuestAddress target, TTD::GuestAddress fall_through,
|
||||
TTD::Replay::IThreadView const* thread) noexcept {
|
||||
bool is_call = (static_cast<uint64_t>(fall_through) != 0);
|
||||
uint64_t utid = static_cast<uint64_t>(thread->GetThreadInfo().UniqueId);
|
||||
++events_seen;
|
||||
|
||||
// Rate-limited to a few lines a second. The event counter is masked first so
|
||||
// the common path is an AND rather than a clock read.
|
||||
if (opt.report_progress && (events_seen & 0x3FF) == 0) {
|
||||
uint64_t now = ::GetTickCount64();
|
||||
if (now >= next_progress_ms) {
|
||||
next_progress_ms = now + 200;
|
||||
uint64_t at = static_cast<uint64_t>(thread->GetPosition().Sequence);
|
||||
double pct = final_sequence != 0 ? (100.0 * static_cast<double>(at)
|
||||
/ static_cast<double>(final_sequence))
|
||||
: 0.0;
|
||||
if (pct > 100.0) {
|
||||
pct = 100.0;
|
||||
}
|
||||
// Positions from different threads are not perfectly ordered, so a
|
||||
// sample occasionally reads behind the one before it (about 4% of
|
||||
// them). Clamp here rather than letting a progress bar walk
|
||||
// backwards.
|
||||
if (pct < highest_pct) {
|
||||
pct = highest_pct;
|
||||
} else {
|
||||
highest_pct = pct;
|
||||
}
|
||||
std::fprintf(stderr, "[progress] %.2f %zu\n", pct,
|
||||
g_report.process.calls.size());
|
||||
std::fflush(stderr);
|
||||
}
|
||||
}
|
||||
|
||||
if (is_call) {
|
||||
auto it = resolvedTraceModuleExports.find(static_cast<uint64_t>(target));
|
||||
@@ -105,8 +262,11 @@ int wmain(int argc, wchar_t** argv) {
|
||||
ttdcapa::CallRecord rec;
|
||||
rec.tid = utid;
|
||||
rec.seq = seq++;
|
||||
rec.module = it->second.first;
|
||||
rec.api = it->second.second;
|
||||
rec.module = it->second.module;
|
||||
rec.api = it->second.api;
|
||||
// The callback's fall-through address is the return address; no extra
|
||||
// work to obtain it.
|
||||
rec.return_address = static_cast<uint64_t>(fall_through);
|
||||
|
||||
// Retrieve usable TTD timestamp
|
||||
TTD::Replay::Position pos = thread->GetPosition();
|
||||
@@ -121,33 +281,89 @@ int wmain(int argc, wchar_t** argv) {
|
||||
|
||||
TTD::Replay::RegisterContext regs = thread->GetCrossPlatformContext();
|
||||
auto const* ctx = reinterpret_cast<AMD64_CONTEXT const*>(®s);
|
||||
rec.args.push_back(captureCallArg(thread, ctx->Rcx));
|
||||
rec.args.push_back(captureCallArg(thread, ctx->Rdx));
|
||||
rec.args.push_back(captureCallArg(thread, ctx->R8));
|
||||
rec.args.push_back(captureCallArg(thread, ctx->R9));
|
||||
|
||||
if (opt.with_stack_args) { // Capture arguments on stack based on offsets from RSP
|
||||
for (int k = 0; k < 4; ++k) {
|
||||
uint64_t slot = ctx->Rsp + 0x28 + static_cast<uint64_t>(k) * 8;
|
||||
uint64_t v = 0;
|
||||
if (thread->QueryMemoryBuffer(TTD::GuestAddress{ slot }, TTD::BufferView{ &v, sizeof(v) })
|
||||
.Memory.Size == sizeof(v)) {
|
||||
rec.args.push_back(captureCallArg(thread, v));
|
||||
// The target module's PE header decided this; in a WoW64 trace the
|
||||
// 32-bit and 64-bit halves alternate call by call.
|
||||
CallFrame frame;
|
||||
frame.arch = it->second.is64 ? GuestArch::X64 : GuestArch::X86;
|
||||
frame.thread = thread;
|
||||
if (frame.arch == GuestArch::X64) {
|
||||
frame.x64 = ctx;
|
||||
} else {
|
||||
frame.x86 = reinterpret_cast<X86_NT5_CONTEXT const*>(®s);
|
||||
}
|
||||
|
||||
std::vector<PendingOut> pending;
|
||||
win32meta::FuncSig const* sig = win32meta::index().lookup(rec.api);
|
||||
// decodeArgs declines a signature it cannot lay out for this
|
||||
// architecture, in which case the heuristic path below is still
|
||||
// better than parameters read from the wrong offsets.
|
||||
if (sig != nullptr && !sig->unsupported()
|
||||
&& decodeArgs(*sig, frame, decode_opt, rec.params, pending)) {
|
||||
// We know the real arity, so capture exactly that many arguments
|
||||
// -- no stale RDX/R8/R9 residue masquerading as parameters.
|
||||
rec.args = toCapaArgs(rec.params);
|
||||
rec.metadata = true;
|
||||
} else {
|
||||
rec.params.clear();
|
||||
pending.clear();
|
||||
|
||||
if (frame.arch == GuestArch::X86) {
|
||||
// Nothing arrives in registers on x86, so the heuristic has to
|
||||
// read the stack even without --with-stack-args. Four words
|
||||
// keeps it comparable to the x64 fallback's four registers.
|
||||
int words = opt.with_stack_args ? 8 : 4;
|
||||
for (int k = 0; k < words; ++k) {
|
||||
uint64_t addr = static_cast<uint64_t>(frame.x86->Esp) + 4 + static_cast<uint64_t>(k) * 4;
|
||||
uint32_t v = 0;
|
||||
if (thread->QueryMemoryBuffer(TTD::GuestAddress{ addr }, TTD::BufferView{ &v, sizeof(v) })
|
||||
.Memory.Size == sizeof(v)) {
|
||||
rec.args.push_back(captureCallArg(thread, v));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
rec.args.push_back(captureCallArg(thread, ctx->Rcx));
|
||||
rec.args.push_back(captureCallArg(thread, ctx->Rdx));
|
||||
rec.args.push_back(captureCallArg(thread, ctx->R8));
|
||||
rec.args.push_back(captureCallArg(thread, ctx->R9));
|
||||
|
||||
if (opt.with_stack_args) { // Capture arguments on stack based on offsets from RSP
|
||||
for (int k = 0; k < 4; ++k) {
|
||||
uint64_t slot = ctx->Rsp + 0x28 + static_cast<uint64_t>(k) * 8;
|
||||
uint64_t v = 0;
|
||||
if (thread->QueryMemoryBuffer(TTD::GuestAddress{ slot }, TTD::BufferView{ &v, sizeof(v) })
|
||||
.Memory.Size == sizeof(v)) {
|
||||
rec.args.push_back(captureCallArg(thread, v));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
size_t idx = g_report.process.calls.size();
|
||||
in_flight[utid].push_back({ static_cast<uint64_t>(fall_through), idx });
|
||||
in_flight[utid].push_back(
|
||||
InFlight{ static_cast<uint64_t>(fall_through), idx, frame.arch, std::move(pending) });
|
||||
g_report.process.calls.push_back(std::move(rec));
|
||||
} else {
|
||||
// Log most recent function called as returned and capture return value
|
||||
auto it = in_flight.find(utid);
|
||||
if (it != in_flight.end() && !it->second.empty()) {
|
||||
auto& top = it->second.back();
|
||||
if (top.first == static_cast<uint64_t>(target)) {
|
||||
g_report.process.calls[top.second].ret = thread->GetBasicReturnValue();
|
||||
g_report.process.calls[top.second].has_ret = true;
|
||||
InFlight& top = it->second.back();
|
||||
if (top.ret_addr == static_cast<uint64_t>(target)) {
|
||||
CallRecord& call = g_report.process.calls[top.call_index];
|
||||
call.ret = thread->GetBasicReturnValue();
|
||||
call.has_ret = true;
|
||||
if (!top.pending.empty()) {
|
||||
// The payoff of a time-travel trace: [Out] parameters can be
|
||||
// rendered filled in, which a live debugger can't easily do.
|
||||
// Pointer width has to match the entry pass, so carry the
|
||||
// architecture over rather than re-deriving it here.
|
||||
CallFrame retFrame;
|
||||
retFrame.arch = top.arch;
|
||||
retFrame.thread = thread;
|
||||
resolvePendingOuts(top.pending, retFrame, decode_opt, call.params);
|
||||
call.args = toCapaArgs(call.params);
|
||||
}
|
||||
it->second.pop_back();
|
||||
}
|
||||
}
|
||||
@@ -155,17 +371,84 @@ int wmain(int argc, wchar_t** argv) {
|
||||
};
|
||||
|
||||
sweep->SetCallReturnCallback(on_call_return);
|
||||
sweep->SetReplayFlags(TTD::Replay::ReplayFlags::ReplayAllSegmentsWithoutFiltering | TTD::Replay::ReplayFlags::ReplaySegmentsSequentially);
|
||||
// Without ReplayAllSegmentsWithoutFiltering the engine only replays segments it
|
||||
// thinks can hit an event, and a call/return callback alone doesn't qualify --
|
||||
// the sweep then completes instantly having seen nothing.
|
||||
sweep->SetReplayFlags(TTD::Replay::ReplayFlags::ReplaySegmentsSequentially |
|
||||
TTD::Replay::ReplayFlags::ReplayAllSegmentsWithoutFiltering);
|
||||
sweep->SetPosition(TTD::Replay::Position::Min);
|
||||
std::cerr << "[+] Beginning execution sweep...\n";
|
||||
sweep->ReplayForward();
|
||||
if (opt.report_progress) {
|
||||
// The sweep is only part of the wall clock -- serialising a multi-hundred-MB
|
||||
// report takes comparable time -- so name the phases rather than let a caller's
|
||||
// progress bar sit at 100% looking wedged.
|
||||
std::fprintf(stderr, "[phase] sweep\n");
|
||||
std::fflush(stderr);
|
||||
}
|
||||
|
||||
// A UI driving this as a child process can ask to stop early. Everything recorded
|
||||
// so far is already in g_report, so an interrupted sweep still yields a usable
|
||||
// report -- only the calls that had not returned yet lose their return values and
|
||||
// [Out] parameters.
|
||||
g_replayingCursor.store(sweep.get());
|
||||
std::thread cancel_watcher;
|
||||
if (opt.cancel_on_stdin) {
|
||||
cancel_watcher = std::thread([] {
|
||||
std::string line;
|
||||
while (std::getline(std::cin, line)) {
|
||||
if (line.rfind("cancel", 0) == 0) {
|
||||
if (auto* cursor = g_replayingCursor.load()) {
|
||||
cursor->InterruptReplay();
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
});
|
||||
// Detached because it is parked in a blocking read; the process exits out from
|
||||
// under it once the report is written.
|
||||
cancel_watcher.detach();
|
||||
}
|
||||
|
||||
TTD::Replay::ICursorView::ReplayResult result = sweep->ReplayForward();
|
||||
g_replayingCursor.store(nullptr);
|
||||
|
||||
const bool interrupted = result.StopReason == TTD::Replay::EventType::Interrupted;
|
||||
if (interrupted) {
|
||||
std::cerr << "[!] Sweep cancelled; writing the " << g_report.process.calls.size()
|
||||
<< " calls recorded so far\n";
|
||||
}
|
||||
|
||||
if (limit_hit) {
|
||||
std::cerr << "[!] Reached --max-calls limit (" << opt.max_calls << ")\n";
|
||||
}
|
||||
std::cerr << "[+] Recorded " << g_report.process.calls.size() << " API calls\n";
|
||||
std::cerr << "[+] Recorded " << g_report.process.calls.size() << " API calls from "
|
||||
<< events_seen << " call/return events\n";
|
||||
|
||||
writeReport(opt.output);
|
||||
if (opt.report_progress) {
|
||||
std::fprintf(stderr, "[phase] write\n");
|
||||
std::fflush(stderr);
|
||||
}
|
||||
if (!opt.binary_output.empty()) {
|
||||
uint64_t began = ::GetTickCount64();
|
||||
std::string err;
|
||||
if (binreport::write(opt.binary_output, g_report, err)) {
|
||||
uint64_t took = ::GetTickCount64() - began;
|
||||
std::cerr << "[+] Wrote binary report " << opt.binary_output.string() << " in "
|
||||
<< (static_cast<double>(took) / 1000.0) << "s\n";
|
||||
// Machine-readable for a UI that wants to show where the time went.
|
||||
std::fprintf(stderr, "[timing] write %.3f\n", static_cast<double>(took) / 1000.0);
|
||||
std::fflush(stderr);
|
||||
} else {
|
||||
std::cerr << "[-] " << err << "\n";
|
||||
}
|
||||
}
|
||||
if (!opt.output.empty()) {
|
||||
uint64_t began = ::GetTickCount64();
|
||||
writeReport(opt.output);
|
||||
std::fprintf(stderr, "[timing] write-json %.3f\n",
|
||||
static_cast<double>(::GetTickCount64() - began) / 1000.0);
|
||||
std::fflush(stderr);
|
||||
}
|
||||
|
||||
std::cerr << "[!] Exiting...\n";
|
||||
return 0;
|
||||
|
||||
+88
-35
@@ -41,21 +41,76 @@ std::string readCSTR(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress addr,
|
||||
return s;
|
||||
}
|
||||
|
||||
// Locate the NT headers for the image at `base`. Returns the file-relative offset
|
||||
// of the NT headers (e_lfanew) and validates the PE/PE32+ signatures.
|
||||
bool readNTHeaders(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress, IMAGE_NT_HEADERS64& nt) {
|
||||
// The signature and IMAGE_FILE_HEADER that precede the optional header are identical
|
||||
// in PE32 and PE32+, so the offset of the optional header is the same in both.
|
||||
constexpr uint32_t kOptionalHeaderOffset = offsetof(IMAGE_NT_HEADERS64, OptionalHeader);
|
||||
static_assert(kOptionalHeaderOffset == offsetof(IMAGE_NT_HEADERS32, OptionalHeader),
|
||||
"PE32 and PE32+ must agree on where the optional header starts");
|
||||
|
||||
// Just the parts of an image's headers the parsers below need, read in a way that
|
||||
// works for both PE32 and PE32+.
|
||||
struct PeHeaders {
|
||||
bool is64 = false;
|
||||
uint32_t ntOffset = 0; // e_lfanew
|
||||
uint16_t numberOfSections = 0;
|
||||
uint16_t sizeOfOptionalHeader = 0;
|
||||
IMAGE_DATA_DIRECTORY dataDir[IMAGE_NUMBEROF_DIRECTORY_ENTRIES] {};
|
||||
|
||||
// Width of a thunk/pointer in this image.
|
||||
uint32_t pointerSize() const { return is64 ? 8u : 4u; }
|
||||
|
||||
// The section table follows the optional header, whose size varies by format.
|
||||
uint32_t sectionTableOffset() const { return ntOffset + kOptionalHeaderOffset + sizeOfOptionalHeader; }
|
||||
};
|
||||
|
||||
// Locate and validate the headers of the image at `base`, for either PE format.
|
||||
bool readNTHeaders(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress, PeHeaders& pe) {
|
||||
IMAGE_DOS_HEADER dos{};
|
||||
if (!readMemory(cursor, moduleBaseAddress, &dos, sizeof(dos)) || dos.e_magic != IMAGE_DOS_SIGNATURE) {
|
||||
return false;
|
||||
}
|
||||
if (!readMemory(cursor, moduleBaseAddress + static_cast<uint32_t>(dos.e_lfanew), &nt, sizeof(nt))) {
|
||||
pe.ntOffset = static_cast<uint32_t>(dos.e_lfanew);
|
||||
|
||||
struct {
|
||||
uint32_t Signature;
|
||||
IMAGE_FILE_HEADER FileHeader;
|
||||
} head {};
|
||||
if (!readMemory(cursor, moduleBaseAddress + pe.ntOffset, &head, sizeof(head))) {
|
||||
return false;
|
||||
}
|
||||
if (nt.Signature != IMAGE_NT_SIGNATURE ||
|
||||
nt.OptionalHeader.Magic != IMAGE_NT_OPTIONAL_HDR64_MAGIC) {
|
||||
if (head.Signature != IMAGE_NT_SIGNATURE) {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
pe.numberOfSections = head.FileHeader.NumberOfSections;
|
||||
pe.sizeOfOptionalHeader = head.FileHeader.SizeOfOptionalHeader;
|
||||
|
||||
// The magic decides which optional header layout follows, and with it the image's
|
||||
// bitness -- the one thing about a module we cannot get from the TTD module list.
|
||||
TTD::GuestAddress optAddr = moduleBaseAddress + pe.ntOffset + kOptionalHeaderOffset;
|
||||
uint16_t magic = 0;
|
||||
if (!readMemory(cursor, optAddr, &magic, sizeof(magic))) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (magic == IMAGE_NT_OPTIONAL_HDR64_MAGIC) {
|
||||
IMAGE_OPTIONAL_HEADER64 opt{};
|
||||
if (!readMemory(cursor, optAddr, &opt, sizeof(opt))) {
|
||||
return false;
|
||||
}
|
||||
pe.is64 = true;
|
||||
std::memcpy(pe.dataDir, opt.DataDirectory, sizeof(pe.dataDir));
|
||||
return true;
|
||||
}
|
||||
if (magic == IMAGE_NT_OPTIONAL_HDR32_MAGIC) {
|
||||
IMAGE_OPTIONAL_HEADER32 opt{};
|
||||
if (!readMemory(cursor, optAddr, &opt, sizeof(opt))) {
|
||||
return false;
|
||||
}
|
||||
pe.is64 = false;
|
||||
std::memcpy(pe.dataDir, opt.DataDirectory, sizeof(pe.dataDir));
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool isPrintableASCII(unsigned char c) {
|
||||
@@ -82,13 +137,17 @@ std::string getModuleBaseName(const std::wstring& full) {
|
||||
return out;
|
||||
}
|
||||
|
||||
bool getModuleExports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress, std::vector<std::pair<uint64_t, std::string>>& out) {
|
||||
IMAGE_NT_HEADERS64 nt{};
|
||||
if (!readNTHeaders(cursor, moduleBaseAddress, nt)) {
|
||||
bool getModuleExports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress,
|
||||
std::vector<std::pair<uint64_t, std::string>>& out, bool* is64Bit) {
|
||||
PeHeaders pe{};
|
||||
if (!readNTHeaders(cursor, moduleBaseAddress, pe)) {
|
||||
return false;
|
||||
}
|
||||
if (is64Bit != nullptr) {
|
||||
*is64Bit = pe.is64;
|
||||
}
|
||||
|
||||
const IMAGE_DATA_DIRECTORY& dir = nt.OptionalHeader.DataDirectory[IMAGE_DIRECTORY_ENTRY_EXPORT];
|
||||
const IMAGE_DATA_DIRECTORY& dir = pe.dataDir[IMAGE_DIRECTORY_ENTRY_EXPORT];
|
||||
if (dir.VirtualAddress == 0 || dir.Size == 0) {
|
||||
return true; // valid PE, just no exports
|
||||
}
|
||||
@@ -144,13 +203,13 @@ bool getModuleExports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress modul
|
||||
}
|
||||
|
||||
bool getModuleImports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress, std::vector<ImportRecord>& out) {
|
||||
IMAGE_NT_HEADERS64 nt{};
|
||||
PeHeaders pe{};
|
||||
|
||||
if (!readNTHeaders(cursor, moduleBaseAddress, nt)) {
|
||||
if (!readNTHeaders(cursor, moduleBaseAddress, pe)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const IMAGE_DATA_DIRECTORY& dir = nt.OptionalHeader.DataDirectory[IMAGE_DIRECTORY_ENTRY_IMPORT];
|
||||
const IMAGE_DATA_DIRECTORY& dir = pe.dataDir[IMAGE_DIRECTORY_ENTRY_IMPORT];
|
||||
if (dir.VirtualAddress == 0 || dir.Size == 0) {
|
||||
return true;
|
||||
}
|
||||
@@ -176,15 +235,19 @@ bool getModuleImports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress modul
|
||||
continue;
|
||||
}
|
||||
|
||||
// Thunks are pointer-sized, so a PE32 image's import tables are half as wide.
|
||||
const uint32_t thunkSize = pe.pointerSize();
|
||||
const uint64_t ordinalFlag = pe.is64 ? IMAGE_ORDINAL_FLAG64 : IMAGE_ORDINAL_FLAG32;
|
||||
|
||||
for (uint32_t t = 0;; ++t) {
|
||||
uint64_t thunk = 0;
|
||||
if (!readMemory(cursor, moduleBaseAddress + int_rva + t * sizeof(uint64_t), &thunk, sizeof(thunk)) || thunk == 0) {
|
||||
if (!readMemory(cursor, moduleBaseAddress + int_rva + t * thunkSize, &thunk, thunkSize) || thunk == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
TTD::GuestAddress slot_va = moduleBaseAddress + iat_rva + t * sizeof(uint64_t);
|
||||
TTD::GuestAddress slot_va = moduleBaseAddress + iat_rva + t * thunkSize;
|
||||
|
||||
if (thunk & IMAGE_ORDINAL_FLAG64) {
|
||||
if (thunk & ordinalFlag) {
|
||||
// import by ordinal: no name to match against; record a synthetic name
|
||||
ImportRecord rec;
|
||||
rec.dll = dll;
|
||||
@@ -208,20 +271,15 @@ bool getModuleImports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress modul
|
||||
}
|
||||
|
||||
bool getModuleSections(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress, std::vector<SectionRecord>& out) {
|
||||
IMAGE_NT_HEADERS64 nt{};
|
||||
PeHeaders pe{};
|
||||
|
||||
if (!readNTHeaders(cursor, moduleBaseAddress, nt)) {
|
||||
if (!readNTHeaders(cursor, moduleBaseAddress, pe)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
IMAGE_DOS_HEADER dos{};
|
||||
if (!readMemory(cursor, moduleBaseAddress, &dos, sizeof(dos))) {
|
||||
return false;
|
||||
}
|
||||
// section headers follow the optional header
|
||||
TTD::GuestAddress sect_va = moduleBaseAddress + dos.e_lfanew + offsetof(IMAGE_NT_HEADERS64, OptionalHeader) + nt.FileHeader.SizeOfOptionalHeader;
|
||||
TTD::GuestAddress sect_va = moduleBaseAddress + pe.sectionTableOffset();
|
||||
|
||||
for (uint16_t i = 0; i < nt.FileHeader.NumberOfSections; ++i) {
|
||||
for (uint16_t i = 0; i < pe.numberOfSections; ++i) {
|
||||
IMAGE_SECTION_HEADER sh{};
|
||||
|
||||
if (!readMemory(cursor, sect_va + i * sizeof(IMAGE_SECTION_HEADER), &sh, sizeof(sh))) {
|
||||
@@ -239,20 +297,15 @@ bool getModuleSections(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress modu
|
||||
}
|
||||
|
||||
bool getModuleStrings(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress, std::vector<std::string>& out, size_t minLength, size_t maxStrings) {
|
||||
IMAGE_NT_HEADERS64 nt{};
|
||||
PeHeaders pe{};
|
||||
|
||||
if (!readNTHeaders(cursor, moduleBaseAddress, nt)) {
|
||||
if (!readNTHeaders(cursor, moduleBaseAddress, pe)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
IMAGE_DOS_HEADER dos{};
|
||||
if (!readMemory(cursor, moduleBaseAddress, &dos, sizeof(dos))) {
|
||||
return false;
|
||||
}
|
||||
TTD::GuestAddress sect_va = moduleBaseAddress + pe.sectionTableOffset();
|
||||
|
||||
TTD::GuestAddress sect_va = moduleBaseAddress + dos.e_lfanew + offsetof(IMAGE_NT_HEADERS64, OptionalHeader) + nt.FileHeader.SizeOfOptionalHeader;
|
||||
|
||||
for (uint16_t i = 0; i < nt.FileHeader.NumberOfSections && out.size() < maxStrings; ++i) {
|
||||
for (uint16_t i = 0; i < pe.numberOfSections && out.size() < maxStrings; ++i) {
|
||||
IMAGE_SECTION_HEADER sh{};
|
||||
if (!readMemory(cursor, sect_va + i * sizeof(IMAGE_SECTION_HEADER), &sh, sizeof(sh))) {
|
||||
break;
|
||||
|
||||
@@ -16,8 +16,13 @@ namespace ttdcapa {
|
||||
|
||||
// Parse the export directory of the PE image mapped at `base`. Appends one entry
|
||||
// per named export (forwarders are skipped). Returns false if `base` is not a
|
||||
// readable PE32+ image at this position.
|
||||
bool getModuleExports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress, std::vector<std::pair<uint64_t, std::string>>& out);
|
||||
// readable PE32 or PE32+ image at this position.
|
||||
//
|
||||
// `is64Bit`, when given, receives the image's bitness. A WoW64 trace contains both
|
||||
// kinds at once, and the bitness of the module owning a call target is what selects
|
||||
// the calling convention used to decode that call's arguments.
|
||||
bool getModuleExports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress,
|
||||
std::vector<std::pair<uint64_t, std::string>>& out, bool* is64Bit = nullptr);
|
||||
|
||||
// Parse the import directory of the PE image mapped at `base`.
|
||||
bool getModuleImports(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress moduleBaseAddress, std::vector<ImportRecord>& out);
|
||||
|
||||
+139
-9
@@ -19,8 +19,8 @@
|
||||
extern ttdcapa::Report g_report;
|
||||
|
||||
namespace ttdcapa {
|
||||
std::unordered_map<uint64_t, std::pair<std::wstring, std::string>> resolveTraceModuleExports(TTD::Replay::UniqueReplayEngine& engine, TTD::Replay::UniqueCursor& cursor) {
|
||||
std::unordered_map<uint64_t, std::pair<std::wstring, std::string>> resolvedTraceModuleExports; // function VA -> (module, api)
|
||||
std::unordered_map<uint64_t, ResolvedExport> resolveTraceModuleExports(TTD::Replay::UniqueReplayEngine& engine, TTD::Replay::UniqueCursor& cursor) {
|
||||
std::unordered_map<uint64_t, ResolvedExport> resolvedTraceModuleExports; // function VA -> resolved target
|
||||
|
||||
size_t count = engine->GetModuleLoadedEventCount();
|
||||
TTD::Replay::ModuleLoadedEvent const* moduleLoadEvents = engine->GetModuleLoadedEventList();
|
||||
@@ -38,9 +38,14 @@ namespace ttdcapa {
|
||||
std::wstring moduleNameStripped(moduleBaseName.begin(), moduleBaseName.end());
|
||||
|
||||
std::vector<std::pair<uint64_t, std::string>> moduleExports;
|
||||
if (getModuleExports(&cursor, moduleLoadEvent.pModule->Address, moduleExports)) {
|
||||
bool is64 = true;
|
||||
if (getModuleExports(&cursor, moduleLoadEvent.pModule->Address, moduleExports, &is64)) {
|
||||
for (auto& moduleExport : moduleExports) {
|
||||
resolvedTraceModuleExports.emplace(moduleExport.first, std::make_pair(moduleNameStripped, std::move(moduleExport.second)));
|
||||
ResolvedExport resolved;
|
||||
resolved.module = moduleNameStripped;
|
||||
resolved.api = std::move(moduleExport.second);
|
||||
resolved.is64 = is64;
|
||||
resolvedTraceModuleExports.emplace(moduleExport.first, std::move(resolved));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -48,13 +53,30 @@ namespace ttdcapa {
|
||||
return resolvedTraceModuleExports;
|
||||
}
|
||||
|
||||
std::string convertWstringToString(const std::wstring& ws) {
|
||||
if (ws.empty()) {
|
||||
return {};
|
||||
}
|
||||
int needed = ::WideCharToMultiByte(CP_UTF8, 0, ws.data(), static_cast<int>(ws.size()), nullptr, 0, nullptr, nullptr);
|
||||
if (needed <= 0) {
|
||||
return {};
|
||||
}
|
||||
std::string out(static_cast<size_t>(needed), '\0');
|
||||
::WideCharToMultiByte(CP_UTF8, 0, ws.data(), static_cast<int>(ws.size()), out.data(), needed, nullptr, nullptr);
|
||||
return out;
|
||||
}
|
||||
|
||||
size_t readMemory(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress addr, void* dest, unsigned __int64 size) {
|
||||
TTD::Replay::MemoryBuffer memoryBuffer = cursor->get()->QueryMemoryBuffer(addr, TTD::BufferView{ dest, size });
|
||||
return memoryBuffer.Memory.Size;
|
||||
}
|
||||
|
||||
bool initializeTTDEngine(TTD::Replay::UniqueReplayEngine& engine, std::wstring traceFilePath) {
|
||||
ConsoleErrorReporting reporter;
|
||||
// The engine holds this pointer for its whole lifetime and calls into it
|
||||
// whenever it reports an error. A stack local would dangle the moment this
|
||||
// function returns, and the first derailment during ReplayForward would then
|
||||
// dispatch a virtual call through reclaimed stack memory.
|
||||
static ConsoleErrorReporting reporter;
|
||||
engine->RegisterDebugModeAndLogging(TTD::Replay::DebugModeType::None, &reporter);
|
||||
|
||||
if (!engine->Initialize(traceFilePath.c_str())) {
|
||||
@@ -64,13 +86,22 @@ namespace ttdcapa {
|
||||
|
||||
if (engine->GetIndexStatus() != TTD::Replay::IndexStatus::IndexFileLoaded) {
|
||||
std::cerr << "[+] Building index (first run may be slow)...\n";
|
||||
auto progress = [](void const*, TTD::Replay::IndexBuildProgressType const* d) noexcept -> void {
|
||||
if (d->KeyframeCount > 0) {
|
||||
std::cerr << "\r" << (d->KeyframesProcessed * 100 / d->KeyframeCount) << "% ";
|
||||
|
||||
auto reportProgress = [](void const* ctx, TTD::Replay::IndexBuildProgressType const* progress) noexcept -> void {
|
||||
if (progress != nullptr && progress->KeyframeCount > 0) {
|
||||
uint32_t pct = progress->KeyframesProcessed * 100 / progress->KeyframeCount;
|
||||
std::cerr << "\r[+] Building index... " << pct << "% ";
|
||||
}
|
||||
};
|
||||
engine->BuildIndex(progress, nullptr, TTD::Replay::IndexBuildFlags::None);
|
||||
|
||||
TTD::Replay::IndexStatus status =
|
||||
engine->BuildIndex(reportProgress, nullptr, TTD::Replay::IndexBuildFlags::None);
|
||||
std::cerr << "\n";
|
||||
|
||||
if (status == TTD::Replay::IndexStatus::IndexFileNotPresent) {
|
||||
std::cerr << "[-] Failed to build trace index\n";
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
std::cerr << "[+] Initialized TTD engine\n";
|
||||
@@ -138,6 +169,105 @@ namespace ttdcapa {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
namespace {
|
||||
// Shared by both typed readers: an address below the first 64 KiB is never a
|
||||
// valid user-mode string pointer, and treating one as such is how a flags
|
||||
// DWORD ends up rendered as text.
|
||||
constexpr uint64_t kMinStringAddr = 0x10000;
|
||||
|
||||
bool isPlausibleTextByte(unsigned char c) {
|
||||
return c == '\t' || c == '\r' || c == '\n' || (c >= 0x20 && c != 0x7f);
|
||||
}
|
||||
|
||||
// The -A entry points take strings in the ANSI code page, so their high
|
||||
// bytes are not UTF-8. Emitting them raw would produce a report that the
|
||||
// JSON serializer refuses to write, so transcode before anything else sees
|
||||
// them. Pure-ASCII input (the overwhelming majority) short-circuits.
|
||||
std::string ansiToUtf8(std::string bytes) {
|
||||
bool ascii = true;
|
||||
for (unsigned char c : bytes) {
|
||||
if (c >= 0x80) {
|
||||
ascii = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (ascii) {
|
||||
return bytes;
|
||||
}
|
||||
|
||||
int wide = ::MultiByteToWideChar(CP_ACP, 0, bytes.data(), static_cast<int>(bytes.size()), nullptr, 0);
|
||||
if (wide > 0) {
|
||||
std::wstring ws(static_cast<size_t>(wide), L'\0');
|
||||
if (::MultiByteToWideChar(CP_ACP, 0, bytes.data(), static_cast<int>(bytes.size()), ws.data(), wide) == wide) {
|
||||
return convertWstringToString(ws);
|
||||
}
|
||||
}
|
||||
|
||||
// Unconvertible: keep the ASCII skeleton rather than dropping the string.
|
||||
for (char& c : bytes) {
|
||||
if (static_cast<unsigned char>(c) >= 0x80) {
|
||||
c = '?';
|
||||
}
|
||||
}
|
||||
return bytes;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
std::optional<std::string> readAnsiString(TTD::Replay::IThreadView const* thread, uint64_t addr, size_t maxChars) {
|
||||
if (addr < kMinStringAddr) {
|
||||
return std::nullopt;
|
||||
}
|
||||
std::vector<char> buf(maxChars + 1);
|
||||
auto result = thread->QueryMemoryBuffer(TTD::GuestAddress{ addr }, TTD::BufferView{ buf.data(), maxChars });
|
||||
size_t avail = result.Memory.Size;
|
||||
if (avail == 0) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
std::string s;
|
||||
for (size_t i = 0; i < avail; ++i) {
|
||||
unsigned char c = static_cast<unsigned char>(buf[i]);
|
||||
if (c == 0) {
|
||||
return ansiToUtf8(std::move(s));
|
||||
}
|
||||
if (!isPlausibleTextByte(c)) {
|
||||
return std::nullopt; // control bytes mean this wasn't a string after all
|
||||
}
|
||||
s.push_back(static_cast<char>(c));
|
||||
}
|
||||
// Ran out of readable memory before the terminator; keep what we have as
|
||||
// long as it looked like text the whole way.
|
||||
return s.empty() ? std::nullopt : std::optional<std::string>(ansiToUtf8(std::move(s)));
|
||||
}
|
||||
|
||||
std::optional<std::string> readWideString(TTD::Replay::IThreadView const* thread, uint64_t addr, size_t maxChars) {
|
||||
if (addr < kMinStringAddr) {
|
||||
return std::nullopt;
|
||||
}
|
||||
std::vector<wchar_t> buf(maxChars + 1);
|
||||
auto result = thread->QueryMemoryBuffer(
|
||||
TTD::GuestAddress{ addr }, TTD::BufferView{ buf.data(), maxChars * sizeof(wchar_t) });
|
||||
size_t availChars = result.Memory.Size / sizeof(wchar_t);
|
||||
if (availChars == 0) {
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
size_t len = 0;
|
||||
while (len < availChars && buf[len] != L'\0') {
|
||||
wchar_t wc = buf[len];
|
||||
// Reject C0 controls (other than the usual whitespace) rather than
|
||||
// emitting mojibake for a pointer that only looked like a string.
|
||||
if (wc < 0x20 && wc != L'\t' && wc != L'\r' && wc != L'\n') {
|
||||
return std::nullopt;
|
||||
}
|
||||
++len;
|
||||
}
|
||||
if (len == 0) {
|
||||
return std::string{};
|
||||
}
|
||||
return convertWstringToString(std::wstring(buf.data(), len));
|
||||
}
|
||||
|
||||
// Capture one candidate argument: a dereferenced string if it points to one, otherwise the raw integer value.
|
||||
ttdcapa::ArgValue captureCallArg(TTD::Replay::IThreadView const* thread, uint64_t value) {
|
||||
if (auto s = tryReadString(thread, value)) {
|
||||
|
||||
+66
-2
@@ -14,6 +14,8 @@
|
||||
#include <TTD/IReplayEngineStl.h>
|
||||
#include <TTD/ErrorReporting.h>
|
||||
|
||||
#include "win32meta.hpp"
|
||||
|
||||
class ConsoleErrorReporting : public TTD::ErrorReporting {
|
||||
public:
|
||||
void __fastcall VPrintError(char const* fmt, va_list args) override {
|
||||
@@ -43,15 +45,57 @@ namespace ttdcapa {
|
||||
// One call argument: either an integer or a dereferenced string
|
||||
using ArgValue = std::variant<int64_t, std::string>;
|
||||
|
||||
// One parameter decoded with the help of the Win32 metadata index. `name` and
|
||||
// `type` point into the index blob, which outlives every record, so they cost
|
||||
// nothing to copy. Enum values keep their index rather than their decoded flag
|
||||
// names so millions of in-memory calls stay cheap; names are resolved once at
|
||||
// report-write time.
|
||||
struct DecodedArg {
|
||||
const char* name = nullptr;
|
||||
const char* type = nullptr;
|
||||
win32meta::ArgKind kind = win32meta::ArgKind::Unknown;
|
||||
uint32_t enum_index = 0xFFFFFFFFu;
|
||||
uint64_t raw = 0; // the register/stack value as captured
|
||||
uint64_t deref = 0; // pointee, for PtrToInt and friends
|
||||
double fval = 0.0; // for Float/Double params (read from XMM)
|
||||
std::string str; // decoded string contents
|
||||
std::vector<uint8_t> bytes; // bounded buffer preview
|
||||
uint64_t bytes_total = 0; // the buffer's real length, when `bytes` is only a prefix
|
||||
bool bytes_capped = false; // --max-buffer cut the capture short
|
||||
bool has_str = false;
|
||||
bool has_deref = false;
|
||||
bool has_fval = false;
|
||||
bool is_out = false;
|
||||
bool from_return = false; // contents were read at the return position
|
||||
};
|
||||
|
||||
// A dereference deferred until the call returns, so [Out] parameters can be
|
||||
// rendered filled in -- something only a time-travel trace makes easy.
|
||||
struct PendingOut {
|
||||
uint16_t param_index = 0;
|
||||
win32meta::ArgKind kind = win32meta::ArgKind::Unknown;
|
||||
uint64_t ptr = 0;
|
||||
uint16_t pointee_size = 0;
|
||||
win32meta::AuxKind aux_kind = win32meta::AuxKind::None;
|
||||
int32_t aux_value = 0;
|
||||
uint64_t in_cap = 0; // caller-supplied upper bound on length, 0 if unknown
|
||||
};
|
||||
|
||||
struct CallRecord {
|
||||
uint64_t tid = 0; // TTD Thread ID
|
||||
uint64_t seq = 0; // monotonic record order (consistent with timeline order)
|
||||
std::string position; // TTD navigable position "Sequence:Steps" (hex), for WinDbg
|
||||
std::wstring module; // resolved owning module, e.g. "kernel32" (no extension)
|
||||
std::string api; // resolved export name, e.g. "CreateFileA"
|
||||
std::vector<ArgValue> args;
|
||||
std::vector<ArgValue> args; // flat view consumed by capa (ints and strings)
|
||||
std::vector<DecodedArg> params; // rich view; only meaningful when `metadata` is set
|
||||
bool metadata = false; // args came from a real signature, not the heuristic
|
||||
bool has_ret = false;
|
||||
uint64_t ret = 0;
|
||||
// Where the callee will return to, i.e. the instruction after the CALL. Identifies
|
||||
// the call site, which is what tells you who in the sample made the call. The
|
||||
// replay engine hands this to the callback already, so recording it is free.
|
||||
uint64_t return_address = 0;
|
||||
};
|
||||
|
||||
struct ProcessRecord {
|
||||
@@ -72,18 +116,38 @@ namespace ttdcapa {
|
||||
std::vector<std::pair<uint64_t, std::string>> exports; // exported function VA -> export name
|
||||
};
|
||||
|
||||
// One resolved call target: which module it belongs to, its export name, and the
|
||||
// bitness of that module. The bitness is per module rather than per trace because
|
||||
// a WoW64 process runs 32-bit and 64-bit code side by side, and it decides which
|
||||
// calling convention the call's arguments follow.
|
||||
struct ResolvedExport {
|
||||
std::wstring module;
|
||||
std::string api;
|
||||
bool is64 = true;
|
||||
};
|
||||
|
||||
// Navigates all module load events, and returns a map of all function VAs along with their associated module and function name
|
||||
std::unordered_map<uint64_t, std::pair<std::wstring, std::string>> resolveTraceModuleExports(TTD::Replay::UniqueReplayEngine& engine, TTD::Replay::UniqueCursor& cursor);
|
||||
std::unordered_map<uint64_t, ResolvedExport> resolveTraceModuleExports(TTD::Replay::UniqueReplayEngine& engine, TTD::Replay::UniqueCursor& cursor);
|
||||
|
||||
// TTD memory read utility function
|
||||
size_t readMemory(TTD::Replay::UniqueCursor* cursor, TTD::GuestAddress addr, void* dest, unsigned __int64 size);
|
||||
|
||||
// UTF-16 -> UTF-8, for the wide strings TTD and the PE headers hand back
|
||||
std::string convertWstringToString(const std::wstring& ws);
|
||||
|
||||
// Initializes the TTD engine based off a given trace file (.run) path
|
||||
bool initializeTTDEngine(TTD::Replay::UniqueReplayEngine& engine, std::wstring trace_file_path);
|
||||
|
||||
// Attempts to interpret the memory at a certain address as a string. If not a string, will return null
|
||||
std::optional<std::string> tryReadString(TTD::Replay::IThreadView const* thread, uint64_t addr);
|
||||
|
||||
// Read a NUL-terminated string the metadata told us is really there. Unlike
|
||||
// tryReadString these do not guess: no minimum length, and the wide reader
|
||||
// converts real UTF-16 (not just its ASCII subset) to UTF-8. Returns nullopt
|
||||
// only when the memory is unreadable or the bytes aren't a plausible string.
|
||||
std::optional<std::string> readAnsiString(TTD::Replay::IThreadView const* thread, uint64_t addr, size_t maxChars = 512);
|
||||
std::optional<std::string> readWideString(TTD::Replay::IThreadView const* thread, uint64_t addr, size_t maxChars = 512);
|
||||
|
||||
// Attempts to capture an argument as a string. If it doesn't look like a valid string, this function will return the same argument value
|
||||
ArgValue captureCallArg(TTD::Replay::IThreadView const* thread, uint64_t value);
|
||||
}
|
||||
|
||||
+88
-20
@@ -56,18 +56,6 @@ namespace ttdcapa {
|
||||
return result;
|
||||
}
|
||||
|
||||
std::string convertWstringToString(const std::wstring& ws) {
|
||||
if (ws.empty()) {
|
||||
return {};
|
||||
}
|
||||
int needed = ::WideCharToMultiByte(CP_UTF8, 0, ws.data(), static_cast<int>(ws.size()), nullptr, 0, nullptr, nullptr);
|
||||
if (needed <= 0) {
|
||||
return {};
|
||||
}
|
||||
std::string out(static_cast<size_t>(needed), '\0');
|
||||
::WideCharToMultiByte(CP_UTF8, 0, ws.data(), static_cast<int>(ws.size()), out.data(), needed, nullptr, nullptr);
|
||||
return out;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
SampleHashes hashFile(const std::filesystem::path& path) {
|
||||
@@ -130,7 +118,12 @@ namespace ttdcapa {
|
||||
bool ok = getModuleImports(&cursor, mbase, g_report.imports);
|
||||
ok = ttdcapa::getModuleSections(&cursor, mbase, g_report.sections) && ok;
|
||||
|
||||
if (getModuleExports(&cursor, mbase, exps)) {
|
||||
// The main image's PE format is what the report calls the trace's
|
||||
// architecture. A WoW64 trace also holds 64-bit system modules, but the
|
||||
// process it recorded is the 32-bit one.
|
||||
bool mainIs64 = true;
|
||||
if (getModuleExports(&cursor, mbase, exps, &mainIs64)) {
|
||||
g_report.arch = mainIs64 ? "x64" : "x86";
|
||||
for (auto& e : exps) {
|
||||
ExportRecord exportRecord;
|
||||
exportRecord.name = e.second;
|
||||
@@ -156,11 +149,8 @@ namespace ttdcapa {
|
||||
}
|
||||
}
|
||||
|
||||
size_t thread_count = engine->GetThreadCount();
|
||||
TTD::Replay::ThreadInfo const* thread_list = engine->GetThreadList();
|
||||
for (size_t i = 0; i < thread_count; ++i) {
|
||||
g_report.process.threads.push_back(static_cast<uint64_t>(thread_list[i].UniqueId));
|
||||
}
|
||||
// The thread list is filled in by the caller after this returns; collecting it
|
||||
// here too duplicated every entry.
|
||||
}
|
||||
|
||||
bool parse_args(int argc, wchar_t** argv, Options& opt) {
|
||||
@@ -178,6 +168,30 @@ namespace ttdcapa {
|
||||
else if (a == L"--with-stack-args") {
|
||||
opt.with_stack_args = true;
|
||||
}
|
||||
else if (a == L"--ttd-dlls" && i + 1 < argc) {
|
||||
opt.ttd_dlls = argv[++i];
|
||||
}
|
||||
else if (a == L"--win32-index" && i + 1 < argc) {
|
||||
opt.win32_index = argv[++i];
|
||||
}
|
||||
else if (a == L"--no-metadata") {
|
||||
opt.no_metadata = true;
|
||||
}
|
||||
else if (a == L"--max-buffer" && i + 1 < argc) {
|
||||
opt.max_buffer = static_cast<size_t>(std::wcstoull(argv[++i], nullptr, 10));
|
||||
}
|
||||
else if (a == L"--dump-sig" && i + 1 < argc) {
|
||||
opt.dump_sig = convertWstringToString(argv[++i]);
|
||||
}
|
||||
else if ((a == L"-b" || a == L"--binary") && i + 1 < argc) {
|
||||
opt.binary_output = argv[++i];
|
||||
}
|
||||
else if (a == L"--progress") {
|
||||
opt.report_progress = true;
|
||||
}
|
||||
else if (a == L"--cancel-on-stdin") {
|
||||
opt.cancel_on_stdin = true;
|
||||
}
|
||||
else if (!a.empty() && a[0] == L'-') {
|
||||
std::cerr << "unknown option\n";
|
||||
return false;
|
||||
@@ -190,7 +204,8 @@ namespace ttdcapa {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return !opt.trace.empty();
|
||||
// --dump-sig is a metadata-only debug mode, so it doesn't need a trace.
|
||||
return !opt.trace.empty() || !opt.dump_sig.empty();
|
||||
}
|
||||
|
||||
bool writeReport(std::filesystem::path outputFilePath) {
|
||||
@@ -239,6 +254,9 @@ namespace ttdcapa {
|
||||
call["module"] = moduleStr;
|
||||
|
||||
call["api"] = callRecord.api;
|
||||
// Always an array, even when empty -- a zero-parameter function is a
|
||||
// real result, not a missing field.
|
||||
call["args"] = json::array();
|
||||
for (ArgValue& argValue : callRecord.args) {
|
||||
if (std::holds_alternative<int64_t>(argValue)) {
|
||||
call["args"].push_back(std::get<int64_t>(argValue));
|
||||
@@ -247,7 +265,53 @@ namespace ttdcapa {
|
||||
}
|
||||
}
|
||||
|
||||
// The rich, metadata-derived view. capa matches on "args"; this is for
|
||||
// ttd-timeline and human triage, so it can afford to be verbose. Absent
|
||||
// entirely when we had no signature, so consumers can tell "no
|
||||
// parameters" apart from "we didn't know".
|
||||
if (callRecord.metadata) {
|
||||
call["params"] = json::array();
|
||||
}
|
||||
for (DecodedArg& decoded : callRecord.params) {
|
||||
json param;
|
||||
param["name"] = decoded.name ? decoded.name : "";
|
||||
param["type"] = decoded.type ? decoded.type : "";
|
||||
param["kind"] = win32meta::kindName(decoded.kind);
|
||||
param["value"] = decoded.raw;
|
||||
if (decoded.has_str) {
|
||||
param["str"] = decoded.str;
|
||||
}
|
||||
if (decoded.has_deref) {
|
||||
param["deref"] = decoded.deref;
|
||||
}
|
||||
if (decoded.has_fval) {
|
||||
param["float"] = decoded.fval;
|
||||
}
|
||||
if (decoded.enum_index != 0xFFFFFFFFu) {
|
||||
std::vector<std::string> names = win32meta::index().decodeEnum(decoded.enum_index, decoded.raw);
|
||||
if (!names.empty()) {
|
||||
param["flags"] = names;
|
||||
}
|
||||
}
|
||||
if (!decoded.bytes.empty()) {
|
||||
param["bytes"] = to_hex(decoded.bytes);
|
||||
// Only when --max-buffer actually cut it short: a string buffer
|
||||
// trimmed at its NUL is complete, not truncated.
|
||||
if (decoded.bytes_capped) {
|
||||
param["bytes_total"] = decoded.bytes_total;
|
||||
}
|
||||
}
|
||||
if (decoded.is_out) {
|
||||
param["out"] = true;
|
||||
}
|
||||
if (decoded.from_return) {
|
||||
param["at_return"] = true;
|
||||
}
|
||||
call["params"].push_back(std::move(param));
|
||||
}
|
||||
|
||||
call["ret"] = callRecord.has_ret ? callRecord.ret : 0;
|
||||
call["return_address"] = callRecord.return_address;
|
||||
|
||||
process["calls"].push_back(call);
|
||||
}
|
||||
@@ -260,8 +324,12 @@ namespace ttdcapa {
|
||||
return false;
|
||||
}
|
||||
|
||||
reportFile << report.dump() << std::endl;
|
||||
// Recovered guest memory can always surprise us with a byte sequence that
|
||||
// isn't valid UTF-8. Substituting it beats throwing away an entire trace's
|
||||
// worth of extraction over one bad string.
|
||||
reportFile << report.dump(-1, ' ', false, nlohmann::json::error_handler_t::replace) << std::endl;
|
||||
reportFile.close();
|
||||
std::cerr << "DONE!\n";
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
+29
-3
@@ -31,9 +31,35 @@ namespace ttdcapa {
|
||||
|
||||
struct Options {
|
||||
std::filesystem::path trace;
|
||||
std::filesystem::path sample; // optional on-disk sample for hashing
|
||||
std::filesystem::path output; // empty will output to stdout
|
||||
uint64_t max_calls = 0; // 0 means unlimited
|
||||
std::filesystem::path sample; // optional on-disk sample for hashing
|
||||
std::filesystem::path output; // JSON report; empty means none
|
||||
// Binary report (see binreport.hpp). Much faster to write and effectively free to
|
||||
// load, but only the JSON form is what capa consumes.
|
||||
std::filesystem::path binary_output;
|
||||
std::filesystem::path win32_index; // empty means search the default locations
|
||||
// Directory holding TTDReplay.dll and TTDReplayCPU.dll -- normally WinDbg's
|
||||
// amd64 td folder. Those are Microsoft's and not ours to redistribute, so
|
||||
// rather than requiring them beside the executable, a caller points at wherever
|
||||
// WinDbg already put them. TTDReplay.dll is delay-loaded so this can take effect
|
||||
// at runtime; without the flag the usual DLL search order applies.
|
||||
std::filesystem::path ttd_dlls;
|
||||
std::string dump_sig; // print one signature and exit; no trace needed
|
||||
uint64_t max_calls = 0; // 0 means unlimited
|
||||
// Bytes kept from any one counted buffer. Generous on purpose: buffer parameters
|
||||
// are well under 1% of the calls in a typical trace -- on a 654 MB report measured
|
||||
// here, 23k buffers across 3.4M calls totalling 1.1 MB -- so the cap is nowhere
|
||||
// near the dominant cost, while a buffer truncated below the interesting part is
|
||||
// simply lost. 64 KiB covers typical socket and file reads whole.
|
||||
size_t max_buffer = 65536;
|
||||
bool no_metadata = false; // force the pre-metadata heuristic capture
|
||||
// Emit "[progress] <percent> <calls>" lines on stderr as the sweep runs, and
|
||||
// interrupt it when a line reading "cancel" arrives on stdin -- writing out
|
||||
// whatever was collected up to that point. Both are for a UI driving this as a
|
||||
// child process; interactive use is unaffected.
|
||||
bool report_progress = false;
|
||||
bool cancel_on_stdin = false;
|
||||
// Only affects calls with no metadata: blindly grab four extra stack slots.
|
||||
// Functions we have a signature for always capture their true arity.
|
||||
bool with_stack_args = false;
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,277 @@
|
||||
#include "win32meta.hpp"
|
||||
|
||||
#ifndef WIN32_LEAN_AND_MEAN
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#endif
|
||||
#include <windows.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
|
||||
namespace ttdcapa::win32meta {
|
||||
namespace {
|
||||
constexpr char kMagic[8] = { 'W', '3', '2', 'I', 'D', 'X', '0', '1' };
|
||||
constexpr uint32_t kFormatVersion = 1;
|
||||
|
||||
// Record sizes must match the struct.pack formats in tools/build-win32-index.py.
|
||||
constexpr size_t kHeaderSize = 32;
|
||||
constexpr size_t kFuncRecSize = 16;
|
||||
constexpr size_t kParamRecSize = 24;
|
||||
constexpr size_t kEnumRecSize = 16;
|
||||
constexpr size_t kEnumValRecSize = 16;
|
||||
|
||||
template <typename T>
|
||||
T readAt(const uint8_t* p, size_t off) {
|
||||
T v{};
|
||||
std::memcpy(&v, p + off, sizeof(T));
|
||||
return v;
|
||||
}
|
||||
|
||||
int popcount64(uint64_t v) {
|
||||
int n = 0;
|
||||
while (v) {
|
||||
v &= v - 1;
|
||||
++n;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
bool Index::load(const std::filesystem::path& path, std::string& error) {
|
||||
std::ifstream f(path, std::ios::binary);
|
||||
if (!f) {
|
||||
error = "cannot open index";
|
||||
return false;
|
||||
}
|
||||
std::vector<uint8_t> blob((std::istreambuf_iterator<char>(f)), std::istreambuf_iterator<char>());
|
||||
if (blob.size() < kHeaderSize || std::memcmp(blob.data(), kMagic, sizeof(kMagic)) != 0) {
|
||||
error = "not a win32 index file (bad magic)";
|
||||
return false;
|
||||
}
|
||||
|
||||
const uint8_t* p = blob.data();
|
||||
uint32_t version = readAt<uint32_t>(p, 8);
|
||||
if (version != kFormatVersion) {
|
||||
error = "index format version " + std::to_string(version) +
|
||||
", expected " + std::to_string(kFormatVersion) + " (regenerate with tools/build-win32-index.py)";
|
||||
return false;
|
||||
}
|
||||
|
||||
uint32_t funcCount = readAt<uint32_t>(p, 12);
|
||||
uint32_t paramCount = readAt<uint32_t>(p, 16);
|
||||
uint32_t enumCount = readAt<uint32_t>(p, 20);
|
||||
uint32_t enumValCount = readAt<uint32_t>(p, 24);
|
||||
uint32_t strtabSize = readAt<uint32_t>(p, 28);
|
||||
|
||||
size_t funcOff = kHeaderSize;
|
||||
size_t paramOff = funcOff + static_cast<size_t>(funcCount) * kFuncRecSize;
|
||||
size_t enumOff = paramOff + static_cast<size_t>(paramCount) * kParamRecSize;
|
||||
size_t enumValOff = enumOff + static_cast<size_t>(enumCount) * kEnumRecSize;
|
||||
size_t strOff = enumValOff + static_cast<size_t>(enumValCount) * kEnumValRecSize;
|
||||
if (strOff + strtabSize != blob.size()) {
|
||||
error = "index is truncated or corrupt";
|
||||
return false;
|
||||
}
|
||||
// Every string offset is dereferenced without a bounds check below, so the
|
||||
// pool must be NUL-terminated for that to be safe.
|
||||
if (strtabSize == 0 || blob[blob.size() - 1] != 0) {
|
||||
error = "index string pool is not NUL-terminated";
|
||||
return false;
|
||||
}
|
||||
|
||||
blob_ = std::move(blob);
|
||||
p = blob_.data();
|
||||
const char* strtab = reinterpret_cast<const char*>(p + strOff);
|
||||
auto str = [&](uint32_t off) -> const char* {
|
||||
return off < strtabSize ? strtab + off : "";
|
||||
};
|
||||
|
||||
params_.resize(paramCount);
|
||||
for (uint32_t i = 0; i < paramCount; ++i) {
|
||||
size_t o = paramOff + static_cast<size_t>(i) * kParamRecSize;
|
||||
ParamSig& ps = params_[i];
|
||||
ps.name = str(readAt<uint32_t>(p, o));
|
||||
ps.type = str(readAt<uint32_t>(p, o + 4));
|
||||
ps.kind = static_cast<ArgKind>(p[o + 8]);
|
||||
ps.attrs = p[o + 9];
|
||||
ps.slot = p[o + 10];
|
||||
ps.auxKind = static_cast<AuxKind>(p[o + 11]);
|
||||
ps.auxValue = readAt<int32_t>(p, o + 12);
|
||||
ps.enumIndex = readAt<uint32_t>(p, o + 16);
|
||||
ps.pointeeSize = readAt<uint16_t>(p, o + 20);
|
||||
if (ps.enumIndex != 0xFFFFFFFFu && ps.enumIndex >= enumCount) {
|
||||
ps.enumIndex = 0xFFFFFFFFu;
|
||||
}
|
||||
}
|
||||
|
||||
funcs_.resize(funcCount);
|
||||
byName_.reserve(funcCount * 2);
|
||||
for (uint32_t i = 0; i < funcCount; ++i) {
|
||||
size_t o = funcOff + static_cast<size_t>(i) * kFuncRecSize;
|
||||
FuncSig& fs = funcs_[i];
|
||||
fs.name = str(readAt<uint32_t>(p, o));
|
||||
fs.dll = str(readAt<uint32_t>(p, o + 4));
|
||||
uint32_t firstParam = readAt<uint32_t>(p, o + 8);
|
||||
fs.paramCount = p[o + 12];
|
||||
fs.flags = p[o + 13];
|
||||
if (static_cast<size_t>(firstParam) + fs.paramCount > params_.size()) {
|
||||
error = "index parameter range out of bounds";
|
||||
return false;
|
||||
}
|
||||
fs.params = fs.paramCount ? ¶ms_[firstParam] : nullptr;
|
||||
byName_.emplace(std::string_view(fs.name), i);
|
||||
}
|
||||
|
||||
enumValues_.resize(enumValCount);
|
||||
for (uint32_t i = 0; i < enumValCount; ++i) {
|
||||
size_t o = enumValOff + static_cast<size_t>(i) * kEnumValRecSize;
|
||||
enumValues_[i].name = str(readAt<uint32_t>(p, o));
|
||||
enumValues_[i].value = readAt<int64_t>(p, o + 8);
|
||||
}
|
||||
|
||||
enums_.resize(enumCount);
|
||||
for (uint32_t i = 0; i < enumCount; ++i) {
|
||||
size_t o = enumOff + static_cast<size_t>(i) * kEnumRecSize;
|
||||
EnumTable& et = enums_[i];
|
||||
et.name = str(readAt<uint32_t>(p, o));
|
||||
et.valueOffset = readAt<uint32_t>(p, o + 4);
|
||||
et.valueCount = readAt<uint32_t>(p, o + 8);
|
||||
et.isFlags = p[o + 12] != 0;
|
||||
et.width = p[o + 13];
|
||||
if (static_cast<size_t>(et.valueOffset) + et.valueCount > enumValues_.size()) {
|
||||
et.valueOffset = 0;
|
||||
et.valueCount = 0;
|
||||
}
|
||||
}
|
||||
|
||||
path_ = path;
|
||||
return true;
|
||||
}
|
||||
|
||||
const FuncSig* Index::lookup(std::string_view api) const {
|
||||
auto it = byName_.find(api);
|
||||
return it == byName_.end() ? nullptr : &funcs_[it->second];
|
||||
}
|
||||
|
||||
const char* Index::enumName(uint32_t enumIndex) const {
|
||||
return enumIndex < enums_.size() ? enums_[enumIndex].name : "";
|
||||
}
|
||||
|
||||
std::vector<std::string> Index::decodeEnum(uint32_t enumIndex, uint64_t value) const {
|
||||
std::vector<std::string> out;
|
||||
if (enumIndex >= enums_.size()) {
|
||||
return out;
|
||||
}
|
||||
const EnumTable& et = enums_[enumIndex];
|
||||
// The captured value is a full 64-bit register; mask to the enum's real width
|
||||
// so sign-extension and upper garbage don't defeat the comparisons.
|
||||
uint64_t mask = et.width >= 8 ? ~0ull : ((1ull << (et.width * 8)) - 1);
|
||||
uint64_t v = value & mask;
|
||||
|
||||
const EnumValue* vals = enumValues_.data() + et.valueOffset;
|
||||
for (uint32_t i = 0; i < et.valueCount; ++i) {
|
||||
if ((static_cast<uint64_t>(vals[i].value) & mask) == v) {
|
||||
out.emplace_back(vals[i].name);
|
||||
return out; // exact match wins, flags or not
|
||||
}
|
||||
}
|
||||
if (!et.isFlags || v == 0) {
|
||||
return out;
|
||||
}
|
||||
|
||||
// Greedy decomposition: consume the widest matching bit groups first so
|
||||
// composites like GENERIC_WRITE beat their individual constituent bits.
|
||||
std::vector<uint32_t> order(et.valueCount);
|
||||
for (uint32_t i = 0; i < et.valueCount; ++i) {
|
||||
order[i] = i;
|
||||
}
|
||||
std::sort(order.begin(), order.end(), [&](uint32_t a, uint32_t b) {
|
||||
return popcount64(static_cast<uint64_t>(vals[a].value) & mask) >
|
||||
popcount64(static_cast<uint64_t>(vals[b].value) & mask);
|
||||
});
|
||||
|
||||
uint64_t remaining = v;
|
||||
for (uint32_t i : order) {
|
||||
uint64_t bits = static_cast<uint64_t>(vals[i].value) & mask;
|
||||
if (bits != 0 && (remaining & bits) == bits) {
|
||||
out.emplace_back(vals[i].name);
|
||||
remaining &= ~bits;
|
||||
}
|
||||
}
|
||||
if (remaining != 0) {
|
||||
char buf[32];
|
||||
std::snprintf(buf, sizeof(buf), "0x%llx", static_cast<unsigned long long>(remaining));
|
||||
out.emplace_back(buf);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
Index& index() {
|
||||
static Index instance;
|
||||
return instance;
|
||||
}
|
||||
|
||||
namespace {
|
||||
std::filesystem::path executableDir() {
|
||||
wchar_t buf[MAX_PATH * 4];
|
||||
DWORD n = ::GetModuleFileNameW(nullptr, buf, static_cast<DWORD>(std::size(buf)));
|
||||
if (n == 0 || n >= std::size(buf)) {
|
||||
return {};
|
||||
}
|
||||
return std::filesystem::path(buf, buf + n).parent_path();
|
||||
}
|
||||
} // namespace
|
||||
|
||||
bool loadIndex(const std::filesystem::path& explicitPath, std::string& error) {
|
||||
std::vector<std::filesystem::path> candidates;
|
||||
if (!explicitPath.empty()) {
|
||||
candidates.push_back(explicitPath);
|
||||
} else {
|
||||
std::filesystem::path dir = executableDir();
|
||||
if (!dir.empty()) {
|
||||
candidates.push_back(dir / L"win32-index.bin");
|
||||
// running straight out of ttd\bin\<plat>\<config>\ during development
|
||||
candidates.push_back(dir / L".." / L".." / L".." / L"data" / L"win32-index.bin");
|
||||
}
|
||||
}
|
||||
|
||||
std::error_code ec;
|
||||
for (const auto& c : candidates) {
|
||||
if (!std::filesystem::exists(c, ec)) {
|
||||
continue;
|
||||
}
|
||||
if (index().load(c, error)) {
|
||||
return true;
|
||||
}
|
||||
return false; // found but unusable: surface the real reason
|
||||
}
|
||||
error = "win32-index.bin not found (run tools/build-win32-index.py, or pass --win32-index)";
|
||||
return false;
|
||||
}
|
||||
|
||||
const char* kindName(ArgKind kind) {
|
||||
switch (kind) {
|
||||
case ArgKind::Integer: return "int";
|
||||
case ArgKind::Bool: return "bool";
|
||||
case ArgKind::Handle: return "handle";
|
||||
case ArgKind::Enum: return "enum";
|
||||
case ArgKind::Float: return "float";
|
||||
case ArgKind::Double: return "double";
|
||||
case ArgKind::AnsiString: return "str";
|
||||
case ArgKind::WideString: return "wstr";
|
||||
case ArgKind::AnsiBuffer: return "strbuf";
|
||||
case ArgKind::WideBuffer: return "wstrbuf";
|
||||
case ArgKind::ByteBuffer: return "buf";
|
||||
case ArgKind::PtrToInt: return "int*";
|
||||
case ArgKind::StructPtr: return "struct*";
|
||||
case ArgKind::FuncPtr: return "fnptr";
|
||||
case ArgKind::Guid: return "guid";
|
||||
case ArgKind::Pointer: return "ptr";
|
||||
case ArgKind::PtrToAnsiString: return "str*";
|
||||
case ArgKind::PtrToWideString: return "wstr*";
|
||||
case ArgKind::Unknown:
|
||||
default: return "unknown";
|
||||
}
|
||||
}
|
||||
} // namespace ttdcapa::win32meta
|
||||
@@ -0,0 +1,156 @@
|
||||
#ifndef WIN32META_HPP
|
||||
#define WIN32META_HPP
|
||||
|
||||
// Loader for the pre-baked Win32 API metadata index (win32-index.bin) produced by
|
||||
// tools/build-win32-index.py from the win32json submodule.
|
||||
//
|
||||
// The index answers one question per recorded call: "given this export name, what
|
||||
// are its parameters?" -- count, names, types, direction, and how to decode each
|
||||
// one. That turns the extractor's blind 4-register grab into an exact capture.
|
||||
//
|
||||
// Lookup is keyed on the bare function name. Module is carried for display only:
|
||||
// API sets mean the metadata's "CreateFileW -> KERNEL32.dll" shows up in a trace as
|
||||
// KERNELBASE.dll!CreateFileW, so matching on module would lose most calls.
|
||||
// See WIN32JSON-TTD-INTEGRATION-NOTES.md section 4.
|
||||
|
||||
#include <cstdint>
|
||||
#include <filesystem>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
namespace ttdcapa::win32meta {
|
||||
|
||||
// How to decode one parameter. Values must match the K_* constants in
|
||||
// tools/build-win32-index.py.
|
||||
enum class ArgKind : uint8_t {
|
||||
Unknown = 0,
|
||||
Integer, // plain scalar, never dereferenced
|
||||
Bool,
|
||||
Handle, // opaque HANDLE/HKEY/HWND; pointer-sized but not a pointer
|
||||
Enum, // scalar with a symbolic value table (see enumIndex)
|
||||
Float, // 4-byte float, arrives in XMM<slot>
|
||||
Double, // 8-byte float, arrives in XMM<slot>
|
||||
AnsiString, // char*, NUL-terminated
|
||||
WideString, // wchar_t*, NUL-terminated
|
||||
AnsiBuffer, // char[], length from aux
|
||||
WideBuffer, // wchar_t[], length from aux
|
||||
ByteBuffer, // void*/byte[], length from aux
|
||||
PtrToInt, // pointer to a pointeeSize-byte scalar
|
||||
StructPtr, // pointer to a struct/union (not expanded in v1)
|
||||
FuncPtr,
|
||||
Guid, // pointer to a 16-byte GUID
|
||||
Pointer, // opaque pointer, contents unknown
|
||||
PtrToAnsiString, // char** -- out-param that receives an allocated string
|
||||
PtrToWideString, // wchar_t**
|
||||
};
|
||||
|
||||
// Bitmask of a parameter's direction/nullability attributes.
|
||||
enum ParamAttr : uint8_t {
|
||||
AttrIn = 0x01,
|
||||
AttrOut = 0x02,
|
||||
AttrOptional = 0x04,
|
||||
AttrConst = 0x08,
|
||||
AttrReserved = 0x10,
|
||||
AttrNotNulTerminated = 0x20,
|
||||
AttrNulNulTerminated = 0x40,
|
||||
AttrComOutPtr = 0x80,
|
||||
};
|
||||
|
||||
// Where a buffer parameter's length comes from.
|
||||
enum class AuxKind : uint8_t {
|
||||
None = 0,
|
||||
BytesFromParam, // auxValue is the index of a parameter holding a byte count
|
||||
CountFromParam, // auxValue is the index of a parameter holding an element count
|
||||
CountConst, // auxValue is the element count itself
|
||||
};
|
||||
|
||||
enum FuncFlag : uint8_t {
|
||||
FlagHiddenRetPtr = 0x01, // returns a large aggregate; RCX is the hidden return buffer
|
||||
FlagUnsupported = 0x02, // at least one parameter could not be classified
|
||||
FlagSetLastError = 0x04,
|
||||
};
|
||||
|
||||
struct ParamSig {
|
||||
const char* name = "";
|
||||
const char* type = "";
|
||||
ArgKind kind = ArgKind::Unknown;
|
||||
uint8_t attrs = 0;
|
||||
uint8_t slot = 0; // positional ABI slot, already shifted for FlagHiddenRetPtr
|
||||
AuxKind auxKind = AuxKind::None;
|
||||
int32_t auxValue = 0;
|
||||
uint32_t enumIndex = 0xFFFFFFFFu;
|
||||
uint16_t pointeeSize = 0; // bytes per pointee/element, 0 if unknown
|
||||
|
||||
bool isIn() const { return (attrs & AttrIn) != 0; }
|
||||
bool isOut() const { return (attrs & AttrOut) != 0; }
|
||||
bool isFloat() const { return kind == ArgKind::Float || kind == ArgKind::Double; }
|
||||
bool hasEnum() const { return enumIndex != 0xFFFFFFFFu; }
|
||||
};
|
||||
|
||||
struct FuncSig {
|
||||
const char* name = "";
|
||||
const char* dll = "";
|
||||
const ParamSig* params = nullptr;
|
||||
uint8_t paramCount = 0;
|
||||
uint8_t flags = 0;
|
||||
|
||||
bool unsupported() const { return (flags & FlagUnsupported) != 0; }
|
||||
bool hiddenRetPtr() const { return (flags & FlagHiddenRetPtr) != 0; }
|
||||
};
|
||||
|
||||
// Loaded once at startup and read-only thereafter, so the replay sweep can hit
|
||||
// it from the call callback without synchronisation.
|
||||
class Index {
|
||||
public:
|
||||
bool load(const std::filesystem::path& path, std::string& error);
|
||||
bool loaded() const { return !funcs_.empty(); }
|
||||
size_t functionCount() const { return funcs_.size(); }
|
||||
const std::filesystem::path& path() const { return path_; }
|
||||
|
||||
// Bare export name, e.g. "CreateFileW". Returns nullptr when unknown, which
|
||||
// is the caller's cue to fall back to the heuristic capture.
|
||||
const FuncSig* lookup(std::string_view api) const;
|
||||
|
||||
const char* enumName(uint32_t enumIndex) const;
|
||||
|
||||
// Symbolic names for `value`: an exact match if one exists, otherwise a
|
||||
// greedy bit decomposition for flag enums. Empty when nothing matches.
|
||||
std::vector<std::string> decodeEnum(uint32_t enumIndex, uint64_t value) const;
|
||||
|
||||
private:
|
||||
struct EnumTable {
|
||||
const char* name = "";
|
||||
uint32_t valueOffset = 0;
|
||||
uint32_t valueCount = 0;
|
||||
bool isFlags = false;
|
||||
uint8_t width = 4;
|
||||
};
|
||||
struct EnumValue {
|
||||
const char* name = "";
|
||||
int64_t value = 0;
|
||||
};
|
||||
|
||||
std::vector<uint8_t> blob_;
|
||||
std::vector<FuncSig> funcs_;
|
||||
std::vector<ParamSig> params_;
|
||||
std::vector<EnumTable> enums_;
|
||||
std::vector<EnumValue> enumValues_;
|
||||
std::unordered_map<std::string_view, uint32_t> byName_;
|
||||
std::filesystem::path path_;
|
||||
};
|
||||
|
||||
// Process-wide instance; empty until loadIndex() succeeds.
|
||||
Index& index();
|
||||
|
||||
// Try `explicitPath` if non-empty, else the conventional locations next to the
|
||||
// executable and in the source tree. Returns false with `error` set; the caller
|
||||
// is expected to warn and continue in heuristic mode.
|
||||
bool loadIndex(const std::filesystem::path& explicitPath, std::string& error);
|
||||
|
||||
const char* kindName(ArgKind kind);
|
||||
|
||||
} // namespace ttdcapa::win32meta
|
||||
|
||||
#endif
|
||||
@@ -1,4 +1,4 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
@@ -22,13 +22,13 @@
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'" Label="Configuration">
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
<PlatformToolset>v145</PlatformToolset>
|
||||
<PlatformToolset Condition="'$(PlatformToolset)'==''">v145</PlatformToolset>
|
||||
<CharacterSet>Unicode</CharacterSet>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'" Label="Configuration">
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<UseDebugLibraries>false</UseDebugLibraries>
|
||||
<PlatformToolset>v145</PlatformToolset>
|
||||
<PlatformToolset Condition="'$(PlatformToolset)'==''">v145</PlatformToolset>
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
<CharacterSet>Unicode</CharacterSet>
|
||||
</PropertyGroup>
|
||||
@@ -55,7 +55,8 @@
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>TTDReplay.lib;bcrypt.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalDependencies>TTDReplay.lib;bcrypt.lib;delayimp.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<DelayLoadDLLs>TTDReplay.dll;%(DelayLoadDLLs)</DelayLoadDLLs>
|
||||
</Link>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
@@ -70,15 +71,21 @@
|
||||
</Link>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<ClCompile Include="src\abi.cpp" />
|
||||
<ClCompile Include="src\binreport.cpp" />
|
||||
<ClCompile Include="src\main.cpp" />
|
||||
<ClCompile Include="src\ttd_pe_utils.cpp" />
|
||||
<ClCompile Include="src\ttdutils.cpp" />
|
||||
<ClCompile Include="src\utils.cpp" />
|
||||
<ClCompile Include="src\win32meta.cpp" />
|
||||
</ItemGroup>
|
||||
<ItemGroup>
|
||||
<ClInclude Include="src\abi.hpp" />
|
||||
<ClInclude Include="src\binreport.hpp" />
|
||||
<ClInclude Include="src\ttd_pe_utils.hpp" />
|
||||
<ClInclude Include="src\ttdutils.hpp" />
|
||||
<ClInclude Include="src\utils.hpp" />
|
||||
<ClInclude Include="src\win32meta.hpp" />
|
||||
</ItemGroup>
|
||||
<ItemGroup>
|
||||
<None Include="packages.config" />
|
||||
@@ -101,6 +108,13 @@
|
||||
</ItemGroup>
|
||||
<Copy SourceFiles="@(TTDRuntime)" DestinationFolder="$(OutDir)" SkipUnchangedFiles="true" Condition="'@(TTDRuntime)'!=''" ContinueOnError="true" />
|
||||
</Target>
|
||||
<!-- The Win32 metadata index the argument decoder reads at startup. Regenerate
|
||||
with tools\build-win32-index.py after bumping the win32json submodule; the
|
||||
extractor still runs (heuristically) if it is absent. -->
|
||||
<Target Name="CopyWin32Index" AfterTargets="Build">
|
||||
<Copy SourceFiles="data\win32-index.bin" DestinationFolder="$(OutDir)" SkipUnchangedFiles="true" Condition="Exists('data\win32-index.bin')" />
|
||||
<Warning Condition="!Exists('data\win32-index.bin')" Text="data\win32-index.bin is missing; run 'python tools\build-win32-index.py' to enable metadata-driven argument decoding." />
|
||||
</Target>
|
||||
<Import Project="packages\nlohmann.json.3.12.0\build\native\nlohmann.json.targets" Condition="Exists('packages\nlohmann.json.3.12.0\build\native\nlohmann.json.targets')" />
|
||||
<Target Name="EnsureNuGetPackageBuildImports" BeforeTargets="PrepareForBuild">
|
||||
<PropertyGroup>
|
||||
|
||||
@@ -1,9 +1,15 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<ItemGroup>
|
||||
<ClCompile Include="src\abi.cpp">
|
||||
<Filter>Source Files</Filter>
|
||||
</ClCompile>
|
||||
<ClCompile Include="src\main.cpp">
|
||||
<Filter>Source Files</Filter>
|
||||
</ClCompile>
|
||||
<ClCompile Include="src\win32meta.cpp">
|
||||
<Filter>Source Files</Filter>
|
||||
</ClCompile>
|
||||
<ClCompile Include="src\ttd_pe_utils.cpp">
|
||||
<Filter>Source Files</Filter>
|
||||
</ClCompile>
|
||||
@@ -13,17 +19,29 @@
|
||||
<ClCompile Include="src\utils.cpp">
|
||||
<Filter>Source Files</Filter>
|
||||
</ClCompile>
|
||||
<ClCompile Include="src\binreport.cpp">
|
||||
<Filter>Source Files</Filter>
|
||||
</ClCompile>
|
||||
</ItemGroup>
|
||||
<ItemGroup>
|
||||
<ClInclude Include="src\abi.hpp">
|
||||
<Filter>Header Files</Filter>
|
||||
</ClInclude>
|
||||
<ClInclude Include="src\ttd_pe_utils.hpp">
|
||||
<Filter>Header Files</Filter>
|
||||
</ClInclude>
|
||||
<ClInclude Include="src\win32meta.hpp">
|
||||
<Filter>Header Files</Filter>
|
||||
</ClInclude>
|
||||
<ClInclude Include="src\ttdutils.hpp">
|
||||
<Filter>Header Files</Filter>
|
||||
</ClInclude>
|
||||
<ClInclude Include="src\utils.hpp">
|
||||
<Filter>Header Files</Filter>
|
||||
</ClInclude>
|
||||
<ClInclude Include="src\binreport.hpp">
|
||||
<Filter>Header Files</Filter>
|
||||
</ClInclude>
|
||||
</ItemGroup>
|
||||
<ItemGroup>
|
||||
<None Include="packages.config" />
|
||||
|
||||
Submodule
+1
Submodule win32json added at 071df49952
Reference in New Issue
Block a user