#!/usr/bin/env python3
"""Dump bytecode for Python source files as JSON.
Designed to compare raw bytecode streams across different Python
implementations while normalizing only display-only details such as memory
addresses in argument reprs.
Usage:
python dis_dump.py Lib/
python dis_dump.py --base-dir Lib path/to/file.py
python dis_dump.py --base-dir Lib --output dump.json path/to/file.py
"""
import argparse
import ast
import builtins
import dis
import json
import os
import re
import sys
import types
# Raw bytecode parity mode: do not skip any instructions.
SKIP_OPS = frozenset()
_OPNAME_NORMALIZE = {}
_SUPER_DECOMPOSE = {}
# Jump instruction names (fallback when hasjrel/hasjabs is incomplete)
_JUMP_OPNAMES = frozenset(
{
"JUMP",
"JUMP_FORWARD",
"JUMP_BACKWARD",
"JUMP_BACKWARD_NO_INTERRUPT",
"POP_JUMP_IF_TRUE",
"POP_JUMP_IF_FALSE",
"POP_JUMP_IF_NONE",
"POP_JUMP_IF_NOT_NONE",
"JUMP_IF_TRUE_OR_POP",
"JUMP_IF_FALSE_OR_POP",
"FOR_ITER",
"END_ASYNC_FOR",
"SEND",
}
)
_JUMP_OPCODES = None
_ABSOLUTE_JUMP_OPCODES = frozenset(getattr(dis, "hasjabs", ()))
def _jump_opcodes():
global _JUMP_OPCODES
if _JUMP_OPCODES is None:
_JUMP_OPCODES = set()
if hasattr(dis, "hasjrel"):
_JUMP_OPCODES.update(dis.hasjrel)
if hasattr(dis, "hasjabs"):
_JUMP_OPCODES.update(dis.hasjabs)
return _JUMP_OPCODES
def _is_jump(inst):
"""Check if an instruction is a jump (by opcode set or name)."""
return inst.opcode in _jump_opcodes() or inst.opname in _JUMP_OPNAMES
def _normalize_argrepr(argrepr):
"""Strip runtime-specific details from arg repr."""
if argrepr.startswith("").strip()
# Remove memory addresses from other reprs
argrepr = re.sub(r" at 0x[0-9a-fA-F]+", "", argrepr)
# Remove LOAD_ATTR/LOAD_SUPER_ATTR suffixes: " + NULL|self", " + NULL"
argrepr = re.sub(r" \+ NULL\|self$", "", argrepr)
argrepr = re.sub(r" \+ NULL$", "", argrepr)
# Normalize unicode escapes
def _unescape(m):
try:
cp = int(m.group(1), 16)
if 0xD800 %d" % target_idx])
elif inst.opname == "COMPARE_OP":
if _IS_RUSTPYTHON:
cmp_idx = inst.arg >> 5 if isinstance(inst.arg, int) else inst.arg
cmp_str = _RP_CMP_OPS.get(cmp_idx, inst.argrepr)
if isinstance(inst.arg, int) and inst.arg & 16:
cmp_str = f"bool({cmp_str})"
else:
cmp_str = inst.argrepr if inst.argrepr else str(inst.arg)
result.append([opname, cmp_str])
elif inst.arg is not None and inst.argrepr:
# If argrepr is just a number, try to resolve it via fallback
# (RustPython may return raw index instead of variable name)
argrepr = inst.argrepr
if argrepr.isdigit() or (argrepr.startswith("-") and argrepr[1:].isdigit()):
resolved = _resolve_arg_fallback(code, opname, inst.arg)
if isinstance(resolved, str) and not resolved.isdigit():
argrepr = resolved
result.append([opname, _normalize_argrepr(argrepr)])
elif inst.arg is not None:
resolved = _resolve_arg_fallback(code, opname, inst.arg)
result.append([opname, resolved])
else:
result.append([opname])
return result
def _dump_code(code):
"""Recursively dump a code object and its nested code objects."""
name = getattr(code, "co_qualname", None) or code.co_name
children = [_dump_code(c) for c in code.co_consts if isinstance(c, types.CodeType)]
r = {"name": name, "insts": _extract_instructions(code)}
if children:
r["children"] = children
return r
def process_file(path):
"""Compile a single file and return its bytecode dump."""
try:
with open(path, "rb") as f:
source = f.read()
code = compile(source, path, "exec")
return {"status": "ok", "code": _dump_code(code)}
except SyntaxError as e:
return {"status": "error", "error": "%s (line %s)" % (e.msg, e.lineno)}
except Exception as e:
return {"status": "error", "error": str(e)}
def main():
parser = argparse.ArgumentParser(description="Dump normalized bytecode as JSON")
parser.add_argument(
"--base-dir",
default=None,
help="Base directory used to compute relative output paths",
)
parser.add_argument(
"--files-from",
default=None,
help="Read newline-separated target paths from this file",
)
parser.add_argument(
"targets", nargs="*", help="Python files or directories to process"
)
parser.add_argument(
"--progress",
type=int,
default=0,
help="Print a dot to stderr every N files processed",
)
parser.add_argument(
"--output",
default=None,
help="Write JSON output to this file instead of stdout",
)
args = parser.parse_args()
targets = list(args.targets)
if args.files_from:
with open(args.files_from, encoding="utf-8") as f:
targets.extend(line.strip() for line in f if line.strip())
results = {}
count = 0
for target in targets:
if os.path.isdir(target):
for root, dirs, files in os.walk(target):
dirs[:] = sorted(
d for d in dirs if d != "__pycache__" and not d.startswith(".")
)
for fname in sorted(files):
if fname.endswith(".py"):
fpath = os.path.join(root, fname)
rel_base = args.base_dir or target
relpath = os.path.relpath(fpath, rel_base)
results[relpath] = process_file(fpath)
count += 1
if args.progress and count % args.progress == 0:
sys.stderr.write(".")
sys.stderr.flush()
elif target.endswith(".py"):
rel_base = args.base_dir or os.path.dirname(target) or "."
relpath = os.path.relpath(target, rel_base)
results[relpath] = process_file(target)
count += 1
if args.progress and count % args.progress == 0:
sys.stderr.write(".")
sys.stderr.flush()
output = open(args.output, "w", encoding="utf-8") if args.output else sys.stdout
try:
json.dump(results, output, ensure_ascii=False, separators=(",", ":"))
finally:
if args.output:
output.close()
if __name__ == "__main__":
main()