aboutsummaryrefslogtreecommitdiffstats
path: root/server
diff options
context:
space:
mode:
Diffstat (limited to 'server')
-rw-r--r--server/patch_server.py2258
1 files changed, 0 insertions, 2258 deletions
diff --git a/server/patch_server.py b/server/patch_server.py
deleted file mode 100644
index 6667e12..0000000
--- a/server/patch_server.py
+++ /dev/null
@@ -1,2258 +0,0 @@
-#!/usr/bin/env python3
-"""Inject idatui's extra ida-pro-mcp tools into the installed server package.
-
-DEPRECATED along with the ida-pro-mcp transport: the default backend is now the
-idalib worker (idatui/worker.py), which registers these same tools in-process and
-needs no patching. Kept only for `--backend mcp`; slated for removal.
-
-ida-pro-mcp lacks a few tools idatui needs. Rather than vendor/fork the server,
-we keep the tool source here and inject it (idempotently) into the installed
-``api_types.py``. That module is imported by every worker
-(``python -m ida_pro_mcp.idalib_server``), so the tools register themselves via
-``@tool`` on the shared ``MCP_SERVER`` — no server code is forked, and re-running
-this (spawn.sh does, on every start) re-applies it after a reinstall/upgrade.
-
-Injected tools:
- * ``del_type`` — delete a named local type (struct editor CRUD).
- * ``func_types`` — structured decompiler types for a function (prototype +
- local variables), so clients don't parse pseudocode text.
- * ``set_lvar_type`` — set a decompiler local variable's type; works on auto/
- register vars too (the stock set_type only updates lvars
- that already have user-saved info).
-
-The block between the BEGIN/END markers is *replaced* on each run, so editing
-BODY here and restarting the supervisor updates the tools.
-
-Run with the *same* interpreter the server uses (the idalib-mcp entry point's
-``/usr/bin/python``), so it patches the file the workers actually import.
-Changing a tool needs a supervisor restart so workers respawn.
-"""
-from __future__ import annotations
-
-import importlib.util
-import pathlib
-import sys
-
-BEGIN = "# >>> idatui-ext: begin (auto-injected by server/patch_server.py) >>>"
-END = "# <<< idatui-ext: end <<<"
-
-# Appended to ida_pro_mcp/ida_mcp/api_types.py, which already imports
-# ``Annotated``, ``tool``, ``idasync``, ``ida_typeinf``, ``parse_address`` and
-# ``_parse_type_tinfo``.
-BODY = '''
-def _idatui_lv_get(x):
- return x() if callable(x) else x
-
-
-@tool
-@idasync
-def resolve_names(
- queries: Annotated[list, "Symbol name(s) to resolve to their OWN address"],
-) -> list:
- """Resolve named locations (functions, labels like loc_/locret_, data) to the
- exact address the NAME denotes, via get_name_ea. Unlike lookup_funcs, a
- mid-function label resolves to the label's address, not the containing
- function's entry."""
- import idaapi
- qs = queries if isinstance(queries, list) else [queries]
- out = []
- for q in qs:
- q = str(q).strip()
- ea = idaapi.get_name_ea(idaapi.BADADDR, q)
- out.append({"query": q, "ea": (hex(ea) if ea != idaapi.BADADDR else None)})
- return out
-
-
-@tool
-@idasync
-def del_type(
- name: Annotated[str, "Local type name to delete (struct/union/enum/typedef)"],
-) -> dict:
- """Delete a named local type from the local type library."""
- til = ida_typeinf.get_idati()
- ok = ida_typeinf.del_named_type(til, name, ida_typeinf.NTF_TYPE)
- if not ok:
- return {"name": name, "error": f"Type '{name}' not found or could not be deleted"}
- return {"name": name, "deleted": True}
-
-
-@tool
-@idasync
-def func_types(
- addr: Annotated[str, "Function address or name"],
-) -> dict:
- """Structured decompiler types for a function: its prototype plus each local
- variable (name/type/is_arg). Lets clients read/edit types without parsing
- pseudocode text."""
- import ida_hexrays
- import idaapi
-
- def _tstr(tif):
- try:
- s = tif.dstr()
- if s:
- return s
- except Exception:
- pass
- return str(tif)
-
- ea = parse_address(addr)
- f = idaapi.get_func(ea)
- if not f:
- return {"addr": str(addr), "error": "no function at address"}
- try:
- cf = ida_hexrays.decompile(f.start_ea)
- except Exception as e:
- return {"addr": hex(f.start_ea), "error": f"decompile failed: {e}"}
- if cf is None:
- return {"addr": hex(f.start_ea), "error": "decompilation failed"}
- name = idaapi.get_func_name(f.start_ea) or ""
- try:
- proto = ida_typeinf.print_tinfo(
- "", 0, 0, ida_typeinf.PRTYPE_1LINE, cf.type, name, "")
- except Exception:
- proto = ""
- lvars = []
- for lv in cf.get_lvars():
- try:
- ty = _tstr(_idatui_lv_get(lv.type))
- except Exception:
- ty = ""
- lvars.append({
- "name": _idatui_lv_get(lv.name),
- "type": ty,
- "is_arg": bool(_idatui_lv_get(lv.is_arg_var)),
- })
- return {
- "addr": hex(f.start_ea),
- "name": name,
- "prototype": (proto or "").strip(),
- "lvars": lvars,
- }
-
-
-@tool
-@idasync
-def set_lvar_type(
- addr: Annotated[str, "Function address or name"],
- variable: Annotated[str, "Local variable name"],
- type: Annotated[str, "New C type for the variable"],
-) -> dict:
- """Set a decompiler local variable's type. Handles auto/register vars (unlike
- set_type, which only updates lvars that already have user-saved info)."""
- import ida_hexrays
- import idaapi
-
- ea = parse_address(addr)
- f = idaapi.get_func(ea)
- if not f:
- return {"error": "no function at address"}
- try:
- cf = ida_hexrays.decompile(f.start_ea)
- except Exception as e:
- return {"error": f"decompile failed: {e}"}
- if cf is None:
- return {"error": "decompilation failed"}
- target = None
- for lv in cf.get_lvars():
- if _idatui_lv_get(lv.name) == variable:
- target = lv
- break
- if target is None:
- return {"error": f"local variable {variable!r} not found"}
- try:
- tif = _parse_type_tinfo(type)
- except Exception as e:
- return {"error": f"bad type {type!r}: {e}"}
- lsi = ida_hexrays.lvar_saved_info_t()
- try:
- lsi.ll = target
- except Exception:
- try:
- lsi.ll.location = _idatui_lv_get(target.location)
- lsi.ll.defea = target.defea
- except Exception as e:
- return {"error": f"could not locate variable: {e}"}
- lsi.type = tif
- ok = bool(ida_hexrays.modify_user_lvar_info(
- f.start_ea, ida_hexrays.MLI_TYPE, lsi))
- return {"addr": hex(f.start_ea), "variable": variable, "type": type, "ok": ok}
-
-
-@tool
-@idasync
-def file_regions() -> dict:
- """Loaded segments mapped to their raw file offsets (get_fileregion_offset),
- so clients can convert a virtual address to an on-disk file offset without a
- format-specific header parser. file_off is -1 for non-file-backed segments
- (e.g. .bss)."""
- import ida_segment
- import idaapi
-
- out = []
- seg = ida_segment.get_first_seg()
- while seg is not None:
- try:
- fo = int(idaapi.get_fileregion_offset(seg.start_ea))
- except Exception:
- fo = -1
- if fo < 0 or fo >= (1 << 48):
- fo = -1
- try:
- nm = ida_segment.get_segm_name(seg) or ""
- except Exception:
- nm = ""
- out.append({"start": hex(seg.start_ea), "end": hex(seg.end_ea),
- "file_off": fo, "name": nm})
- seg = ida_segment.get_next_seg(seg.start_ea)
- return {"regions": out}
-
-
-@tool
-@idasync
-def make_string(
- addr: Annotated[str, "Address of the string start"],
- length: Annotated[int, "Length in bytes (0 = auto-detect to the terminator)"] = 0,
- kind: Annotated[str, "String kind: c | c16 | c32 | pascal"] = "c",
-) -> dict:
- """Create a string literal at ``addr`` (IDA's 'A'). ``length`` 0 auto-detects
- to the terminator. Undefines any items in the way first, like the UI does.
- Returns the created byte size and the decoded contents."""
- import ida_bytes
- import ida_nalt
-
- ea = parse_address(addr)
- strtype = {
- "c": ida_nalt.STRTYPE_C,
- "c16": ida_nalt.STRTYPE_C_16,
- "c32": ida_nalt.STRTYPE_C_32,
- "pascal": ida_nalt.STRTYPE_PASCAL,
- }.get(str(kind).lower(), ida_nalt.STRTYPE_C)
- n = max(int(length), 0)
- # Free any existing item(s) so create_strlit can carve the literal.
- ida_bytes.del_items(ea, ida_bytes.DELIT_SIMPLE, n if n > 0 else 1)
- ok = bool(ida_bytes.create_strlit(ea, n, strtype))
- if not ok:
- return {"addr": addr, "ok": False, "error": "create_strlit failed"}
- size = int(ida_bytes.get_item_size(ea))
- try:
- raw = ida_bytes.get_strlit_contents(ea, -1, strtype)
- text = raw.decode("utf-8", "replace") if raw else ""
- except Exception:
- text = ""
- return {"addr": addr, "ok": True, "size": size, "text": text}
-
-
-@tool
-@idasync
-def read_raw(
- addr: Annotated[str, "Start address (hex or name)"],
- size: Annotated[int, "Number of bytes to read"],
-) -> dict:
- """Read ``size`` bytes at ``addr`` as ONE contiguous lowercase hex string
- (no per-byte '0x'/spaces). The hot path for the hex view and disasm opcode
- bytes.
-
- Fast: does a single bulk ``ida_bytes.get_bytes`` (C-speed) instead of the
- per-byte read_bytes_bss_safe loop (2 IDA calls/byte). Unloaded bytes come
- back from IDA as the 0xFF sentinel, so we only re-check is_loaded for the
- (usually sparse) 0xFF bytes and zero the genuinely-unloaded ones — matching
- get_bytes' bss semantics without paying per-byte for the whole range.
-
- Encoding is compact hex (~2.5x smaller than get_bytes' '0x..'-with-spaces)
- and, unlike get_bytes, does not truncate on large reads."""
- import ida_bytes
-
- ea = parse_address(addr)
- n = max(int(size), 0)
- if n == 0:
- return {"addr": addr, "hex": "", "n": 0}
- raw = ida_bytes.get_bytes(ea, n)
- if raw is None or len(raw) < n: # nothing (or not all) mapped
- base = bytearray(raw or b"")
- base.extend(b"\\xff" * (n - len(base)))
- raw = bytes(base)
- ba = bytearray(raw)
- # Only unloaded bytes read as 0xFF; correct just those to 0 (bss => zero).
- i = ba.find(0xFF)
- while i != -1:
- if not ida_bytes.is_loaded(ea + i):
- ba[i] = 0
- i = ba.find(0xFF, i + 1)
- return {"addr": addr, "hex": bytes(ba).hex(), "n": len(ba)}
-
-
-def _idatui_head_row(ea, flags=None):
- """One flat-listing row for the head at ``ea``: kind (code/data/unknown),
- byte size, rendered text, and any symbol name.
-
- ``flags`` lets a caller that already asked for them say so -- the walk in
- ``heads`` used to fetch them three times per head (here, in _is_unknown from
- _advance, and again from _rows_for).
- """
- import ida_bytes
- import ida_lines
- import ida_name
-
- f = ida_bytes.get_flags(ea) if flags is None else flags
- if ida_bytes.is_code(f):
- kind = "code"
- elif ida_bytes.is_data(f):
- kind = "data"
- else:
- kind = "unknown"
- line = ida_lines.generate_disasm_line(ea, 0)
- text, spans, ops = _idatui_line_parts(line) if line else ("", None, None)
- row = {
- "ea": hex(ea),
- "kind": kind,
- "size": int(ida_bytes.get_item_size(ea)),
- "text": text,
- }
- if spans is not None:
- row["spans"] = spans
- # Where each operand sits in `text`. Comes out of the same tag walk
- # (free), and is what lets the client show WHICH literal a keypress
- # would reformat before you press it.
- if ops:
- row["ops"] = ops
- nm = ida_name.get_ea_name(ea)
- if nm:
- row["name"] = nm
- return row
-
-
-import functools as _idatui_functools
-import os as _idatui_os
-
-#: Entries in the per-line render cache. Sized to hold a whole segment's
-#: DISTINCT lines rather than a working set, because the listing gets rendered
-#: TWICE: once when it is first walked, and again after a rename, which restates
-#: every row's text. bash's .text is 228 659 rows but only 53 363 distinct
-#: lines, and the difference between thrashing and not is the whole win:
-#:
-#: maxsize first sweep second sweep worker RSS
-#: 16 384 17.2 us/row 16.9 us/row +29 MB
-#: 32 768 17.0 17.2 +52 MB
-#: 65 536 17.0 11.1 +75 MB
-#: 131 072 16.9 11.2 +75 MB (working set fits)
-#:
-#: It is a bound, not a proportion: a bigger binary fills it and stops, so the
-#: cost is capped at ~56 MB whatever is open. Lower it with IDATUI_LINE_CACHE if
-#: a pool of workers is competing for memory.
-_IDATUI_LINE_CACHE = int(_idatui_os.environ.get("IDATUI_LINE_CACHE") or 65536)
-
-
-@_idatui_functools.lru_cache(maxsize=_IDATUI_LINE_CACHE)
-def _idatui_line_parts(line):
- """``(text, spans, ops)`` for one tagged disassembly line -- memoised.
-
- A function of the tagged line and nothing else, so the same line always
- gives the same answer: a rename changes the line, which changes the key.
- And listings repeat themselves hard -- 196k lines of bash are 53k distinct
- ones, so a 16k-entry cache serves ~70% of them and takes the per-line cost
- from 10.4us to 3.9us. This is the most expensive thing the backend does per
- listing row, and a jump to an address near the end of a big binary walks
- hundreds of thousands of them.
-
- ``spans`` is None when the tag walk and the plain text disagree about what
- the line says (then the text wins and the row renders unhighlighted).
-
- The returned lists are SHARED between every row that has the same line;
- treat them as read-only. Pickle notices the sharing too, so a page of
- repetitive disassembly also serialises smaller.
- """
- import ida_lines
- text = " ".join(ida_lines.tag_remove(line).split()) # collapse the padding
- spans, ops = _idatui_spans(line)
- # Built from the SAME line as `text`, then whitespace-collapsed identically,
- # so the two can never disagree about what the row says.
- joined = "".join([t for _k, t in spans])
- if " ".join(joined.split()) != text:
- return (text, None, None)
- return (text, spans, ops)
-
-
-#: IDA colour tag -> the semantic kind the TUI styles. IDA already classifies
-#: every token in a disassembly line, for every processor it supports, so there
-#: is nothing to lex: generate_disasm_line emits \x01<tag>text\x02<tag> and the
-#: tag says what the text IS. A pygments assembly lexer would be a worse guess at
-#: this and would need one dialect per architecture.
-_IDATUI_SPAN_KINDS = {
- "insn": ("SCOLOR_INSN", "SCOLOR_KEYWORD", "SCOLOR_ASMDIR", "SCOLOR_MACRO"),
- "reg": ("SCOLOR_REG",),
- "num": ("SCOLOR_NUMBER", "SCOLOR_CHAR", "SCOLOR_BINPREF"),
- "str": ("SCOLOR_STRING",),
- # NB the real constant names: DATNAME/CODNAME, not "DNAME". Guessing here
- # fails silently — an unmapped tag renders as plain body text, so symbols
- # just quietly aren't blue and nothing tells you why.
- "name": ("SCOLOR_DATNAME", "SCOLOR_CODNAME", "SCOLOR_LOCNAME",
- "SCOLOR_IMPNAME", "SCOLOR_DEMNAME", "SCOLOR_LIBNAME",
- "SCOLOR_CNAME", "SCOLOR_DNAME",
- "SCOLOR_CREF", "SCOLOR_DREF", "SCOLOR_CREFTAIL", "SCOLOR_DREFTAIL"),
- "seg": ("SCOLOR_SEGNAME",),
- "cmt": ("SCOLOR_AUTOCMT", "SCOLOR_REGCMT", "SCOLOR_RPTCMT", "SCOLOR_VOIDOP"),
- "punct": ("SCOLOR_SYMBOL", "SCOLOR_ALTOP", "SCOLOR_HIDNAME"),
- "err": ("SCOLOR_ERROR",),
-}
-
-
-def _idatui_tag_map():
- """{tag character: kind}, built once from whatever this IDA actually has."""
- import ida_lines
- out = {}
- for kind, names in _IDATUI_SPAN_KINDS.items():
- for n in names:
- v = getattr(ida_lines, n, None)
- if isinstance(v, str) and v:
- out[v[0]] = kind
- elif isinstance(v, int):
- out[chr(v)] = kind
- return out
-
-
-_IDATUI_TAGS = None
-_IDATUI_OPND_TAGS = None
-_IDATUI_CTL = None # re: a tag = one of three control chars plus its argument
-#: {tag character: (kind, operand index or None)} -- the two maps above merged,
-#: because the span walker wants both for the same tag and a dict lookup per
-#: tag per line is one of the few things it does often enough to matter.
-_IDATUI_TAGINFO = None
-
-
-def _idatui_opnd_tag_map():
- """{tag character: operand index}. IDA wraps each operand of a disassembly
- line in COLOR_OPND1..8, so the line already says where operand N starts and
- ends -- no need to re-render operands with print_operand to find out (and
- the two agree exactly; checked over thousands of instructions)."""
- import ida_lines
- out = {}
- for i in range(1, 9):
- v = getattr(ida_lines, "COLOR_OPND%d" % i, None)
- if isinstance(v, int):
- out[chr(v)] = i - 1
- elif isinstance(v, str) and v:
- out[v[0]] = i - 1
- return out
-
-
-def _idatui_spans(line):
- """(spans, ops) for a tagged disasm line.
-
- ``spans`` is [[kind, text], ...] with colour tags resolved; ``ops`` is
- [[start, end, n], ...], the extent of each operand in the SAME (collapsed)
- coordinates the row's ``text`` uses -- which is what lets a cursor column
- name the operand it is standing on.
-
- Unknown tags become 'text' rather than being dropped: a processor module can
- emit a colour we don't classify, and losing the characters would corrupt the
- line."""
- global _IDATUI_TAGS, _IDATUI_OPND_TAGS, _IDATUI_CTL, _IDATUI_TAGINFO
- import ida_lines
- if _IDATUI_TAGS is None:
- _IDATUI_TAGS = _idatui_tag_map()
- if _IDATUI_OPND_TAGS is None:
- _IDATUI_OPND_TAGS = _idatui_opnd_tag_map()
- if _IDATUI_CTL is None:
- import re as _re
- # One capturing split gives [text, tag, text, tag, ..., text] in a
- # single C pass. A per-character python loop over the line used to be
- # the most expensive thing the `heads` tool did, and a line is ~54
- # characters but only ~13 tags -- everything between two tags is already
- # exactly one span's worth of text.
- _IDATUI_CTL = _re.compile("([\\x01\\x02\\x03](?s:.))")
- if _IDATUI_TAGINFO is None:
- _IDATUI_TAGINFO = {
- tag: (_IDATUI_TAGS.get(tag, "text"), _IDATUI_OPND_TAGS.get(tag))
- for tag in set(_IDATUI_TAGS) | set(_IDATUI_OPND_TAGS)}
- taginfo = _IDATUI_TAGINFO
- plain_tag = ("text", None)
- on, off, esc = "\x01", "\x02", "\x03"
- addr_tag = chr(getattr(ida_lines, "COLOR_ADDR", 0x28))
- addr_len = int(getattr(ida_lines, "COLOR_ADDR_SIZE", 16))
- parts = _IDATUI_CTL.split(line)
- spans, stack = [], [] # stack entries: (kind, operand index|None)
- kind, opnd = "text", None # state the current run of text belongs to
- pend = ""
- skip = 0 # characters of an address payload still due
- i, n = 0, len(parts)
- while i < n:
- txt = parts[i]
- i += 1
- if skip:
- if len(txt) <= skip:
- skip -= len(txt)
- txt = ""
- else:
- txt = txt[skip:]
- skip = 0
- if txt:
- pend += txt
- if i >= n:
- break
- pair = parts[i]
- i += 1
- if skip: # a tag INSIDE an address payload: 2 chars
- skip = skip - 2 if skip > 2 else 0
- continue
- ch = pair[0]
- if ch == esc: # escaped literal: keep the char it guards
- pend += pair[1]
- continue
- tag = pair[1]
- if ch == on and tag == addr_tag:
- # An embedded target address, not display text: 16 hex digits that
- # must not reach the screen. Deliberately NOT a span boundary.
- skip = addr_len
- continue
- if pend:
- spans.append([kind, pend, opnd])
- pend = ""
- if ch == on:
- stack.append((kind, opnd))
- kind, o = taginfo.get(tag, plain_tag)
- if o is not None:
- opnd = o # operands nest: an inner colour keeps the operand
- elif stack:
- kind, opnd = stack.pop()
- else:
- kind, opnd = "text", None
- if pend:
- spans.append([kind, pend, opnd])
- # Collapse IDA's column padding EXACTLY as the plain text does. A run of
- # spaces can straddle two spans, so the leading space of a span is dropped
- # when the previous one ended in space — otherwise the spans and `text`
- # disagree about the line and the row silently loses its highlighting.
- # ``" ".join(txt.split())`` splits on exactly what str.isspace() calls
- # whitespace, which is what the character walk this replaces tested.
- out, prev_space = [], False
- for kind, txt, opnd in spans:
- core = " ".join(txt.split())
- if core == txt:
- # Nothing to collapse and no edge whitespace -- which is the common
- # case ("mov", "rax", ", ") and skips both isspace() probes below.
- prev_space = False
- out.append([kind, txt, opnd])
- continue
- if not core: # the span is nothing but padding
- if not prev_space:
- prev_space = True
- out.append([kind, " ", opnd])
- continue
- acc = core
- if txt[0].isspace() and not prev_space:
- acc = " " + acc
- if txt[-1].isspace():
- acc += " "
- prev_space = acc[-1] == " "
- out.append([kind, acc, opnd])
- while out and out[0][1] == " ":
- out.pop(0)
- while out and out[-1][1] == " ":
- out.pop()
- if out and out[0][1].startswith(" "):
- out[0][1] = out[0][1].lstrip()
- if out and out[-1][1].endswith(" "):
- out[-1][1] = out[-1][1].rstrip()
- out = [s for s in out if s[1]]
- # Operand extents, in the coordinates of the collapsed text these spans
- # spell out. Adjacent spans of the same operand merge, so an operand like
- # ``[rbp+var_40]`` (five differently-coloured tokens) comes back as ONE
- # range -- which is the thing a cursor is inside of, and the thing a format
- # change applies to.
- ops, pos, cur, start = [], 0, None, 0
- for _kind, txt, opnd in out:
- if opnd != cur:
- if cur is not None and pos > start:
- ops.append([start, pos, cur])
- cur, start = opnd, pos
- pos += len(txt)
- if cur is not None and pos > start:
- ops.append([start, pos, cur])
- text = "".join(t for _k, t, _o in out)
- trimmed = []
- for lo, hi, k in ops: # don't let a range own trailing space
- while hi > lo and text[hi - 1].isspace():
- hi -= 1
- while lo < hi and text[lo].isspace():
- lo += 1
- if hi > lo:
- trimmed.append([lo, hi, k])
- return [[k, t] for k, t, _o in out], trimmed
-
-
-def _idatui_rows_digest(rows):
- """A value that changes whenever any of ``rows`` would render differently.
-
- Covers everything a client keeps off a row: address, kind, size, the plain
- text, the symbol name and the colour spans (which is what makes it exact
- rather than a heuristic -- two lines can collapse to the same text and still
- be coloured differently).
-
- Uses the interpreter's own ``hash``, deliberately. It never has to mean
- anything outside this process: the client stores what a page hashed to when
- it loaded it and hands the same number back to ask whether the page still
- hashes to that. One worker, one process, one hash seed.
- """
- acc = 0
- # The per-line render is memoised, so one spans list is shared by every row
- # that says the same thing -- about 45% of them within a page. Hash each
- # distinct list once and key that by identity, rather than rebuilding a
- # tuple of tuples per row (which is the exact cost that was measured and
- # removed from the client side for the same reason).
- seen = {}
- for r in rows:
- sp = r.get("spans")
- if sp is None:
- sh = None
- else:
- key = id(sp)
- sh = seen.get(key)
- if sh is None:
- sh = seen[key] = hash(tuple(map(tuple, sp)))
- acc = hash((acc, r.get("ea"), r.get("kind"), r.get("size"),
- r.get("text"), r.get("name"), sh))
- return acc
-
-
-def _idatui_unknown_row(ea, size):
- """One collapsed row for a run of ``size`` undefined bytes starting at
- ``ea``. A single byte is rendered normally (shows its value); a longer run
- collapses to ``db N dup(?)`` so a big .bss/gap doesn't explode into millions
- of one-byte rows."""
- import ida_name
-
- if size <= 1:
- return _idatui_head_row(ea)
- row = {"ea": hex(ea), "kind": "unknown", "size": int(size),
- "text": f"db {size} dup(?)"}
- nm = ida_name.get_ea_name(ea)
- if nm:
- row["name"] = nm
- return row
-
-
-def _idatui_struct_member_rows(ea):
- """Indented member rows for a struct-typed data item at ``ea`` (expansion),
- or [] if it isn't a struct. Top-level fields only."""
- import ida_nalt
- import ida_typeinf
- import idaapi
-
- tif = ida_typeinf.tinfo_t()
- if not (ida_nalt.get_tinfo(tif, ea) and tif.is_udt()):
- return []
- udt = ida_typeinf.udt_type_data_t()
- if not tif.get_udt_details(udt):
- return []
- rows = []
- for m in udt:
- off = m.begin() // 8
- try:
- mtype = m.type._print() or ""
- except Exception:
- mtype = ""
- try:
- sz = int(m.type.get_size())
- if sz == idaapi.BADSIZE:
- sz = 0
- except Exception:
- sz = 0
- name = m.name or ""
- text = f"+{off:X} {name}" + (f" {mtype}" if mtype else "")
- rows.append({"ea": hex(ea + off), "kind": "member", "size": sz,
- "text": text})
- return rows
-
-
-def _idatui_func_header_rows(ea):
- """IDA-style subroutine banner rows shown just before a function's entry."""
- import ida_funcs
-
- name = ida_funcs.get_func_name(ea) or "sub_%X" % ea
- bar = "=" * 15 + " S U B R O U T I N E " + "=" * 15
- return [
- {"ea": hex(ea), "kind": "sep", "size": 0, "text": ""},
- {"ea": hex(ea), "kind": "sep", "size": 0, "text": "; " + bar},
- {"ea": hex(ea), "kind": "funchdr", "size": 0,
- "text": name + " proc", "name": name},
- ]
-
-
-def _idatui_func_footer_rows(ea, func):
- """End-of-function marker shown just after a function's last item."""
- import ida_funcs
-
- name = ida_funcs.get_func_name(func.start_ea) or "sub_%X" % func.start_ea
- return [
- {"ea": hex(ea), "kind": "funchdr", "size": 0,
- "text": name + " endp", "name": name},
- {"ea": hex(ea), "kind": "sep", "size": 0, "text": "; " + "-" * 60},
- ]
-
-
-@tool
-@idasync
-def heads(
- addr: Annotated[str, "Start address or name to walk from"],
- count: Annotated[int, "Max heads to return (default 200, max 2000)"] = 200,
- offset: Annotated[int, "Skip first N heads from addr (default 0)"] = 0,
- end: Annotated[str, "Optional exclusive end address; default = segment end"] = "",
- back: Annotated[bool, "Walk backwards: return the count heads ENDING just before addr, in forward order"] = False,
- annotate: Annotated[bool, "Emit IDA-style function boundary banner rows (kind sep/funchdr)"] = False,
- expect: Annotated[str, "Digest a caller already holds: the rows are omitted when they still hash to it"] = "",
-) -> dict:
- """Walk item heads from ``addr`` as a flat listing: every head is rendered
- (code OR data OR undefined) via generate_disasm_line and stepped with
- next_head/prev_head. Unlike ``disasm`` (code-only, bails at the first data
- byte) this shows db/dw/dd/... lines for data and undefined regions — IDA's
- real disassembly view. Address-paged: page forward by re-calling with
- ``addr`` = the returned cursor.next; page up with ``back=true``."""
- import ida_bytes
- import ida_segment
- import idaapi
-
- count = 2000 if count > 2000 else (1 if count < 1 else count)
- offset = max(int(offset), 0)
- try:
- start = parse_address(addr)
- except Exception as e:
- return {"addr": str(addr), "error": str(e), "heads": [], "cursor": {"done": True}}
- seg = ida_segment.getseg(start)
- if not seg:
- return {"addr": str(addr), "error": "no segment", "heads": [], "cursor": {"done": True}}
- lo, hi = seg.start_ea, seg.end_ea
- if end:
- try:
- hi = min(hi, parse_address(end))
- except Exception:
- pass
-
- rows = []
- if back:
- # Collect up to (count+offset) heads strictly before `start`, then take
- # the window closest to `start`, returned in forward order.
- walk = []
- cur = ida_bytes.prev_head(start, lo)
- while cur != idaapi.BADADDR and cur >= lo and len(walk) < count + offset:
- walk.append(cur)
- cur = ida_bytes.prev_head(cur, lo)
- walk.reverse()
- chosen = walk[: len(walk) - offset] if offset else walk
- chosen = chosen[-count:]
- rows = [_idatui_head_row(e) for e in chosen]
- first = chosen[0] if chosen else start
- pea = ida_bytes.prev_head(first, lo)
- cursor = {"done": True} if pea == idaapi.BADADDR or pea < lo else {"prev": hex(pea)}
- return {"addr": str(addr), "heads": rows, "cursor": cursor}
-
- # Walk by item END (not next_head): next_head SKIPS undefined bytes, but a
- # flat listing must show them (IDA renders undefined as `db ?` lines, and
- # navigating to an unmarked address must land ON it). Defined items advance
- # by get_item_end; a run of undefined bytes is COLLAPSED into one row (its
- # end found in O(1) via next_head, which skips undefined) so a large .bss or
- # gap doesn't explode into millions of one-byte rows.
- def _is_unknown_f(f):
- return not (ida_bytes.is_code(f) or ida_bytes.is_data(f))
-
- def _run_end(e):
- """End (exclusive) of the undefined run starting at ``e``."""
- nh = ida_bytes.next_head(e, hi)
- return nh if (nh != idaapi.BADADDR and e < nh <= hi) else hi
-
- def _advance(e, f):
- if _is_unknown_f(f):
- return _run_end(e)
- nxt = ida_bytes.get_item_end(e)
- return nxt if nxt > e else e + 1
-
- # The function the walk is currently inside, reused while it stays inside.
- # get_func is ~0.5us and the walk asks per head; a head is nearly always in
- # the same function as the one before it. Only ever consulted when ``e``
- # falls in [start_ea, end_ea), so a tail chunk elsewhere cannot be
- # misattributed -- checked against get_func over 437k heads of
- # bash/ls_ttl/echo with zero disagreements.
- fn_cache = [None]
-
- def _func_at(e):
- cur = fn_cache[0]
- if cur is not None and cur.start_ea <= e < cur.end_ea:
- return cur
- cur = idaapi.get_func(e)
- fn_cache[0] = cur
- return cur
-
- def _rows_for(e, f):
- if _is_unknown_f(f):
- return [_idatui_unknown_row(e, _run_end(e) - e)]
- func = _func_at(e) if annotate else None
- at_start = func is not None and func.start_ea == e
- out = []
- if at_start:
- out.extend(_idatui_func_header_rows(e))
- row = _idatui_head_row(e, f)
- if at_start:
- row = dict(row)
- row["name"] = None # the name is shown on the proc header line
- elif annotate and row.get("kind") == "code" and row.get("name"):
- # A code label (loc_XXX/jump target) gets its OWN line at depth 0,
- # like IDA; strip it from the instruction row below.
- nm = row["name"]
- out.append({"ea": hex(e), "kind": "label", "size": 0,
- "text": nm + ":", "name": nm})
- row = dict(row)
- row["name"] = None
- out.append(row)
- if row.get("kind") == "data":
- out.extend(_idatui_struct_member_rows(e)) # expand struct fields
- if func is not None and ida_bytes.get_item_end(e) >= func.end_ea:
- out.extend(_idatui_func_footer_rows(e, func))
- return out
-
- ea = ida_bytes.get_item_head(start)
- get_flags = ida_bytes.get_flags
- for _ in range(offset):
- if ea >= hi or ea == idaapi.BADADDR:
- break
- ea = _advance(ea, get_flags(ea))
- more = False
- while ea != idaapi.BADADDR and ea < hi:
- if len(rows) >= count:
- more = True
- break
- f = get_flags(ea) # once per head, not once per consumer
- rows.extend(_rows_for(ea, f)) # a struct head expands into member rows
- ea = _advance(ea, f)
- cursor = {"next": hex(ea)} if more else {"done": True}
- dig = _idatui_rows_digest(rows)
- out = {"addr": str(addr), "cursor": cursor, "digest": dig, "count": len(rows)}
- # ``expect`` says "I already hold a page that hashed to this". The rows are
- # built either way -- generate_disasm_line is the floor and there is no way
- # to know a line is unchanged without rendering it -- but pickling several
- # hundred rows with their colour spans, unpickling them and rebuilding Heads
- # is about 40% of what a page costs, and after a rename almost every page
- # comes back identical.
- #
- # It carries the expected value rather than being a yes/no "digest mode" so
- # that a page which HAS changed still costs one round trip: asking first and
- # fetching afterwards made every changed page two.
- if not (expect and str(dig) == expect):
- out["heads"] = rows
- return out
-
-
-@tool
-@idasync
-def xref_types(
- queries: Annotated[list, "[{addr, direction:'to'|'from'|'both', include_fn, dedup, count}]"],
-) -> dict:
- """Like xref_query, but every row carries a fine-grained ``kind`` derived from
- the IDA xref type \u2014 call/jump/flow for code, read/write/offset/text/info for
- data \u2014 alongside the coarse ``type`` (code/data). Feeds the xref dialog's
- r/w/call badges. Same query/envelope shape as xref_query."""
- import idaapi, idautils, ida_funcs, ida_bytes, ida_xref
- code_kind = {ida_xref.fl_CF: "call", ida_xref.fl_CN: "call",
- ida_xref.fl_JF: "jump", ida_xref.fl_JN: "jump",
- ida_xref.fl_F: "flow"}
- data_kind = {ida_xref.dr_O: "offset", ida_xref.dr_W: "write",
- ida_xref.dr_R: "read", ida_xref.dr_T: "text", ida_xref.dr_I: "info"}
-
- def _kind(xr):
- table = code_kind if xr.iscode else data_kind
- return table.get(xr.type, "code" if xr.iscode else "data")
-
- def _fn(ea):
- f = ida_funcs.get_func(ea)
- if not f:
- return None
- return {"addr": hex(f.start_ea), "name": ida_funcs.get_func_name(f.start_ea)}
-
- def _resolve(raw):
- raw = str(raw).strip()
- try:
- return int(raw, 16) # handles '0x2490' and '2490'
- except ValueError:
- return idaapi.get_name_ea(idaapi.BADADDR, raw)
-
- qs = queries if isinstance(queries, list) else [queries]
- result = []
- for q in qs:
- q = q if isinstance(q, dict) else {"addr": q}
- raw = str(q.get("addr", "")).strip()
- direction = str(q.get("direction", "to") or "to").lower()
- include_fn = bool(q.get("include_fn", True))
- dedup = bool(q.get("dedup", True))
- try:
- count = int(q.get("count", 2000) or 2000)
- except (TypeError, ValueError):
- count = 2000
- target = _resolve(raw)
- rows = []
- if target is not None and target != idaapi.BADADDR and ida_bytes.is_mapped(target):
- if direction in ("to", "both"):
- for xr in idautils.XrefsTo(target, 0):
- row = {"direction": "to", "addr": hex(xr.frm), "from": hex(xr.frm),
- "to": hex(target), "type": "code" if xr.iscode else "data",
- "kind": _kind(xr)}
- if include_fn:
- row["fn"] = _fn(xr.frm)
- rows.append(row)
- if direction in ("from", "both"):
- for xr in idautils.XrefsFrom(target, 0):
- row = {"direction": "from", "addr": hex(xr.to), "from": hex(target),
- "to": hex(xr.to), "type": "code" if xr.iscode else "data",
- "kind": _kind(xr)}
- if include_fn:
- row["fn"] = _fn(xr.to)
- rows.append(row)
- if dedup:
- seen = set()
- deduped = []
- for r in rows:
- k = (r["direction"], r["from"], r["to"], r["kind"])
- if k in seen:
- continue
- seen.add(k)
- deduped.append(r)
- rows = deduped
- rows = rows[:count]
- result.append({"query": raw, "data": rows, "next_offset": None})
- return {"result": result}
-
-
-@tool
-@idasync
-def data_type(
- addr: Annotated[str, "Address or name of a data item / global"],
-) -> dict:
- """The current C type of a data item, for prefilling a retype prompt:
- {addr, name, type, size, is_func}. ``type`` is empty when the item is
- untyped; ``is_func`` distinguishes a global from a function so the caller
- knows which flavour of set_type to use."""
- import idaapi
- import ida_bytes
- import ida_name
- import idc
- raw = str(addr).strip()
- try:
- ea = int(raw, 16)
- except ValueError:
- ea = idaapi.get_name_ea(idaapi.BADADDR, raw)
- if ea == idaapi.BADADDR or not ida_bytes.is_mapped(ea):
- return {"addr": raw, "error": f"not a mapped address: {raw}"}
- return {
- "addr": hex(ea),
- "name": ida_name.get_name(ea) or "",
- "type": idc.get_type(ea) or "",
- "size": int(ida_bytes.get_item_size(ea) or 0),
- "is_func": bool(idaapi.get_func(ea)),
- }
-
-
-@tool
-@idasync
-def decomp_map(
- addr: Annotated[str, "Function address or name"],
-) -> dict:
- """Per-pseudocode-line instruction coverage for the split view's region
- highlight: for each line, the set of EAs the decompiler attributes to it,
- swept across the line's columns via get_line_item. Shape:
- {addr, lines:[{ea: primary|None, eas:[hex,...]}, ...]}."""
- import ida_hexrays
- import idaapi
- try:
- ea = int(str(addr), 16)
- except ValueError:
- ea = idaapi.get_name_ea(idaapi.BADADDR, str(addr).strip())
- func = idaapi.get_func(ea)
- if not func:
- return {"error": f"no function at {addr}"}
- try:
- cfunc = ida_hexrays.decompile(func.start_ea)
- except Exception as e: # noqa: BLE001
- return {"error": f"decompile failed: {e}"}
- if cfunc is None:
- return {"error": "decompile failed"}
- import ida_lines
- # Three things this loop must not do, each measured on real functions (the 25
- # largest of bash went 68.3s -> 6.5s; echo's 60 largest 5.4s -> 0.6s, with
- # byte-identical output):
- #
- # * allocate ctree_item_t's per COLUMN. They are SWIG objects and this is
- # the innermost loop; one per call is enough, and head/tail are never
- # read, so don't ask for them at all.
- # * sweep the TAGGED length. ``x`` is a screen column but ``sl.line`` still
- # carries IDA's colour tags, so a 23-column line was swept 124 times.
- # * call dstr() per column. It formats a whole 'EA: description' string --
- # 24us a call, which is 79% of this tool. Comparing against the PREVIOUS
- # column's item id is not enough: items interleave, so `foo(a, b)` flips
- # call -> arg -> call -> arg and every flip re-formats an item already
- # seen (106 594 calls for 15 417 lines of bash). Memoise id -> ea for the
- # whole function instead: obj_id is unique within a cfunc, so the same id
- # always yields the same string, and the result is deduped by ``seen``
- # anyway. Items with no ctree node (it is None) have no id to key on and
- # still pay per occurrence.
- item = ida_hexrays.ctree_item_t()
- tag_remove = ida_lines.tag_remove
- get_line_item = cfunc.get_line_item
- ea_of_id = {}
- lines = []
- for sl in cfunc.get_pseudocode():
- line = sl.line
- eas, seen = [], set()
- prev_id = None
- for x in range(len(tag_remove(line)) + 1):
- if not get_line_item(line, x, False, None, item, None):
- continue
- it = item.it
- if it is not None:
- oid = it.obj_id
- if oid == prev_id:
- continue
- prev_id = oid
- if oid in ea_of_id:
- e = ea_of_id[oid]
- if e is not None and e not in seen:
- seen.add(e)
- eas.append(hex(e))
- continue
- else:
- oid = None
- prev_id = None
- # Match the /*ea*/ marker's source (decompile_function_safe): the
- # item's dstr() is 'EA: description'; get_ea() reports a different ea.
- e = None
- dstr = item.dstr()
- if dstr:
- parts = dstr.split(": ", 1)
- if len(parts) == 2:
- try:
- e = int(parts[0], 16)
- except ValueError:
- e = None
- if oid is not None:
- ea_of_id[oid] = e
- if e is not None and e not in seen:
- seen.add(e)
- eas.append(hex(e))
- lines.append({"ea": eas[0] if eas else None, "eas": eas})
- return {"addr": hex(func.start_ea), "lines": lines}
-
-
-_idatui_strings_cache = {}
-
-
-def _idatui_build_strings(min_len):
- """[(ea, text, length, typename)] for every string IDA found, cached by
- min_len (rebuilding the list is O(n) and the browser pages through it)."""
- import idautils
- import ida_nalt
- hit = _idatui_strings_cache.get(min_len)
- if hit is not None:
- return hit
- tnames = {}
- for nm, lbl in (("STRTYPE_C", "C"), ("STRTYPE_C_16", "utf16"),
- ("STRTYPE_C_32", "utf32"), ("STRTYPE_PASCAL", "pascal")):
- v = getattr(ida_nalt, nm, None)
- if v is not None:
- tnames[v & 0xFF] = lbl
- items = []
- for s in idautils.Strings():
- if s is None:
- continue
- try:
- text = str(s)
- except Exception: # noqa: BLE001 -- undecodable literal
- continue
- if len(text) < min_len:
- continue
- st = getattr(s, "strtype", 0) & 0xFF
- items.append((s.ea, text, getattr(s, "length", len(text)),
- tnames.get(st, "t%d" % st)))
- _idatui_strings_cache[min_len] = items
- return items
-
-
-@tool
-@idasync
-def list_strings(
- offset: Annotated[int, "Start index into the strings list"] = 0,
- count: Annotated[int, "Max strings to return (page size)"] = 2000,
- min_len: Annotated[int, "Minimum string length to include"] = 4,
- refresh: Annotated[bool, "Rebuild the cached strings list"] = False,
-) -> dict:
- """Every string literal IDA found in the binary (IDA's Shift+F12 window),
- paginated: {strings:[{addr,text,len,type}], total, next_offset}. Feeds the
- TUI's strings browser."""
- try:
- min_len = max(int(min_len), 1)
- except (TypeError, ValueError):
- min_len = 4
- try:
- offset = max(int(offset), 0)
- except (TypeError, ValueError):
- offset = 0
- try:
- count = max(int(count), 1)
- except (TypeError, ValueError):
- count = 2000
- if refresh:
- _idatui_strings_cache.pop(min_len, None)
- items = _idatui_build_strings(min_len)
- page = items[offset:offset + count]
- return {
- "strings": [{"addr": hex(ea), "text": text, "len": ln, "type": ty}
- for (ea, text, ln, ty) in page],
- "total": len(items),
- "next_offset": offset + len(page),
- }
-
-@tool
-@idasync
-def list_linkage(
- kind: Annotated[str, "'import', 'export' or 'both'"] = "both",
-) -> dict:
- """What this binary imports from, and exports to, other modules:
- {imports:[{addr,name,module}], exports:[{addr,name,ordinal}]}. Feeds the
- project-wide import/export join, which resolves a PLT stub in one binary to
- the real implementation in another."""
- import idaapi
- import idautils
- import ida_nalt
- want = str(kind or "both").lower()
- imports = []
- exports = []
- if want in ("import", "both"):
- n = ida_nalt.get_import_module_qty()
- for i in range(n):
- mod = ida_nalt.get_import_module_name(i) or ""
-
- def _cb(ea, name, ordinal, _mod=mod):
- # An ordinal-only import has no name; skip rather than invent one.
- if name:
- imports.append({"addr": hex(ea), "name": name, "module": _mod})
- return True
-
- ida_nalt.enum_import_names(i, _cb)
- if want in ("export", "both"):
- for index, ordinal, ea, name in idautils.Entries():
- if name:
- exports.append({"addr": hex(ea), "name": name,
- "ordinal": int(ordinal)})
- return {"imports": imports, "exports": exports,
- "n_imports": len(imports), "n_exports": len(exports)}
-
-@tool
-@idasync
-def define_code_run(
- addr: Annotated[str, "Address to start disassembling from"],
- limit: Annotated[int, "Max instructions to create (safety stop)"] = 20000,
-) -> dict:
- """Disassemble CONSECUTIVELY from ``addr`` until something stops it, the way
- IDA's 'c' does — one instruction is rarely what you want when carving a raw
- image. Returns {start,end,count,stopped} where ``stopped`` says why:
- 'undecodable' (bytes aren't an instruction), 'flow' (the last instruction
- doesn't fall through, e.g. RET/B), 'defined' (ran into existing code/data),
- 'segment' (hit the end) or 'limit'.
-
- Runs in-process: doing this from the client would be one round trip per
- instruction, which is minutes on a real firmware image."""
- import ida_bytes
- import ida_idp
- import ida_segment
- import ida_ua
- import idaapi
-
- try:
- ea = parse_address(addr)
- except Exception as e:
- return {"addr": str(addr), "error": str(e), "count": 0}
-
- seg = ida_segment.getseg(ea)
- if not seg:
- return {"addr": str(addr), "error": "no segment", "count": 0}
- hi = seg.end_ea
- try:
- limit = max(1, min(int(limit), 200000))
- except (TypeError, ValueError):
- limit = 20000
-
- start, count, stopped = ea, 0, "limit"
- while count < limit:
- if ea >= hi:
- stopped = "segment"
- break
- flags = ida_bytes.get_flags(ea)
- if ida_bytes.is_code(flags) or ida_bytes.is_data(flags):
- # Already defined: stop rather than clobber. Undefining someone's
- # existing work to keep a speculative run going is not a trade the
- # user asked for.
- stopped = "defined"
- break
- n = ida_ua.create_insn(ea)
- if n <= 0:
- stopped = "undecodable"
- break
- count += 1
- # Stop where control flow stops. Past a RET the next bytes are usually
- # padding or a new function's data, and running on turns a clean carve
- # into a mess that has to be undone by hand.
- #
- # Ask ida_idp.is_ret_insn, NOT the canonical feature bits: on AArch64
- # get_canon_feature() returns 0 for RET, so a CF_STOP test silently never
- # fires and the run walks straight through the end of the routine.
- insn = ida_ua.insn_t()
- if ida_ua.decode_insn(insn, ea) > 0:
- try:
- is_ret = ida_idp.is_ret_insn(insn)
- except Exception:
- is_ret = False
- if is_ret or (insn.get_canon_feature() & idaapi.CF_STOP):
- ea += n
- stopped = "flow"
- break
- ea += n
-
- return {"start": hex(start), "end": hex(ea), "count": count,
- "stopped": stopped}
-
-@tool
-@idasync
-def set_thumb(
- addr: Annotated[str, "Address to change the ARM decoding mode at"],
- mode: Annotated[str, "'toggle', 'on' (Thumb) or 'off' (ARM)"] = "toggle",
- end: Annotated[str, "Optional exclusive end address (default: this item)"] = "",
-) -> dict:
- """Switch ARM/Thumb decoding at ``addr`` (IDA's T segment register).
-
- Thumb is not a property of the bytes, it's a mode the CPU is in, so a raw
- image gives IDA no way to know: at a Thumb entry point it decodes 16-bit
- instructions as 32-bit ARM and produces confident nonsense
- (``push {r3,lr}`` reads as ``SVCLT 0xBF00``).
-
- Also forces the segment to 32-bit when turning Thumb ON. Thumb does not
- exist in AArch64, and a headerless blob loaded with -parm defaults to
- 64-bit — so setting T alone changes nothing and looks broken. Asking for
- Thumb IS asking for ARM32."""
- import ida_bytes
- import ida_idp
- import ida_segment
- import ida_segregs
-
- try:
- ea = parse_address(addr)
- except Exception as e:
- return {"addr": str(addr), "error": str(e)}
- treg = ida_idp.str2reg("T")
- if treg is None or treg < 0:
- return {"addr": hex(ea), "error": "no T register (not an ARM database)"}
- seg = ida_segment.getseg(ea)
- if not seg:
- return {"addr": hex(ea), "error": "no segment"}
-
- import ida_ida
- db64 = ida_ida.inf_get_app_bitness() == 64
- cur = ida_segregs.get_sreg(ea, treg)
- cur = 0 if cur in (None, 0xFFFFFFFF, -1) else int(cur)
- want = {"on": 1, "off": 0}.get(str(mode).lower(), 0 if cur else 1)
-
- changed_bits = False
- if want and seg.bitness != 1:
- ida_segment.set_segm_addressing(seg, 1)
- changed_bits = True
-
- try:
- stop = parse_address(end) if end else 0
- except Exception:
- stop = 0
- size = max(int(stop) - ea, 0) or max(ida_bytes.get_item_size(ea), 2)
- # The bytes are currently decoded in the OLD mode; leaving that item defined
- # pins the wrong instruction length and the new mode has nothing to apply to.
- ida_bytes.del_items(ea, 0, size)
- ok = bool(ida_segregs.split_sreg_range(ea, treg, want, ida_segregs.SR_user))
- now = ida_segregs.get_sreg(ea, treg)
- return {"addr": hex(ea), "thumb": bool(now), "was": bool(cur), "ok": ok,
- "bitness": ida_segment.getseg(ea).bitness,
- "forced_32bit": changed_bits,
- # The DATABASE's bitness is fixed at load and can't be corrected
- # here (setting it post-hoc makes the decompiler INTERR). In a
- # 64-bit database a 32-bit function disassembles but Hex-Rays
- # refuses it outright, so say so instead of leaving the user to
- # discover that F5 does nothing.
- "db_64bit": bool(db64 and want)}
-
-def _idatui_add_func(ea):
- """add_func at ``ea``, falling back to an explicit end.
-
- ida_funcs.add_func(ea) asks IDA to find the end and on carved or
- freshly-marked code it often can't, failing with no reason given."""
- import ida_bytes
- import ida_funcs
- import ida_segment
- import idaapi
-
- if idaapi.get_func(ea) is not None:
- return True
- if ida_funcs.add_func(ea):
- return True
- seg = ida_segment.getseg(ea)
- hi = seg.end_ea if seg else ea
- end = ea
- while end < hi and ida_bytes.is_code(ida_bytes.get_flags(end)):
- nxt = ida_bytes.get_item_end(end)
- if nxt <= end:
- break
- end = nxt
- return bool(end > ea and ida_funcs.add_func(ea, end))
-
-
-@tool
-@idasync
-def define_func_run(
- addr: Annotated[str, "Entry point of the function to create"],
-) -> dict:
- """Create a function at ``addr``, working out its end if IDA can't.
-
- ida_funcs.add_func(ea) asks IDA to find the end itself, and on hand-carved
- code it often can't — a run that ends in a tail call, or whose last
- instruction isn't recognised as a return, simply fails with no reason given.
- You then have a disassembled routine that refuses to become a function, and
- F5 has nothing to work with.
-
- So: try IDA's way, and if that fails, use the end of the contiguous
- instruction run starting at ``addr``."""
- import ida_bytes
- import ida_funcs
- import ida_segment
- import idaapi
-
- try:
- ea = parse_address(addr)
- except Exception as e:
- return {"addr": str(addr), "error": str(e), "ok": False}
- fn = idaapi.get_func(ea)
- if fn is not None and fn.start_ea == ea:
- return {"addr": hex(ea), "ok": True, "start": hex(fn.start_ea),
- "end": hex(fn.end_ea), "how": "existed"}
- auto = ida_funcs.add_func(ea)
- if not auto and not _idatui_add_func(ea):
- return {"addr": hex(ea), "ok": False,
- "error": f"IDA refused a function at {ea:#x}"}
- f = idaapi.get_func(ea)
- if f is None:
- return {"addr": hex(ea), "ok": False, "error": "function did not stick"}
- return {"addr": hex(ea), "ok": True, "start": hex(f.start_ea),
- "end": hex(f.end_ea), "how": "auto" if auto else "explicit-end"}
-
-@tool
-@idasync
-def decomp_error(
- addr: Annotated[str, "Address of the function that failed to decompile"],
-) -> dict:
- """Why Hex-Rays refused this function, in its own words.
-
- The plain decompile tool reports "Decompilation failed at 0x0" and drops the
- reason, which is the only useful part. Hex-Rays fills in a hexrays_failure_t
- saying things like "only 64-bit functions can be decompiled in the current
- database" — that one is unfixable in place (the database's bitness is set at
- load), so a user who can't see it has no way to know they must reload."""
- import ida_funcs
- import ida_hexrays
- import ida_ida
-
- try:
- ea = parse_address(addr)
- except Exception as e:
- return {"addr": str(addr), "error": str(e)}
- out = {"addr": hex(ea), "bitness": ida_ida.inf_get_app_bitness()}
- fn = ida_funcs.get_func(ea)
- if fn is None:
- out["reason"] = "no function here"
- return out
- try:
- if not ida_hexrays.init_hexrays_plugin():
- out["reason"] = "the decompiler is not available for this processor"
- return out
- hf = ida_hexrays.hexrays_failure_t()
- cf = ida_hexrays.decompile_func(fn, hf)
- if cf is not None:
- out["reason"] = "" # it decompiles now
- return out
- out["reason"] = hf.desc() or f"error {hf.code}"
- out["code"] = int(hf.code)
- out["errea"] = hex(hf.errea)
- except Exception as e: # noqa: BLE001
- out["reason"] = f"{type(e).__name__}: {e}"
- return out
-
-@tool
-@idasync
-def thumb_scan(
- start: Annotated[str, "Start of the range to scan for entry pointers"] = "",
- end: Annotated[str, "Exclusive end of the range (default: 1KB from start)"] = "",
- apply: Annotated[bool, "Mark the targets as Thumb and disassemble them"] = True,
- limit: Annotated[int, "Max entries to act on"] = 512,
-) -> dict:
- """Find Thumb entry points from ODD pointers, e.g. a Cortex-M vector table.
-
- An ARM function pointer carries the mode in bit 0: odd means Thumb. A vector
- table is therefore a list of Thumb entry points that IDA won't follow on a
- headerless image, because nothing tells it those words are pointers at all.
-
- Being wrong here is expensive — marking a data word as code corrupts the
- listing — so a word only counts when it is odd, lands inside a loaded
- segment, and its target is EXECUTABLE and not already defined as data. The
- even words in a vector table (the initial stack pointer) fail the first test,
- which is the point."""
- import ida_bytes
- import ida_funcs
- import ida_idp
- import ida_segment
- import ida_segregs
- import ida_ua
-
- seg0 = ida_segment.getseg(parse_address(start)) if start else None
- if seg0 is None:
- seg0 = ida_segment.getnseg(0)
- if seg0 is None:
- return {"error": "no segments", "found": [], "applied": 0}
- try:
- lo = parse_address(start) if start else seg0.start_ea
- hi = parse_address(end) if end else min(lo + 0x400, seg0.end_ea)
- except Exception as e:
- return {"error": str(e), "found": [], "applied": 0}
-
- treg = ida_idp.str2reg("T")
- found, applied = [], 0
- ea = lo
- while ea + 4 <= hi and len(found) < limit:
- w = ida_bytes.get_dword(ea)
- ea += 4
- if not (w & 1):
- continue # even: not a Thumb pointer
- tgt = w & ~1
- seg = ida_segment.getseg(tgt)
- if seg is None or not (seg.perm & ida_segment.SEGPERM_EXEC or seg.perm == 0):
- continue # points outside the image, or at data
- f = ida_bytes.get_flags(tgt)
- if ida_bytes.is_data(f):
- continue # already something else; don't fight it
- rec = {"at": hex(ea - 4), "value": hex(w), "target": hex(tgt),
- "was_code": bool(ida_bytes.is_code(f))}
- found.append(rec)
- if not apply:
- continue
- if treg is not None and treg >= 0:
- ida_segregs.split_sreg_range(tgt, treg, 1, ida_segregs.SR_user)
- if not ida_bytes.is_code(ida_bytes.get_flags(tgt)):
- ida_bytes.del_items(tgt, 0, 2)
- if ida_ua.create_insn(tgt) <= 0:
- rec["decoded"] = False
- continue
- rec["decoded"] = True
- rec["function"] = _idatui_add_func(tgt)
- applied += 1
- return {"start": hex(lo), "end": hex(hi), "found": found,
- "applied": applied, "n": len(found)}
-
-
-# --------------------------------------------------------------------------- #
-# operand display formats (IDA's 'o' family: hex / dec / char / offset / ...)
-# --------------------------------------------------------------------------- #
-#: The stops a cycle walks, in order, before filtering to the ones that make
-#: sense for the operand in hand. Octal is deliberately NOT one of them -- every
-#: extra stop is another keypress and nobody reads octal -- but it is still
-#: reachable by name. "default" hands the operand back to IDA's own choice,
-#: which for data is how you get an auto-detected offset/string back.
-_IDATUI_FMT_CYCLE = ("hex", "dec", "bin", "char", "offset", "default")
-
-#: Formats we can re-apply from a name alone. enum/stroff/custom carry an id
-#: (which enum, which struct) that a nibble doesn't record, so they are never
-#: cycled INTO -- and cycling out of one is called out in ``warn``.
-_IDATUI_FMT_SETTABLE = ("hex", "dec", "oct", "bin", "char", "offset", "seg",
- "float", "stack", "default")
-
-
-def _idatui_fmt_nibbles():
- """{format name: IDA operand-type nibble}. Built on call, not at import:
- this module is injected into a file that is imported before a database is
- open."""
- import ida_bytes
- return {
- "default": ida_bytes.FF_N_VOID, "hex": ida_bytes.FF_N_NUMH,
- "dec": ida_bytes.FF_N_NUMD, "char": ida_bytes.FF_N_CHAR,
- "seg": ida_bytes.FF_N_SEG, "offset": ida_bytes.FF_N_OFF,
- "bin": ida_bytes.FF_N_NUMB, "oct": ida_bytes.FF_N_NUMO,
- "enum": ida_bytes.FF_N_ENUM, "forced": ida_bytes.FF_N_FOP,
- "stroff": ida_bytes.FF_N_STRO, "stack": ida_bytes.FF_N_STK,
- "float": ida_bytes.FF_N_FLT, "custom": ida_bytes.FF_N_CUST,
- }
-
-
-def _idatui_fmt_name(nib):
- for name, v in _idatui_fmt_nibbles().items():
- if v == nib:
- return name
- return "default"
-
-
-def _idatui_op_fmt(ea, n):
- """The format operand ``n`` of the item at ``ea`` is currently displayed in.
-
- Reads the nibble IDA keeps per operand rather than guessing from the text --
- ``1`` renders identically in hex and decimal, so the rendered line cannot
- answer this."""
- import ida_bytes
- F = ida_bytes.get_flags(ea)
- nib = (F >> ida_bytes.get_operand_type_shift(int(n))) & 0xF
- return _idatui_fmt_name(nib)
-
-
-def _idatui_op_value(ea, n):
- """(value, byte width) of operand ``n``, or (None, 0) if it hasn't got one.
-
- The value is what decides which formats are OFFERED: a character constant
- for 0x38A9 or an offset to an unmapped address are stops worth skipping."""
- import ida_bytes
- import ida_ua
-
- F = ida_bytes.get_flags(ea)
- if ida_bytes.is_code(F):
- insn = ida_ua.insn_t()
- if ida_ua.decode_insn(insn, ea) <= 0:
- return None, 0
- try:
- op = insn.ops[int(n)]
- except Exception:
- return None, 0
- if op.type == ida_ua.o_void:
- return None, 0
- v = op.value if op.type == ida_ua.o_imm else op.addr
- try:
- size = int(ida_ua.get_dtype_size(op.dtype))
- except Exception:
- size = 0
- return int(v), size
- size = int(ida_bytes.get_item_size(ea))
- read = {1: ida_bytes.get_byte, 2: ida_bytes.get_word,
- 4: ida_bytes.get_dword, 8: ida_bytes.get_qword}.get(size)
- if read is None:
- return None, size
- try:
- return int(read(ea)), size
- except Exception:
- return None, size
-
-
-def _idatui_printable(v):
- """Whether ``v`` would actually render as a character constant. IDA accepts
- op_chr on anything and then prints the number anyway, so a cycle that offers
- 'char' for 0x18 has a stop where nothing visibly happens."""
- if v is None or v < 0 or v > 0xFFFFFFFF:
- return False
- bs, x = [], int(v)
- while True:
- bs.append(x & 0xFF)
- x >>= 8
- if not x:
- break
- return all(0x20 <= b <= 0x7E or b in (9, 10, 13) for b in bs)
-
-
-def _idatui_offset_worth(v):
- """Whether 'offset' is worth OFFERING as a cycle stop for value ``v``.
-
- Making an offset is not free: IDA invents a dummy name at the target
- (``off_18``) and that name STAYS once you cycle past it. So the ring only
- stops there when the target is already something you could name -- a symbol,
- a function, or an item something else references. In a PIE at base 0 half
- the small constants in a function are 'mapped' (they land in the ELF
- header); ``sub rsp, 18h`` is not a reference and must not offer to become
- one on the way past.
-
- An explicit request still converts anything mapped: that's a decision, not a
- keypress that happened to land here. After it, the target HAS a name, so the
- ring includes the stop from then on."""
- import ida_bytes
- import ida_name
-
- return bool(v and ida_bytes.is_mapped(v) and ida_name.get_ea_name(v))
-
-
-def _idatui_op_candidates(ea):
- """Operand indices at ``ea`` whose display format is worth changing.
-
- Immediates and displacements -- the literals. Deliberately NOT:
-
- * branch targets (o_near/o_far), or every jump on the listing would offer to
- become a bare number, on a view you navigate by label;
- * memory references (o_mem), e.g. x86-64's RIP-relative ``lea rdi, name``.
- IDA prints those from the reference, not from the operand's number format,
- so setting one is accepted and changes nothing on screen -- a keypress
- that appears to do nothing is worse than one that says it can't.
-
- An explicit ``n`` still reaches them; this is what a bare cursor picks."""
- import ida_bytes
- import ida_ua
-
- F = ida_bytes.get_flags(ea)
- if ida_bytes.is_data(F):
- return [0] # a data item's value is operand 0
- if not ida_bytes.is_code(F):
- return [] # undefined bytes: IDA refuses a format outright
- insn = ida_ua.insn_t()
- if ida_ua.decode_insn(insn, ea) <= 0:
- return []
- want = (ida_ua.o_imm, ida_ua.o_displ)
- out = []
- for i in range(len(insn.ops)):
- op = insn.ops[i]
- if op.type == ida_ua.o_void:
- break
- if op.type in want:
- out.append(i)
- return out
-
-
-def _idatui_op_spans(ea, text):
- """[(start, end, n)] -- where each operand sits inside ``text`` (the
- whitespace-collapsed line the TUI shows), so a cursor column can name the
- operand it is standing on.
-
- Read out of IDA's own COLOR_OPND markers on the line, which is both free
- (the line is generated anyway) and exact. print_operand is kept as a
- fallback for a processor module that emits no operand markers -- it agrees
- with the tags where both exist, but it re-renders every operand to say so.
- """
- import ida_lines
- import ida_ua
-
- line = ida_lines.generate_disasm_line(ea, 0)
- if line:
- _spans, ops = _idatui_spans(line)
- if ops:
- return [tuple(o) for o in ops]
-
- out, pos = [], 0
- for n in range(8):
- try:
- raw = ida_ua.print_operand(ea, n)
- except Exception:
- raw = None
- if not raw:
- continue
- op = " ".join(ida_lines.tag_remove(raw).split())
- if not op:
- continue
- i = text.find(op, pos)
- if i < 0: # duplicated operand text (mov eax, eax)
- i = text.find(op)
- if i < 0:
- continue
- out.append((i, i + len(op), n))
- pos = i + len(op)
- return out
-
-
-def _idatui_line_text(ea):
- import ida_lines
- line = ida_lines.generate_disasm_line(ea, 0)
- return " ".join(ida_lines.tag_remove(line).split()) if line else ""
-
-
-def _idatui_op_text(ea, text, n):
- """How operand ``n`` reads on the line, for a message that names it."""
- for lo, hi, i in _idatui_op_spans(ea, text):
- if i == int(n):
- return text[lo:hi].strip()
- return ""
-
-
-def _idatui_apply_fmt(ea, n, fmt):
- """Set operand ``n``'s display format. Returns (ok, error)."""
- import ida_bytes
- import ida_offset
- import idaapi
-
- n = int(n)
- if fmt == "default":
- return bool(ida_bytes.clr_op_type(ea, n)), ""
- if fmt == "offset":
- base = ida_offset.calc_offset_base(ea, n)
- if base in (idaapi.BADADDR, None) or base < 0:
- base = 0
- return bool(ida_offset.op_plain_offset(ea, n, base)), ""
- fn = {"hex": ida_bytes.op_hex, "dec": ida_bytes.op_dec,
- "oct": ida_bytes.op_oct, "bin": ida_bytes.op_bin,
- "char": ida_bytes.op_chr, "seg": ida_bytes.op_seg,
- "float": ida_bytes.op_flt, "stack": ida_bytes.op_stkvar}.get(fmt)
- if fn is None:
- return False, (f"can't set {fmt!r} from a name alone"
- if fmt in _idatui_fmt_nibbles() else
- f"unknown format {fmt!r}")
- return bool(fn(ea, n)), ""
-
-
-@tool
-@idasync
-def op_format(
- addr: Annotated[str, "Address of the instruction or data item"],
- mode: Annotated[str, "cycle | back | show | hex | dec | oct | bin | char | offset | stack | default"] = "cycle",
- col: Annotated[int, "Cursor column inside the rendered line (-1: first literal)"] = -1,
- n: Annotated[int, "Operand index; -1 derives it from ``col``"] = -1,
-) -> dict:
- """Change how a literal is DISPLAYED (IDA's 'o' family): hex, decimal,
- binary, character, or an offset to the address it names.
-
- The value in the bytes never changes -- only the representation IDA renders
- and remembers. ``cycle``/``back`` step the stops that make sense for THIS
- operand: 'char' is skipped unless the value prints as one, 'offset' unless
- the target is already named, so no press is ever a no-op you have to press
- again. ``show`` reports without changing anything.
-
- A format the ring can't hold (a stack variable, an enum) is reported in
- ``warn`` on the way out, with what to do about it -- ``mode`` takes any of
- the names above outright, which is also how you put one back.
-
- Which operand: ``n`` if given, else the one under ``col`` (a column in the
- whitespace-collapsed line, as ``heads`` renders it), else the first literal
- on the line."""
- import ida_bytes
-
- try:
- ea = ida_bytes.get_item_head(parse_address(addr))
- except Exception as e:
- return {"addr": str(addr), "error": str(e)}
-
- before = _idatui_line_text(ea)
- cands = _idatui_op_candidates(ea)
- n = int(n)
- if n < 0:
- n = -1
- if int(col) >= 0:
- for lo, hi, i in _idatui_op_spans(ea, before):
- if not (lo <= int(col) < hi):
- continue
- if i in cands:
- n = i
- break
- # The cursor IS on an operand, just not one with a format. The
- # client highlights what the cursor is on, so quietly moving to
- # a different operand would make that highlight a lie -- say
- # which one can be changed instead.
- where = before[lo:hi].strip()
- alt = (f"; the literal on this line is operand {cands[0]} "
- f"({_idatui_op_text(ea, before, cands[0])})"
- if cands else "")
- return {"addr": hex(ea), "n": i, "text": before,
- "error": f"operand {i} ({where}) has no format to "
- f"change{alt}"}
- if n < 0:
- if not cands:
- F = ida_bytes.get_flags(ea)
- why = ("no literal on this line to reformat"
- if ida_bytes.is_code(F) or ida_bytes.is_data(F) else
- "undefined bytes have no format to change -- define "
- "them first ('d' makes data, 'c' makes code)")
- return {"addr": hex(ea), "text": before, "error": why}
- n = cands[0]
-
- cur = _idatui_op_fmt(ea, n)
- value, width = _idatui_op_value(ea, n)
- mapped = value is not None and value != 0 and ida_bytes.is_mapped(value)
- # The ring is a property of the OPERAND, not of what you last pressed: every
- # stop is one that changes what you see for this value, and it is the same
- # ring at every step, so a lap always comes home.
- choices = [f for f in _IDATUI_FMT_CYCLE
- if (f != "char" or _idatui_printable(value))
- and (f != "offset" or _idatui_offset_worth(value))]
- # A stack variable is deliberately NOT a stop: ``[rbp+var_40]`` is a frame
- # member, not a way of writing a number, and IDA's own "is this a stack
- # variable" test isn't exposed to Python here (calc_stkvar_struc_offset
- # happily answers for ``[r14+8]`` too, which would put a bogus stop in the
- # ring). Leaving one is reported instead, with the command that undoes it.
- lossy = cur not in choices and cur != "default"
-
- mode = str(mode or "cycle").lower()
- if mode == "show":
- return {"addr": hex(ea), "n": n, "format": cur, "prev": cur,
- "choices": choices, "text": before, "before": before,
- "value": None if value is None else hex(value),
- "width": width, "applied": False}
- if mode in ("cycle", "back"):
- step = 1 if mode == "cycle" else -1
- if cur in choices:
- want = choices[(choices.index(cur) + step) % len(choices)]
- else:
- # Standing on a format the ring can't hold (an enum names a type a
- # nibble doesn't record): enter the ring at its end, don't skip a
- # stop working out where we "would have" been.
- want = choices[0] if step > 0 else choices[-1]
- else:
- want = mode
- if want not in _idatui_fmt_nibbles():
- return {"addr": hex(ea), "n": n, "text": before,
- "error": f"unknown format {mode!r}; one of "
- + ", ".join(_IDATUI_FMT_SETTABLE)}
- if want == "offset" and not mapped:
- return {"addr": hex(ea), "n": n, "text": before, "format": cur,
- "error": (f"{'0x%x' % value if value is not None else 'this operand'}"
- " isn't a mapped address -- an offset to it would"
- " invent a name for nothing")}
-
- ok, err = _idatui_apply_fmt(ea, n, want)
- if err:
- return {"addr": hex(ea), "n": n, "text": before, "format": cur,
- "error": err}
- got = _idatui_op_fmt(ea, n)
- out = {"addr": hex(ea), "n": n, "prev": cur, "format": got,
- "requested": want, "applied": bool(ok), "choices": choices,
- "before": before, "text": _idatui_line_text(ea),
- "value": None if value is None else hex(value), "width": width}
- if not ok:
- out["error"] = f"IDA refused {want} on operand {n}"
- elif lossy:
- out["warn"] = (
- f"operand {n} was {cur} and the ring has no stop there -- "
- + (f"'{cur}' sets it again" if cur in _IDATUI_FMT_SETTABLE else
- f"{cur} names a type this can't put back, reassign it by hand"))
- return out
-
-
-# --------------------------------------------------------------------------- #
-# the same thing in the decompiler (Hex-Rays keeps its own number formats)
-# --------------------------------------------------------------------------- #
-#: Hex-Rays prints C, so two of the listing's stops are missing here: binary
-#: (C has no binary literal -- the format takes and then renders decimal, which
-#: would be a lie on screen) and offset (it makes the function fail to
-#: decompile outright).
-_IDATUI_PC_FMT_CYCLE = ("hex", "dec", "oct", "char", "default")
-
-
-def _idatui_compact(line):
- """The ida-pro-mcp whitespace collapse the pseudocode is served through, so
- a column in what the client SHOWS can be mapped back to Hex-Rays' line."""
- try:
- from ida_pro_mcp.ida_mcp.utils import compact_whitespace
- return compact_whitespace(line)
- except Exception:
- import re as _re
- stripped = line.lstrip(" \t")
- lead = line[: len(line) - len(stripped)]
- return lead + _re.sub(r"[ \t]{2,}", " ", stripped)
-
-
-def _idatui_compact_col(plain, compact, col):
- """The inverse of ``_idatui_uncompact_col``: a column in Hex-Rays' own line,
- expressed in the collapsed line the client shows."""
- j = 0
- for i in range(min(int(col), len(plain))):
- if j < len(compact) and plain[i] == compact[j]:
- j += 1
- return j
-
-
-def _idatui_uncompact_col(plain, compact, col):
- """Map a column in the collapsed line back to the same character in the
- original. The transform only ever DELETES spaces, so walking both in step
- and skipping what vanished is exact."""
- i = 0
- for j in range(min(int(col), len(compact))):
- c = compact[j]
- while i < len(plain) and plain[i] != c:
- i += 1
- i += 1
- return min(i, max(len(plain) - 1, 0))
-
-
-#: Characters that can be part of a C number literal as Hex-Rays prints one
-#: (digits, hex letters, the 0x prefix, u/L suffixes).
-_IDATUI_LIT_CHARS = frozenset("0123456789abcdefABCDEFxXuUlL")
-
-
-def _idatui_lit_extent(plain, x):
- """The [start, end) of the literal token containing column ``x``.
-
- Hex-Rays says WHICH item a column belongs to, but not how wide the printed
- literal is -- and it attributes neighbouring punctuation to the same item,
- so ``if ( a1 > 1 )`` reports the closing paren as part of the number. The
- identity comes from the ctree; the extent is the run of literal characters
- around the column, which cannot reach a ``)`` or a space."""
- if x >= len(plain):
- return None
- if plain[x] == "'": # a character constant: '-'
- end = plain.find("'", x + 1)
- return (x, end + 1) if end > x else None
- lo = plain.rfind("'", 0, x)
- if lo >= 0 and plain.find("'", x) > x and "'" in plain[lo:x] and \
- plain[lo:x].count("'") == 1 and " " not in plain[lo:x]:
- return (lo, plain.find("'", x) + 1) # inside 'c'
- if plain[x] not in _IDATUI_LIT_CHARS:
- return None
- lo = x
- while lo > 0 and plain[lo - 1] in _IDATUI_LIT_CHARS:
- lo -= 1
- hi = x
- while hi < len(plain) and plain[hi] in _IDATUI_LIT_CHARS:
- hi += 1
- if lo > 0 and plain[lo - 1] == "-": # a unary minus is part of it
- lo -= 1
- return (lo, hi)
-
-
-def _idatui_pc_nums(cf, sl):
- """Every number literal on one pseudocode line, as
- [{x0, x1, ea, opnum, value, nbytes, fmt}].
-
- Asks Hex-Rays what each column belongs to rather than pattern-matching the
- text: a regex over ``v6 = a1 - 1;`` has to guess which of those characters
- are a literal, and ``v11`` looks like one."""
- import ida_bytes
- import ida_hexrays
- import ida_lines
- import idaapi
-
- plain = ida_lines.tag_remove(sl.line)
- out = []
- x = 0
- while x < len(plain):
- ch = plain[x]
- if ch not in _IDATUI_LIT_CHARS and ch != "'":
- x += 1
- continue
- head, item, tail = (ida_hexrays.ctree_item_t() for _ in range(3))
- if not cf.get_line_item(sl.line, x, True, head, item, tail):
- x += 1
- continue
- if item.citype != ida_hexrays.VDI_EXPR:
- x += 1
- continue
- e = item.e
- if e.op != ida_hexrays.cot_num:
- x += 1
- continue
- extent = _idatui_lit_extent(plain, x)
- if extent is None:
- x += 1
- continue
- nf = e.n.nf
- opnum = ord(nf.opnum) if isinstance(nf.opnum, str) else int(nf.opnum)
- nbytes = (ord(nf.org_nbytes) if isinstance(nf.org_nbytes, str)
- else int(nf.org_nbytes))
- ea = int(e.ea)
- if ea == idaapi.BADADDR:
- x = extent[1]
- continue # synthesised: nothing to key on
- nib = (nf.flags >> ida_bytes.get_operand_type_shift(opnum)) & 0xF
- # Whether this format is the USER's or Hex-Rays' own guess. The nibble
- # can't say: an untouched number reads back as whatever it happens to
- # be printed as, and cycling from there would skip that stop forever
- # (default already looks like it) and never come back to it.
- loc = ida_hexrays.operand_locator_t(ea, opnum)
- user = (ida_hexrays.user_numforms_find(cf.numforms, loc)
- != ida_hexrays.user_numforms_end(cf.numforms))
- out.append({"x0": extent[0], "x1": extent[1], "ea": ea,
- "opnum": opnum, "value": int(e.n._value),
- "nbytes": nbytes, "user": user,
- "fmt": _idatui_fmt_name(nib) if user else "default",
- "shown": _idatui_fmt_name(nib)})
- x = extent[1] # past this literal, not into it
- return out
-
-
-@tool
-@idasync
-def pc_nums(
- addr: Annotated[str, "Function address (or any address inside it)"],
-) -> dict:
- """Every number literal in a function's pseudocode, as
- [{line, x0, x1, ea, opnum, value, fmt, user}].
-
- One call per decompilation, so a client can show WHICH literal the cursor is
- on (and reformat exactly that one) without a round trip per cursor move.
- Columns are in the same collapsed coordinates the decompile tool serves its
- text in, i.e. what the client actually displays."""
- import ida_hexrays
- import ida_lines
- import idaapi
-
- if not ida_hexrays.init_hexrays_plugin():
- return {"addr": str(addr), "error": "no decompiler", "nums": []}
- try:
- f = idaapi.get_func(parse_address(addr))
- except Exception as e:
- return {"addr": str(addr), "error": str(e), "nums": []}
- if f is None:
- return {"addr": str(addr), "error": "no function here", "nums": []}
- try:
- cf = ida_hexrays.decompile(f.start_ea)
- except Exception as e:
- return {"addr": hex(f.start_ea), "error": f"decompile failed: {e}",
- "nums": []}
- if cf is None:
- return {"addr": hex(f.start_ea), "error": "decompilation failed",
- "nums": []}
- sv = cf.get_pseudocode()
- out = []
- for i in range(len(sv)):
- plain = ida_lines.tag_remove(sv[i].line)
- compact = _idatui_compact(plain)
- for rec in _idatui_pc_nums(cf, sv[i]):
- out.append({
- "line": i,
- "x0": _idatui_compact_col(plain, compact, rec["x0"]),
- "x1": _idatui_compact_col(plain, compact, rec["x1"]),
- "ea": hex(rec["ea"]), "opnum": rec["opnum"],
- "value": hex(rec["value"]), "fmt": rec["fmt"],
- "shown": rec["shown"], "user": bool(rec["user"]),
- })
- return {"addr": hex(f.start_ea), "nums": out, "lines": len(sv)}
-
-
-@tool
-@idasync
-def pc_num_format(
- addr: Annotated[str, "Function address (or any address inside it)"],
- mode: Annotated[str, "cycle | back | show | hex | dec | oct | char | default"] = "cycle",
- line: Annotated[int, "0-based pseudocode line index"] = -1,
- col: Annotated[int, "Cursor column in the DISPLAYED line (-1: first literal)"] = -1,
- ea: Annotated[str, "Address of the number instead of line/col"] = "",
- opnum: Annotated[int, "Operand number, with ``ea``"] = -1,
-) -> dict:
- """Change how a number is displayed in the DECOMPILATION (Hex-Rays keeps its
- own number formats, per (address, operand), independent of the listing).
-
- Same stops as ``op_format`` minus the two C can't express: binary (no such
- literal -- IDA takes the format and prints decimal anyway) and offset (it
- makes the function stop decompiling). Returns the re-rendered line, and
- marks the function dirty so the next decompile is the new text."""
- import ida_hexrays
- import ida_lines
- import idaapi
-
- if not ida_hexrays.init_hexrays_plugin():
- return {"addr": str(addr), "error": "no decompiler"}
- try:
- f = idaapi.get_func(parse_address(addr))
- except Exception as e:
- return {"addr": str(addr), "error": str(e)}
- if f is None:
- return {"addr": str(addr), "error": "no function here"}
- try:
- cf = ida_hexrays.decompile(f.start_ea)
- except Exception as e:
- return {"addr": hex(f.start_ea), "error": f"decompile failed: {e}"}
- if cf is None:
- return {"addr": hex(f.start_ea), "error": "decompilation failed"}
-
- sv = cf.get_pseudocode()
- line = int(line)
- target = None
- if ea:
- try:
- want_ea = parse_address(ea)
- except Exception as e:
- return {"addr": hex(f.start_ea), "error": str(e)}
- for i in range(len(sv)):
- for rec in _idatui_pc_nums(cf, sv[i]):
- if rec["ea"] == want_ea and (int(opnum) < 0
- or rec["opnum"] == int(opnum)):
- target, line = rec, i
- break
- if target:
- break
- elif 0 <= line < len(sv):
- nums = _idatui_pc_nums(cf, sv[line])
- if nums:
- if int(col) >= 0:
- plain = ida_lines.tag_remove(sv[line].line)
- x = _idatui_uncompact_col(plain, _idatui_compact(plain), int(col))
- target = next((r for r in nums if r["x0"] <= x < r["x1"]), None)
- target = target or nums[0]
- else:
- return {"addr": hex(f.start_ea),
- "error": f"line {line} is outside the {len(sv)}-line decompilation"}
- if target is None:
- return {"addr": hex(f.start_ea), "line": line,
- "text": (ida_lines.tag_remove(sv[line].line).strip()
- if 0 <= line < len(sv) else ""),
- "error": "no number literal on this line"}
-
- cur, value = target["fmt"], target["value"]
- choices = [c for c in _IDATUI_PC_FMT_CYCLE
- if c != "char" or _idatui_printable(value)]
- # Same rule as the listing: one ring per literal, every step. A format the
- # ring can't hold (an enum set in the GUI) is reported on the way out
- # instead of being kept for one lap and then lost.
- lossy = cur not in choices and cur != "default"
- out = {"addr": hex(f.start_ea), "ea": hex(target["ea"]),
- "opnum": target["opnum"], "line": line, "prev": cur,
- "format": cur, "shown": target["shown"], "choices": choices,
- "value": hex(value),
- "before": ida_lines.tag_remove(sv[line].line).strip()}
-
- mode = str(mode or "cycle").lower()
- if mode == "show":
- out["text"] = out["before"]
- out["applied"] = False
- return out
- if mode in ("cycle", "back"):
- step = 1 if mode == "cycle" else -1
- if cur in choices:
- want = choices[(choices.index(cur) + step) % len(choices)]
- else:
- want = choices[0] if step > 0 else choices[-1]
- else:
- want = mode
- if want in ("bin", "offset", "stack", "seg", "float"):
- out["error"] = (f"Hex-Rays has no {want} format for a number "
- f"-- set it on the listing instead")
- out["text"] = out["before"]
- return out
- if want not in ("hex", "dec", "oct", "char", "default"):
- out["error"] = (f"unknown format {mode!r}; one of hex, dec, oct, "
- f"char, default")
- out["text"] = out["before"]
- return out
-
- loc = ida_hexrays.operand_locator_t(target["ea"], target["opnum"])
- it = ida_hexrays.user_numforms_find(cf.numforms, loc)
- if it != ida_hexrays.user_numforms_end(cf.numforms):
- # std::map::insert is a no-op on an existing key, so a format already
- # set here would silently win over the new one.
- ida_hexrays.user_numforms_erase(cf.numforms, it)
- if want != "default":
- import ida_bytes
- nf = ida_hexrays.number_format_t(target["opnum"])
- nf.flags = ida_bytes.get_operand_flag(_idatui_fmt_nibbles()[want],
- target["opnum"])
- try:
- nf.org_nbytes = target["nbytes"]
- except Exception:
- pass
- ida_hexrays.user_numforms_insert(cf.numforms, loc, nf)
- cf.save_user_numforms()
- try:
- ida_hexrays.mark_cfunc_dirty(f.start_ea)
- except Exception:
- pass
-
- out["format"] = want
- out["applied"] = True
- if lossy:
- out["warn"] = (f"this number was {cur}, which names a type a radix "
- f"can't put back -- reassign it in IDA")
- try:
- cf2 = ida_hexrays.decompile(f.start_ea,
- flags=ida_hexrays.DECOMP_NO_CACHE)
- sv2 = cf2.get_pseudocode() if cf2 is not None else None
- out["text"] = (ida_lines.tag_remove(sv2[line].line).strip()
- if sv2 is not None and line < len(sv2) else out["before"])
- except Exception as e:
- out["text"] = out["before"]
- out["warn"] = f"re-render failed: {e}"
- return out
-
-
-@tool
-@idasync
-def flowchart(
- addr: Annotated[str, "Address or name inside the function to chart"],
-) -> dict:
- """Basic-block control-flow graph of the function containing ``addr``.
-
- Returns the blocks and the edges between them -- NOT their text: the block
- body is just an address range, which the client already knows how to render
- with ``heads``. Keeping text out means the graph view reuses the exact same
- listing rows (colours, operand marks and all) instead of growing a second
- disassembly renderer.
-
- Edge ``kind`` is what the graph view colours by:
- * ``fall`` -- control falls through to the next address (IDA draws red)
- * ``jump`` -- a taken conditional branch (green)
- * ``uncond`` -- the block's only successor (blue)
- * ``switch`` -- one of an n-way dispatch
- """
- import ida_funcs
- import ida_gdl
-
- try:
- ea = parse_address(addr)
- except Exception as e:
- return {"addr": str(addr), "error": str(e), "blocks": []}
- fn = ida_funcs.get_func(ea)
- if fn is None:
- return {"addr": str(addr), "error": "no function at that address",
- "blocks": []}
-
- fc = ida_gdl.FlowChart(fn, flags=ida_gdl.FC_PREDS)
- index = {}
- order = []
- for bb in fc:
- index[bb.start_ea] = len(order)
- order.append(bb)
- blocks = []
- for bb in order:
- sl = [s for s in bb.succs() if s.start_ea in index]
- succs = []
- for s in sl:
- if len(sl) > 2:
- kind = "switch"
- elif s.start_ea == bb.end_ea:
- kind = "fall"
- else:
- kind = "jump"
- succs.append([index[s.start_ea], kind])
- blocks.append({
- "id": index[bb.start_ea],
- "start": hex(bb.start_ea),
- "end": hex(bb.end_ea),
- "succs": succs,
- })
- return {
- "addr": hex(ea),
- "func": {"addr": hex(fn.start_ea), "end": hex(fn.end_ea),
- "name": ida_funcs.get_func_name(fn.start_ea)},
- "entry": index.get(fn.start_ea, 0),
- "blocks": blocks,
- }
-'''
-
-SNIPPET = f"{BEGIN}\n{BODY.strip()}\n{END}\n"
-
-
-def api_types_path() -> pathlib.Path | None:
- """Locate ida_pro_mcp/ida_mcp/api_types.py without importing it (importing the
- submodule would pull in IDA, which isn't available outside a worker)."""
- spec = importlib.util.find_spec("ida_pro_mcp") # top-level pkg is IDA-free
- if spec is None or not spec.submodule_search_locations:
- return None
- p = pathlib.Path(spec.submodule_search_locations[0]) / "ida_mcp" / "api_types.py"
- return p if p.exists() else None
-
-
-def main() -> int:
- path = api_types_path()
- if path is None:
- print("idatui: ida_pro_mcp not found; skipping tool injection", file=sys.stderr)
- return 0
- text = path.read_text()
- if BEGIN in text and END in text: # replace the existing block in place
- pre = text[: text.index(BEGIN)].rstrip()
- post = text[text.index(END) + len(END):].lstrip("\n")
- new = pre + "\n\n" + SNIPPET + ("\n" + post if post else "")
- else:
- new = text.rstrip() + "\n\n" + SNIPPET
- if new == text:
- return 0
- try:
- path.write_text(new)
- except OSError as e:
- print(f"idatui: could not patch {path}: {e}", file=sys.stderr)
- return 1
- print(f"idatui: injected/updated idatui-ext tools in {path}", file=sys.stderr)
- return 0
-
-
-if __name__ == "__main__":
- raise SystemExit(main())