Download lib/forge/binary_ir.ex from Snapkitty/forge-code: direct link, hf CLI and curl.
- Browser
- Download file 12.9 kB
-
https://huggingface.co/datasets/Snapkitty/forge-code/resolve/main/lib/forge/binary_ir.ex
- Command line
-
hf download hf://datasets/Snapkitty/forge-code/lib/forge/binary_ir.ex
-
curl -L -o binary_ir.ex https://huggingface.co/datasets/Snapkitty/forge-code/resolve/main/lib/forge/binary_ir.ex
12.9 kB
| defmodule Forge.BinaryIR do | |
| @moduledoc """ | |
| BinaryIR - Normalized Binary Intermediate Representation. | |
| This module provides a stable, normalized intermediate representation for | |
| BEAM binary structures. It preserves all original offsets and provides | |
| semantic meaning where possible, while distinguishing between: | |
| - RAW: Direct binary data | |
| - DECODED: Structurally parsed binary data | |
| - NORMALIZED: Implementation-independent representation | |
| - SEMANTIC: Inferred higher-level meaning | |
| ## Design Principles | |
| 1. **Offset Preservation**: Every decoded field retains its original offset | |
| 2. **Provenance Tracking**: Every object is traceable back to source | |
| 3. **State Distinction**: Never represent inferred meaning as raw data | |
| 4. **Version Awareness**: Handle different BEAM versions explicitly | |
| """ | |
| alias Forge.Hash | |
| alias Forge.Beam | |
| @type state :: :raw | :decoded | :normalized | :semantic | |
| @type binary_field :: %{ | |
| section: atom(), | |
| offset: integer(), | |
| length: integer(), | |
| bytes: binary(), | |
| decoded: any(), | |
| normalized: any(), | |
| semantic: any(), | |
| state: state(), | |
| confidence: :low | :medium | :high, | |
| endianness: :big | :little | :unknown, | |
| encoding: atom() | |
| } | |
| @type section :: %{ | |
| name: atom(), | |
| offset: integer(), | |
| length: integer(), | |
| fields: list(binary_field()), | |
| metadata: map() | |
| } | |
| @type binary_ir :: %{ | |
| module: atom() | nil, | |
| version: String.t() | nil, | |
| sections: list(section()), | |
| constants: list(map()), | |
| functions: list(map()), | |
| metadata: map(), | |
| hashes: %{atom() => binary()}, | |
| provenance: map(), | |
| warnings: list(String.t()) | |
| } | |
| @doc """ | |
| Creates a BinaryIR from a BEAM file. | |
| ## Parameters | |
| - path: Path to the BEAM file | |
| - options: Analysis options | |
| ## Returns | |
| - {:ok, binary_ir} with complete normalized representation | |
| - {:error, reason} on failure | |
| """ | |
| @spec from_beam(String.t(), map()) :: {:ok, binary_ir()} | {:error, String.t()} | |
| def from_beam(path, options \\ []) do | |
| # Read the BEAM file | |
| case Beam.read(path) do | |
| {:ok, metadata} -> | |
| # Start building the IR | |
| ir = %{ | |
| module: metadata.module, | |
| version: metadata.version, | |
| sections: [], | |
| constants: [], | |
| functions: [], | |
| metadata: %{ | |
| compiler: metadata.compiler, | |
| source: metadata.source, | |
| md5: metadata.md5, | |
| debug_info: metadata.debug_info, | |
| line_info: metadata.line_info | |
| }, | |
| hashes: %{}, | |
| provenance: %{ | |
| beam_path: path, | |
| timestamp: DateTime.utc_now(), | |
| beam_size: metadata.raw_size | |
| }, | |
| warnings: [] | |
| } | |
| # Add sections | |
| ir = add_sections(ir, metadata) | |
| # Add constants | |
| ir = add_constants(ir, metadata) | |
| # Add functions | |
| ir = add_functions(ir, metadata) | |
| # Add hashes | |
| ir = add_hashes(ir, metadata, path) | |
| {:ok, ir} | |
| {:error, reason} -> | |
| {:error, reason} | |
| end | |
| end | |
| @doc """ | |
| Adds sections to the BinaryIR. | |
| """ | |
| defp add_sections(ir, metadata) do | |
| sections = [] | |
| # Code section | |
| if offset = metadata.section_offsets[:code] do | |
| sections = sections ++ [ | |
| %{ | |
| name: :code, | |
| offset: offset, | |
| length: calculate_code_length(metadata), | |
| fields: extract_code_fields(metadata), | |
| metadata: %{} | |
| } | |
| ] | |
| end | |
| # Atoms section | |
| if offset = metadata.section_offsets[:atoms] do | |
| sections = sections ++ [ | |
| %{ | |
| name: :atoms, | |
| offset: offset, | |
| length: calculate_atoms_length(metadata), | |
| fields: extract_atoms_fields(metadata), | |
| metadata: %{} | |
| } | |
| ] | |
| end | |
| # Literals section | |
| if offset = metadata.section_offsets[:literals] do | |
| sections = sections ++ [ | |
| %{ | |
| name: :literals, | |
| offset: offset, | |
| length: calculate_literals_length(metadata), | |
| fields: extract_literals_fields(metadata), | |
| metadata: %{} | |
| } | |
| ] | |
| end | |
| # Functions section | |
| if offset = metadata.section_offsets[:functions] do | |
| sections = sections ++ [ | |
| %{ | |
| name: :functions, | |
| offset: offset, | |
| length: calculate_functions_length(metadata), | |
| fields: extract_functions_fields(metadata), | |
| metadata: %{} | |
| } | |
| ] | |
| end | |
| # Imports section | |
| if offset = metadata.section_offsets[:imports] do | |
| sections = sections ++ [ | |
| %{ | |
| name: :imports, | |
| offset: offset, | |
| length: calculate_imports_length(metadata), | |
| fields: extract_imports_fields(metadata), | |
| metadata: %{} | |
| } | |
| ] | |
| end | |
| # Exports section | |
| if offset = metadata.section_offsets[:exports] do | |
| sections = sections ++ [ | |
| %{ | |
| name: :exports, | |
| offset: offset, | |
| length: calculate_exports_length(metadata), | |
| fields: extract_exports_fields(metadata), | |
| metadata: %{} | |
| } | |
| ] | |
| end | |
| %{ir | sections: sections} | |
| end | |
| @doc """ | |
| Adds constants to the BinaryIR. | |
| """ | |
| defp add_constants(ir, metadata) do | |
| constants = [] | |
| # Atoms as constants | |
| if atoms = metadata.atoms do | |
| constants = constants ++ Enum.map(atoms, fn atom, idx -> | |
| %{ | |
| type: :atom, | |
| value: atom, | |
| offset: metadata.section_offsets[:atoms] + idx, | |
| index: idx, | |
| state: :decoded, | |
| confidence: :high | |
| } | |
| end) | |
| end | |
| # Literals as constants | |
| if literals = metadata.literals do | |
| constants = constants ++ Enum.map(literals, fn literal, idx -> | |
| %{ | |
| type: type_of(literal), | |
| value: literal, | |
| offset: metadata.section_offsets[:literals] + idx, | |
| index: idx, | |
| state: :decoded, | |
| confidence: :high | |
| } | |
| end) | |
| end | |
| %{ir | constants: constants} | |
| end | |
| @doc """ | |
| Adds functions to the BinaryIR. | |
| """ | |
| defp add_functions(ir, metadata) do | |
| functions = case metadata.functions do | |
| nil -> [] | |
| funcs -> | |
| Enum.map(funcs, fn func -> | |
| %{ | |
| name: func.name, | |
| arity: func.arity, | |
| index: func.index, | |
| labels: func.labels || [], | |
| instructions: extract_normalized_instructions(func), | |
| offsets: extract_instruction_offsets(func), | |
| state: :normalized, | |
| confidence: :high | |
| } | |
| end) | |
| end | |
| %{ir | functions: functions} | |
| end | |
| @doc """ | |
| Adds hashes to the BinaryIR. | |
| """ | |
| defp add_hashes(ir, metadata, path) do | |
| # Hash the entire BEAM file | |
| case Forge.Hash.file(path) do | |
| {:ok, hash, size} -> | |
| %{ir | hashes: Map.put(ir.hashes, :beam_file, hash)} | |
| {:error, _} -> ir | |
| end | |
| end | |
| @doc """ | |
| Extracts normalized instructions from a BEAM function. | |
| Normalizes implementation-specific details while preserving: | |
| - Original offsets | |
| - Instruction semantics | |
| - Operand information | |
| """ | |
| defp extract_normalized_instructions(func) do | |
| case func.code do | |
| nil -> [] | |
| code -> | |
| Enum.map(code, fn {opcode, args, offset} -> | |
| # Normalize the opcode name | |
| normalized_opcode = normalize_opcode(opcode) | |
| # Normalize operands | |
| normalized_args = Enum.map(args, fn arg -> | |
| case arg do | |
| atom when is_atom(atom) -> %{type: :atom, value: atom, raw: arg} | |
| integer when is_integer(integer) -> %{type: :integer, value: integer, raw: arg} | |
| reference when is_reference(reference) -> %{type: :reference, value: reference, raw: arg} | |
| label when is_binary(label) -> %{type: :label, value: label, raw: arg} | |
| float when is_float(float) -> %{type: :float, value: float, raw: arg} | |
| _ -> %{type: :unknown, value: arg, raw: arg} | |
| end | |
| end) | |
| %{ | |
| opcode: normalized_opcode, | |
| args: args, | |
| normalized_args: normalized_args, | |
| offset: offset, | |
| label: nil, | |
| comments: [], | |
| state: :normalized, | |
| confidence: :high | |
| } | |
| end) | |
| end | |
| end | |
| @doc """ | |
| Normalizes opcode names to a consistent format. | |
| """ | |
| defp normalize_opcode(opcode) do | |
| # Convert atoms to strings for consistency | |
| case opcode do | |
| atom when is_atom(atom) -> Atom.to_string(atom) | |
| string when is_binary(string) -> string | |
| _ -> inspect(opcode) | |
| end | |
| end | |
| @doc """ | |
| Gets the type of a literal value. | |
| """ | |
| defp type_of(value) do | |
| cond do | |
| is_atom(value) -> :atom | |
| is_binary(value) -> :binary | |
| is_integer(value) -> :integer | |
| is_float(value) -> :float | |
| is_list(value) -> :list | |
| is_tuple(value) -> :tuple | |
| is_pid(value) -> :pid | |
| is_port(value) -> :port | |
| is_reference(value) -> :reference | |
| true -> :unknown | |
| end | |
| end | |
| @doc """ | |
| Calculates the length of the code section. | |
| """ | |
| defp calculate_code_length(metadata) do | |
| # This would be calculated from the actual binary | |
| # For now, estimate based on instruction count | |
| case metadata.functions do | |
| nil -> 0 | |
| functions -> | |
| Enum.reduce(functions, 0, fn func, acc -> | |
| acc + length(func.instructions || []) | |
| end) * 4 # Rough estimate | |
| end | |
| end | |
| @doc """ | |
| Calculates the length of the atoms section. | |
| """ | |
| defp calculate_atoms_length(metadata) do | |
| case metadata.atoms do | |
| nil -> 0 | |
| atoms -> length(atoms) * 4 # Atoms are typically 4 bytes each | |
| end | |
| end | |
| @doc """ | |
| Calculates the length of the literals section. | |
| """ | |
| defp calculate_literals_length(metadata) do | |
| case metadata.literals do | |
| nil -> 0 | |
| literals -> length(literals) * 8 # Rough estimate | |
| end | |
| end | |
| @doc """ | |
| Calculates the length of the functions section. | |
| """ | |
| defp calculate_functions_length(metadata) do | |
| case metadata.functions do | |
| nil -> 0 | |
| functions -> length(functions) * 20 # Rough estimate | |
| end | |
| end | |
| @doc """ | |
| Calculates the length of the imports section. | |
| """ | |
| defp calculate_imports_length(metadata) do | |
| case metadata.imports do | |
| nil -> 0 | |
| imports -> length(imports) * 8 # Rough estimate | |
| end | |
| end | |
| @doc """ | |
| Calculates the length of the exports section. | |
| """ | |
| defp calculate_exports_length(metadata) do | |
| case metadata.exports do | |
| nil -> 0 | |
| exports -> length(exports) * 8 # Rough estimate | |
| end | |
| end | |
| @doc """ | |
| Extracts code fields from BEAM metadata. | |
| """ | |
| defp extract_code_fields(metadata) do | |
| case metadata.functions do | |
| nil -> [] | |
| functions -> | |
| # This is a placeholder - actual implementation would parse the binary | |
| [] | |
| end | |
| end | |
| @doc """ | |
| Extracts atoms fields from BEAM metadata. | |
| """ | |
| defp extract_atoms_fields(metadata) do | |
| case metadata.atoms do | |
| nil -> [] | |
| atoms -> | |
| Enum.map_with_index(atoms, fn atom, idx -> | |
| %{ | |
| section: :atoms, | |
| offset: metadata.section_offsets[:atoms] + idx * 4, | |
| length: 4, | |
| bytes: <<>>, # Would be actual bytes from binary | |
| decoded: atom, | |
| normalized: atom, | |
| semantic: %{type: :module, value: atom}, | |
| state: :decoded, | |
| confidence: :high, | |
| endianness: :big, | |
| encoding: :atom | |
| } | |
| end) | |
| end | |
| end | |
| @doc """ | |
| Extracts literals fields from BEAM metadata. | |
| """ | |
| defp extract_literals_fields(metadata) do | |
| case metadata.literals do | |
| nil -> [] | |
| literals -> | |
| [] # Placeholder - actual implementation would parse | |
| end | |
| end | |
| @doc """ | |
| Extracts functions fields from BEAM metadata. | |
| """ | |
| defp extract_functions_fields(metadata) do | |
| [] # Placeholder - actual implementation would parse | |
| end | |
| @doc """ | |
| Extracts imports fields from BEAM metadata. | |
| """ | |
| defp extract_imports_fields(metadata) do | |
| [] # Placeholder - actual implementation would parse | |
| end | |
| @doc """ | |
| Extracts exports fields from BEAM metadata. | |
| """ | |
| defp extract_exports_fields(metadata) do | |
| [] # Placeholder - actual implementation would parse | |
| end | |
| @doc """ | |
| Extracts instruction offsets from a BEAM function. | |
| """ | |
| defp extract_instruction_offsets(func) do | |
| case func.code do | |
| nil -> [] | |
| code -> Enum.map(code, fn {_, _, offset} -> offset end) | |
| end | |
| end | |
| end | |