Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 34 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
name: CI

on:
push:
branches: [master]
pull_request:

jobs:
test:
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
python-version: ['3.10', '3.11', '3.12']
steps:
- uses: actions/checkout@v4

- name: Set up Python ${{ matrix.python-version }}
uses: actions/setup-python@v5
with:
python-version: ${{ matrix.python-version }}

- name: Install system libraries
# libfuzzy-dev: build ssdeep; libmagic1: runtime for python-magic
run: sudo apt-get update && sudo apt-get install -y libfuzzy-dev libmagic1

- name: Install Python dependencies
run: |
python -m pip install --upgrade pip
pip install -r requirements.txt
pip install pytest

- name: Run tests
run: pytest -v tests/
3 changes: 3 additions & 0 deletions requirements-dev.txt
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,9 @@

-r requirements.txt

# Testing
pytest>=8.0.0

# Experimentation / analysis
nltk>=3.9.4
matplotlib>=3.9.2
Expand Down
25 changes: 20 additions & 5 deletions src/MANTILLA.py
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,12 @@ def parse_arguments():
parser.add_argument("-b", "--binary", type=str, help="Specify the binary to analyze")
parser.add_argument("-j", "--json", type=str, help="Specify the features of a binary in JSON format")
parser.add_argument("-m", "--metric", type=str, default="euclidean", help="Specify the distance metric")
parser.add_argument("-t", "--threshold", type=float, default=1.0, help="Specify the distance threshold")
parser.add_argument("-t", "--threshold", type=float, default=1.0,
help="Maximum neighbor distance for a vote. This is an ABSOLUTE distance "
"in the chosen metric (euclidean by default) computed over UNNORMALIZED "
"features, so its scale depends on the feature magnitudes. A neighbor "
"only votes if its distance is <= THRESHOLD. Pass a negative value to "
"disable filtering and let every k-neighbor vote (default: 1.0)")
parser.add_argument("-k", "--neighbors", type=int, default=5, help="Specify the number of k-neighbors")
parser.add_argument("-f", "--file_model", type=str, default=os.path.join(os.path.dirname(os.path.abspath(__file__)), "features_model.csv"), help="Specify the features model CSV file")
parser.add_argument("-d", "--directory", type=str, help="Specify a directory with test files")
Expand Down Expand Up @@ -87,22 +92,32 @@ def train_knn_model(features_file, n_neighbors, metric):
return model, X, y


def classify_file(model, test_features, X, y, threshold=1):
def classify_file(model, test_features, y, threshold=None):
"""Collect the labels of each test function's k nearest neighbors.

``threshold`` is an ABSOLUTE distance in the model's metric (euclidean by
default) computed over UNNORMALIZED features: a neighbor only contributes
its label if its distance is <= ``threshold``. Pass ``None`` (or a negative
value) to disable filtering and let every one of the k neighbors vote.
"""
results = {"predictions": []}
df_test = pd.DataFrame(test_features)
X_test = df_test.values

if X_test.size == 0:
return results

# A negative CLI threshold means "no filtering"; normalize it to None here.
if threshold is not None and threshold < 0:
threshold = None

kneighbors_distance, kneighbors_index_labels = model.kneighbors(X_test)

predicted_labels = [
y[kneighbors_index_labels[d][index]]
for d in range(len(kneighbors_distance))
for index in range(len(kneighbors_distance[d]))
if kneighbors_distance[d][index] <= threshold
if threshold is None or kneighbors_distance[d][index] <= threshold
]

results["predictions"] = predicted_labels
Expand All @@ -114,11 +129,11 @@ def main():
args = parse_arguments()
test_files = prepare_test_files(args)

model, X, y = train_knn_model(args.file_model, args.neighbors, args.metric)
model, _, y = train_knn_model(args.file_model, args.neighbors, args.metric)

for test_file in test_files:
test_features = get_features_test([test_file])
results = classify_file(model, test_features, X, y, args.threshold)
results = classify_file(model, test_features, y, args.threshold)
print(f"File: {test_file}")

all_predictions = [pred for pred in results["predictions"]]
Expand Down
2 changes: 1 addition & 1 deletion src/binutils_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -89,7 +89,7 @@ def main():
label_predict = []
if len(X_test) != 0:

results = classify_file(model, X_test, X_train, y_train, 0.5)
results = classify_file(model, X_test, y_train, 0.5)
all_predictions = [pred for pred in results["predictions"]]

if all_predictions:
Expand Down
7 changes: 7 additions & 0 deletions tests/conftest.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
import os
import sys

# Make the modules under src/ importable as top-level modules in tests.
SRC = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "src"))
if SRC not in sys.path:
sys.path.insert(0, SRC)
59 changes: 59 additions & 0 deletions tests/test_feature_extraction.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,59 @@
"""Tests for the pure (non-radare2) logic in feature_extraction.py.

Importing the module pulls in native deps (magic, tlsh, ssdeep, r2pipe); if any
is unavailable the whole module is skipped rather than failing collection.
"""
import pytest

fe = pytest.importorskip("feature_extraction")


def test_func_offset_accepts_addr_or_offset():
# radare2 5.x exposes 'offset', 6.x exposes 'addr'.
assert fe.BinaryAnalyzer._func_offset({'offset': 0x1000}) == 0x1000
assert fe.BinaryAnalyzer._func_offset({'addr': 0x2000}) == 0x2000


def test_entropy_calculator_uniform_bytes():
# 256 distinct byte values -> 8 bits of entropy.
data = bytes(range(256))
assert fe.EntropyCalculator.calculate_entropy(data) == pytest.approx(8.0)
# A single repeated byte -> 0 entropy.
assert fe.EntropyCalculator.calculate_entropy(b"\x00" * 32) == pytest.approx(0.0)


def test_clean_function_names_strips_prefixes_and_suffixes():
out = fe.FunctionFilter.clean_function_names([
{'name': 'sym.imp.printf', 'offset': 1},
{'name': 'sym.my_func', 'offset': 2},
{'name': 'dbg.helper', 'offset': 3},
{'name': 'foo_2', 'offset': 4},
])
names = [c['name'] for c in out]
assert names == ['printf', 'my_func', 'helper', 'foo']


def test_remove_fnc_c_plus_exact_match_no_substring_overremoval():
clean = [{'name': n, 'offset': i} for i, n in enumerate(
['address_of', 'padding', 'read_config', 'add', 'main', 'real'])]
out = [c['name'] for c in fe.FunctionFilter.remove_fnc_c_plus(clean, {'add', 'read', 'main'})]
# 'add' removed (exact); 'main' kept; substring matches NOT removed.
assert 'add' not in out
assert 'main' in out
assert {'address_of', 'padding', 'read_config', 'real'} <= set(out)


def test_remove_glibc_uses_bundled_lists():
# Resolved against the script dir, so this works from any CWD.
clean = [{'name': 'malloc', 'offset': 1},
{'name': 'zzz_not_a_libc_fn_123', 'offset': 2}]
out = [c['name'] for c in fe.FunctionFilter.remove_glibc(clean)]
assert 'malloc' not in out # present in glibc_functions.txt
assert 'zzz_not_a_libc_fn_123' in out


def test_remove_known_fnc_suffix_and_prefix_rules():
clean = [{'name': n, 'offset': i} for i, n in enumerate(
['foo.cold', '_IO_helper', 'memcpy_avx2', 'my_unique_fn_xyz'])]
out = [c['name'] for c in fe.FunctionFilter.remove_known_fnc(clean)]
assert out == ['my_unique_fn_xyz']
92 changes: 92 additions & 0 deletions tests/test_mantilla.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
"""Tests for the classifier logic in MANTILLA.py.

These need only pandas + scikit-learn; importing MANTILLA does not pull in the
native binary-analysis dependencies (that import is lazy, used only by -b).
"""
import json

import MANTILLA

FEATURES = [
'cc', 'cost', 'size', 'stackframe', 'nbbs', 'ninst', 'edges', 'ebbs',
'noreturn', 'outdegree', 'nlocals', 'nargs', 'entropy', 'fnc_callgraph',
]


def _func(**overrides):
base = {k: 1 for k in FEATURES}
base['noreturn'] = False
base['fnc_callgraph'] = [0]
base.update(overrides)
return base


def test_get_feature_dict_full():
d = MANTILLA.get_feature_dict(_func(cc=3, size=40, fnc_callgraph=[0, 0, 0]))
assert d['cc'] == 3
assert d['size'] == 40
assert d['noreturn'] == 0 # bool -> int
assert d['fnc_callgraph'] == 3 # list -> length
assert set(d.keys()) == set(FEATURES)


def test_get_feature_dict_partial_uses_defaults():
# An incomplete function entry must not raise; missing fields default.
d = MANTILLA.get_feature_dict({'cc': 7, 'size': 12})
assert d['cc'] == 7 and d['size'] == 12
assert d['cost'] == 0 and d['nargs'] == 0
assert d['entropy'] == -1 # entropy sentinel default
assert d['fnc_callgraph'] == 0


def test_get_features_test_skips_file_key(tmp_path):
data = {
'file': {'file_name': 'whatever'},
'f0': _func(cc=2),
'f1': _func(cc=5),
}
p = tmp_path / "b.json"
p.write_text(json.dumps(data))
feats = MANTILLA.get_features_test([str(p)])
assert len(feats) == 2 # 'file' entry excluded
assert all(set(f.keys()) == set(FEATURES) for f in feats)


def _write_model(tmp_path):
# Two classes well separated in 'size'. Rows within a class are made
# distinct via 'cc' so train_knn_model's drop_duplicates() keeps them all.
rows = [",".join(FEATURES) + ",label"]
for size, label in ((1, "classA"), (1000, "classB")):
for i in range(1, 6):
vals = ["1"] * len(FEATURES)
vals[FEATURES.index('size')] = str(size)
vals[FEATURES.index('cc')] = str(i)
rows.append(",".join(vals) + "," + label)
p = tmp_path / "model.csv"
p.write_text("\n".join(rows) + "\n")
return str(p)


def test_train_and_classify_roundtrip(tmp_path):
model, _, y = MANTILLA.train_knn_model(_write_model(tmp_path), 3, "euclidean")
near_a = MANTILLA.get_feature_dict(_func(size=1))
near_b = MANTILLA.get_feature_dict(_func(size=1000))
ra = MANTILLA.classify_file(model, [near_a], y, threshold=None)
rb = MANTILLA.classify_file(model, [near_b], y, threshold=None)
assert set(ra["predictions"]) == {"classA"}
assert set(rb["predictions"]) == {"classB"}


def test_classify_threshold_semantics(tmp_path):
model, _, y = MANTILLA.train_knn_model(_write_model(tmp_path), 3, "euclidean")
far = MANTILLA.get_feature_dict(_func(size=500)) # ~499 away from class A
# None and negative disable filtering -> all k neighbors vote.
assert len(MANTILLA.classify_file(model, [far], y, threshold=None)["predictions"]) == 3
assert len(MANTILLA.classify_file(model, [far], y, threshold=-1)["predictions"]) == 3
# A tiny threshold filters out the far point's neighbors entirely.
assert MANTILLA.classify_file(model, [far], y, threshold=0.5)["predictions"] == []


def test_classify_empty_input(tmp_path):
model, _, y = MANTILLA.train_knn_model(_write_model(tmp_path), 3, "euclidean")
assert MANTILLA.classify_file(model, [], y, threshold=None)["predictions"] == []
Loading