Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 6 additions & 4 deletions admin/scripts/unwrap_berla_iva.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,10 +12,12 @@
AcquireDB.ive iVe's own parsed database, encrypted
Manifest.json, ECUData.json, DLCData.json, CaseData.json, Audit.json

VLEAPP's seekers do not descend into nested archives, so pointing the tool at a .iVa
matches nothing and produces an empty report. This script lifts DCASourceFilesUpload.zip
out, verifies it against the SHA-256 iVe recorded for it, and leaves a zip that can be
passed straight to VLEAPP with -t zip.
VLEAPP reads a .iVa directly with -t iva, so this script is no longer the only
route. It remains the way to KEEP the intermediate zip: -t iva unpacks to a
temporary directory and removes it after the run, so on a large export the unwrap
cost is paid again on every re-run, while this script pays it once. It also
verifies the lifted zip against the SHA-256 iVe recorded for it, which the seeker
route leaves to the zip member CRCs.

python3 admin/scripts/unwrap_berla_iva.py CASE.iVa -o outdir
python3 vleapp.py -t zip -i outdir/DCASourceFilesUpload.zip -o reports
Expand Down
26 changes: 21 additions & 5 deletions admin/scripts/validate_sample_data.py
Original file line number Diff line number Diff line change
Expand Up @@ -261,14 +261,25 @@ def check_registry(artifacts, registry, registry_path, verify_hashes, report):
f"artifacts cite {len(cited)}")


def input_type_for(source):
RUNNABLE_TYPES = ("fs", "tar", "zip", "gz", "file", "raw", "iva")


def input_type_for(source, declared=None):
"""Map an evidence source to the tool's -t input type.

The registry's match key is spelled "zip" for historical reasons but may
point at any container the tool accepts, and --emit may also hand this an
extraction directory. Returns None when nothing names a known type, so the
caller reports it instead of guessing.

declared is an entry's optional "input_type" field and wins outright. It
exists because an extension cannot decide the raw case: the registry holds
.bin files that are readable disk images and .bin files that are chip-level
LUN dumps no walker opens, so raw is opt-in per entry rather than guessed.
.iVa is mapped by extension because it is unambiguous.
"""
if declared:
return declared if declared in RUNNABLE_TYPES else None
if source.is_dir():
return "fs"
name = source.name.lower()
Expand All @@ -278,6 +289,8 @@ def input_type_for(source):
return "tar"
if name.endswith(".zip"):
return "zip"
if name.endswith(".iva"):
return "iva"
return None


Expand Down Expand Up @@ -307,7 +320,7 @@ def lava_output_predicate(report):
return None


def execute_tool(source, label, log_name, report, keep=False):
def execute_tool(source, label, log_name, report, keep=False, declared_type=None):
"""Parse one evidence source and return its per-artifact LAVA row counts.

Returns (produced, output_root, log_path); produced is None when the run
Expand All @@ -318,10 +331,12 @@ def execute_tool(source, label, log_name, report, keep=False):
deleted unless keep is set; the run log is written beside it and survives,
because an artifact that produced nothing has usually logged why.
"""
input_type = input_type_for(source)
input_type = input_type_for(source, declared_type)
if input_type is None:
report.error(f"{label}: cannot pick a -t input type for {source.name}; "
"known inputs are a directory, .zip, .tar, .tar.gz, .tgz")
"known inputs are a directory, .zip, .tar, .tar.gz, .tgz, "
".iVa, or an entry-level \"input_type\" field (a raw disk "
"image is never guessed from its extension)")
return None, None, None

output_root = tempfile.mkdtemp(prefix=RUN_PREFIX)
Expand Down Expand Up @@ -412,7 +427,8 @@ def run_corpus(registry, registry_path, corpus, report, keep=False):
report.error(f"--run '{corpus}' points at a missing file: {source}")
return

produced, _, _ = execute_tool(source, f"--run '{corpus}'", corpus, report, keep=keep)
produced, _, _ = execute_tool(source, f"--run '{corpus}'", corpus, report, keep=keep,
declared_type=entry.get("input_type"))
if produced is None:
return

Expand Down
22 changes: 11 additions & 11 deletions scripts/artifacts/berla_ive_export.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@
Vehicle.json sits uncompressed at the top of the export, so this artifact fires on that
and reports what the export says it holds, including iVe's own per-acquisition counts.
That turns the empty run into a record of what is present and names the step that
reaches it: admin/scripts/unwrap_berla_iva.py.
reaches it: -t iva, or admin/scripts/unwrap_berla_iva.py to keep the intermediate zip.
"""

import json
Expand All @@ -29,11 +29,11 @@
"category": "Vehicle Acquisition",
"notes": "From Vehicle.json at the top of a Berla iVe .iVa export. The .iVa is a "
"ZIP holding another ZIP, and the seekers do not descend into nested "
"archives, so running VLEAPP against a .iVa directly reaches only this "
"file and the vehicle's own data is not seen. Unwrap it first with "
"admin/scripts/unwrap_berla_iva.py, which lifts DCASourceFilesUpload.zip "
"out and verifies it against the SHA-256 the export records, then run "
"VLEAPP against that zip. The counts in these rows are what iVe reported "
"archives, so with any input type other than iva a .iVa reaches only "
"this file and the vehicle's own data is not seen. Run the export "
"with -t iva, which reaches through to the raw image inside; "
"admin/scripts/unwrap_berla_iva.py remains the way to keep the "
"intermediate zip for cheap re-runs. The counts in these rows are what iVe reported "
"for its own parse; they are not produced by VLEAPP and this artifact does "
"not verify them. iVe's parsed database, AcquireDB.ive, is encrypted and "
"is not read. A row here records that an acquisition was attempted and "
Expand All @@ -49,11 +49,11 @@
"and VLEAPP reads only the extracted files. On the tested export those "
"filesystems are QNX6, which no filesystem type Sleuth Kit supports can "
"walk, so the raw image is not reachable with that tooling. It is "
"reachable with qnxprobe, which reads QNX6 superblocks directly and "
"writes the logical files to a zip; on the tested export that route "
"produced the same rows from the same bytes, and it also surfaced a "
"fourth QNX6 volume that the export did not carry extracted files "
"for.",
"reachable with the vendored qnxprobe, which is what -t iva uses: it "
"reaches through the export to the raw image and reads its QNX6 "
"volumes directly. On the tested export that route produced the same "
"rows as the vendor's own extracted file set and also surfaced a "
"fourth QNX6 volume the export carried no extracted files for.",
"paths": ('*/Vehicle.json',),
"sample_data": {
"adams_ford_syncgen3_iva": "Berla iVe export, Ford Sync Gen3 | 4 rows",
Expand Down
24 changes: 11 additions & 13 deletions scripts/artifacts/ford_sync_bluetooth.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,19 +30,17 @@
"unit downloaded over Bluetooth; it does not establish that any number was "
"dialled or that the handset owner was present. TelType is not surfaced "
"because nothing available here documents its values. The store carries no "
"write-ahead log or journal on the tested unit. Where the input came "
"from a Berla iVe export, these rows are read from the file set iVe "
"extracted from the head unit's QNX6 volumes, not from the raw image "
"the export also carries. The values themselves are the unit's own "
"rather than iVe's parse of them. That route is not the only one: "
"these same rows were reproduced from the raw image directly, by "
"extracting its QNX6 volumes with qnxprobe and running this module "
"against that output. Both paths gave 507 contacts, 126 calls and 2 "
"paired devices, and the two stores were byte-identical by SHA-256, so "
"neither extraction is a bottleneck for what these artifacts report.",
"write-ahead log or journal on the tested unit. A Berla iVe export run "
"with -t iva reads these rows from the raw image the export carries, "
"through the head unit's own QNX6 volumes, so the values are the "
"unit's rather than iVe's parse of them. The same rows were also "
"reproduced from the file set iVe itself extracted: both routes gave "
"507 contacts, 126 calls and 2 paired devices, and the two stores "
"were byte-identical by SHA-256, so neither extraction is a "
"bottleneck for what these artifacts report.",
"paths": ('*/BT/btpbk*',),
"sample_data": {
"adams_ford_syncgen3": "Ford Sync Gen3 | 507 rows",
"adams_ford_syncgen3_iva": "Ford Sync Gen3, via -t iva | 507 rows",
"ford_syncg4_logical": "Ford Sync G4 | 0 rows, BT/btpbk not present",
},
"output_types": "standard",
Expand Down Expand Up @@ -75,7 +73,7 @@
"not establish who used the handset or that the vehicle was moving.",
"paths": ('*/BT/btpbk*',),
"sample_data": {
"adams_ford_syncgen3": "Ford Sync Gen3 | 126 rows",
"adams_ford_syncgen3_iva": "Ford Sync Gen3, via -t iva | 126 rows",
"ford_syncg4_logical": "Ford Sync G4 | 0 rows, BT/btpbk not present",
},
"output_types": "standard",
Expand All @@ -102,7 +100,7 @@
"of device, vendor id and product id are reported as stored.",
"paths": ('*/BT/btpersist*',),
"sample_data": {
"adams_ford_syncgen3": "Ford Sync Gen3 | 2 rows",
"adams_ford_syncgen3_iva": "Ford Sync Gen3, via -t iva | 2 rows",
"ford_syncg4_logical": "Ford Sync G4 | 0 rows, BT/btpersist not present",
},
"output_types": "standard",
Expand Down
184 changes: 144 additions & 40 deletions scripts/search_files.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,7 @@

from pathlib import Path
from scripts.ilapfuncs import *
from shutil import copy2
from shutil import copy2, copyfileobj
from zipfile import ZipFile
from fnmatch import _compile_pattern
from functools import lru_cache
Expand Down Expand Up @@ -566,6 +566,49 @@ def update(self, event):
self.volumes += 1


def _extract_image_volumes(probe, image_path, staged_zip, exclude=None):
"""Run the vendored reader over a raw image, streaming its output to the log.

-u so the reader's stdout is unbuffered and its report arrives while it
works rather than in one block at the end. --progress puts machine readable
progress on stderr; stderr is merged into stdout here and the two are told
apart by the leading brace, which no report line has, so one stream is read
and there is no second pipe to deadlock on. The report names the partition
table, every filesystem confirmed and what was extracted; that belongs in
the run log, because it is the record of which volumes the rows came from.
"""
cmd = [sys.executable, '-u', probe, '--progress', '--extract', staged_zip]
for text in (exclude or ()):
cmd += ['--exclude', text]
cmd.append(image_path)

logfunc(f'Reading volumes out of {os.path.basename(image_path)} with the '
f'vendored qnxprobe. This is the slow part of a raw image run.')
progress = _RawExtractProgress()
tail = []
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE,
stderr=subprocess.STDOUT, text=True, bufsize=1)
for line in proc.stdout:
line = line.rstrip()
if not line:
continue
if line.lstrip().startswith('{'):
try:
progress.update(json.loads(line))
continue
except ValueError:
pass # not ours after all, fall through and log it
logfunc(line)
tail.append(line)
del tail[:-20]
proc.wait()
if proc.returncode != 0 or not os.path.isfile(staged_zip):
detail = '\n'.join(tail) or 'no output'
raise RuntimeError(
f'qnxprobe could not extract any filesystem from {image_path}. '
f'Exit {proc.returncode}. Last output:\n{detail}')


class FileSeekerRaw(FileSeekerZip):
"""Read a raw disk image by extracting its volumes to a zip first.

Expand Down Expand Up @@ -601,52 +644,113 @@ def __init__(self, image_path, data_folder, exclude=None):
f'the vendored reader is missing: {probe}. Raw image input needs '
'scripts/vendor/qnxprobe.py.')

# -u so the reader's stdout is unbuffered and its report arrives while it
# works rather than in one block at the end. --progress puts machine
# readable progress on stderr; stderr is merged into stdout here and the
# two are told apart by the leading brace, which no report line has, so
# one stream is read and there is no second pipe to deadlock on.
cmd = [sys.executable, '-u', probe, '--progress', '--extract', staged_zip]
for text in (exclude or ()):
cmd += ['--exclude', text]
cmd.append(image_path)

logfunc(f'Reading volumes out of {os.path.basename(image_path)} with the '
f'vendored qnxprobe. This is the slow part of a raw image run.')
progress = _RawExtractProgress()
tail = []
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE,
stderr=subprocess.STDOUT, text=True, bufsize=1)
for line in proc.stdout:
line = line.rstrip()
if not line:
continue
if line.lstrip().startswith('{'):
try:
progress.update(json.loads(line))
continue
except ValueError:
pass # not ours after all, fall through and log it
# The report names the partition table, every filesystem confirmed and
# what was extracted. That belongs in the run log: it is the record of
# which volumes the rows below came from.
logfunc(line)
tail.append(line)
del tail[:-20]
proc.wait()
if proc.returncode != 0 or not os.path.isfile(staged_zip):
detail = '\n'.join(tail) or 'no output'
raise RuntimeError(
f'qnxprobe could not extract any filesystem from {image_path}. '
f'Exit {proc.returncode}. Last output:\n{detail}')

_extract_image_volumes(probe, image_path, staged_zip, exclude)
FileSeekerZip.__init__(self, staged_zip, data_folder)

def cleanup(self):
FileSeekerZip.cleanup(self)
shutil.rmtree(getattr(self, '_stage_dir', ''), ignore_errors=True)



class FileSeekerIva(FileSeekerZip):
"""Read a Berla iVe .iVa export directly.

An .iVa is a ZIP holding another ZIP, which holds the vehicle's source
files: usually a raw disk image of the head unit plus the file set iVe
extracted from it, beside Vehicle.json and iVe's own encrypted database.
The seekers do not descend into nested archives, so before this a .iVa had
to be unwrapped by hand (admin/scripts/unwrap_berla_iva.py) and the report
of a direct run was empty.

This reaches through the nesting itself. When the export carries a raw
image, that image is read with the vendored qnxprobe, which is the more
complete route: on the tested export it reaches a volume the vendor's own
extracted file set does not include. When no raw image is present, the
extracted file set is used as it stands. Either way Vehicle.json is staged
at the root so the export's acquisition record is reported alongside the
vehicle data.

Everything intermediate lands in a temporary directory and cleanup()
removes it. The unwrap script remains the way to KEEP the intermediate
zip, which makes re-runs cheap on a large export.
"""

SOURCE_FILES_MEMBER = 'DCASourceFilesUpload.zip'

def __init__(self, iva_path, data_folder, exclude=None):
self._stage_dir = tempfile.mkdtemp(prefix='vleapp_iva_')
probe = os.path.join(os.path.dirname(os.path.abspath(__file__)),
'vendor', 'qnxprobe.py')
if not os.path.isfile(probe):
raise FileNotFoundError(
f'the vendored reader is missing: {probe}. .iVa input needs '
'scripts/vendor/qnxprobe.py.')

vehicle_json = None
with ZipFile(iva_path) as outer:
names = outer.namelist()
if 'Vehicle.json' in names:
vehicle_json = outer.read('Vehicle.json')
else:
logfunc('This .iVa carries no Vehicle.json, so no acquisition '
'record will be reported for it.')
inner_names = [n for n in names if n.lower().endswith('.zip')]
if not inner_names:
raise RuntimeError(
f'{os.path.basename(iva_path)} holds no inner zip, so it '
'does not look like an iVe export.')
inner_path = os.path.join(self._stage_dir, 'inner.zip')
logfunc(f'Unpacking {os.path.basename(iva_path)}: '
f'{inner_names[0]} ...')
with outer.open(inner_names[0]) as src, open(inner_path, 'wb') as dst:
copyfileobj(src, dst, 16 << 20)

with ZipFile(inner_path) as inner:
if self.SOURCE_FILES_MEMBER not in inner.namelist():
raise RuntimeError(
f'{inner_names[0]} carries no {self.SOURCE_FILES_MEMBER}; '
'this export does not include the vehicle source files.')
source_zip = os.path.join(self._stage_dir, self.SOURCE_FILES_MEMBER)
logfunc(f'Unpacking {self.SOURCE_FILES_MEMBER} ...')
with inner.open(self.SOURCE_FILES_MEMBER) as src, \
open(source_zip, 'wb') as dst:
copyfileobj(src, dst, 16 << 20)
os.remove(inner_path)

with ZipFile(source_zip) as source:
images = [n for n in source.namelist()
if n.startswith('DiskImages/')
and n.lower().endswith(('.img', '.bin', '.dd', '.raw'))]
image_path = None
if images:
image_path = os.path.join(self._stage_dir,
os.path.basename(images[0]))
logfunc(f'The export carries a raw image, {images[0]}; reading '
'the vehicle data from the image itself.')
with source.open(images[0]) as src, open(image_path, 'wb') as dst:
copyfileobj(src, dst, 16 << 20)

staged_zip = os.path.join(self._stage_dir, 'iva_volumes.zip')
if image_path is not None:
_extract_image_volumes(probe, image_path, staged_zip, exclude)
os.remove(image_path)
os.remove(source_zip)
else:
logfunc('The export carries no raw image; using the file set iVe '
'extracted.')
staged_zip = source_zip

if vehicle_json is not None:
with ZipFile(staged_zip, 'a') as add:
add.writestr('Vehicle.json', vehicle_json)

FileSeekerZip.__init__(self, staged_zip, data_folder)

def cleanup(self):
FileSeekerZip.cleanup(self)
shutil.rmtree(getattr(self, '_stage_dir', ''), ignore_errors=True)

class FileSeekerFile(FileSeekerBase):
"""
This is a class that extends FileSeekerBase to facilitate searching for and copying a specific file
Expand Down
Loading
Loading