Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
241 changes: 241 additions & 0 deletions admin/test/scripts/test_html_report_size.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,241 @@
"""The HTML report holds back tables above the row limit and summarises oversized index tabs.

Every row used to go into the artifact's page, and the table script then read them all
back into memory. Measured 2026-09-19 in Chromium on a 16-column messages page: 100,000
rows were usable after 25.8 s with the browser at 1.7 GB, and at 150,000 the DataTables
call failed with "Maximum call stack size exceeded", left the loading spinner in place and
the browser at 9 GB. The index page copied the run log and the processed files list into
two tabs; a 43 MB list of 229,869 paths opened fast but clicking its tab blocked the page
for over a minute at 12 GB. These tests pin the notice written in place of a large table,
the complete TSV and LAVA output behind it, the index entry that lists it, and the
summarised tabs.
"""
import csv
import inspect
import os
import pathlib
import shutil
import sqlite3
import sys
import tempfile
import types
import unittest

REPO_ROOT = pathlib.Path(__file__).resolve().parents[3]
sys.path.insert(0, str(REPO_ROOT))

from scripts import artifact_report, ilapfuncs, lavafuncs, report # pylint: disable=wrong-import-position
from scripts.context import Context # pylint: disable=wrong-import-position

HEADERS = ('Timestamp', 'Value')


def _rows(count):
return [(f'2026-09-19 00:00:{i % 60:02d}', f'value-{i}') for i in range(count)]


class TableRowLimitTests(unittest.TestCase):

def setUp(self):
self.tmpdir = tempfile.mkdtemp()
self.report_folder = os.path.join(self.tmpdir, '_HTML', 'Category')
os.makedirs(self.report_folder)
self.saved_limit = artifact_report.HTML_TABLE_ROW_LIMIT
artifact_report.set_html_row_limit(10)

def tearDown(self):
artifact_report.set_html_row_limit(self.saved_limit)
shutil.rmtree(self.tmpdir, ignore_errors=True)

def _write_page(self, rows, **kwargs):
page = artifact_report.ArtifactHtmlReport('Test Table')
page.start_artifact_report(self.report_folder, 'Test Table', 'a description')
page.add_script()
held_back = page.write_artifact_data_table(HEADERS, _rows(rows), 'a/b.db', **kwargs)
page.end_artifact_report()
with open(os.path.join(self.report_folder, 'Test Table.temphtml'), encoding='utf8') as fh:
return held_back, fh.read()

def test_a_table_over_the_limit_is_replaced_by_a_notice(self):
held_back, html = self._write_page(11)
self.assertTrue(held_back)
self.assertIn('Total number of entries: 11', html)
self.assertIn('This table has 11 rows, above the 10-row limit', html)
self.assertIn('--html_row_limit 0', html)
self.assertNotIn('<td>value-0</td>', html)
self.assertNotIn('<table', html)

def test_a_table_at_the_limit_is_written_in_full(self):
held_back, html = self._write_page(10)
self.assertFalse(held_back)
self.assertEqual(html.count('<td>value-'), 10)

def test_a_limit_of_zero_writes_every_table(self):
artifact_report.set_html_row_limit(0)
held_back, html = self._write_page(11)
self.assertFalse(held_back)
self.assertEqual(html.count('<td>value-'), 11)

def test_the_notice_names_where_the_complete_rows_are(self):
locations = [artifact_report.LAVA_DATABASE_LOCATION, artifact_report.tsv_export_location('Test Table')]
_, html = self._write_page(11, full_data_locations=locations)
self.assertIn('_lava_artifacts.db', html)
self.assertIn('_TSV Exports/Test Table.tsv', html)

def test_a_caller_that_names_no_locations_gets_the_general_notice(self):
_, html = self._write_page(11)
self.assertIn("the report's other outputs", html)

def test_an_explicit_row_limit_overrides_the_module_setting(self):
held_back, _ = self._write_page(11, row_limit=20)
self.assertFalse(held_back)


class ArtifactProcessorHeldBackTests(unittest.TestCase):
"""A decorated artifact above the limit keeps its TSV and LAVA rows and is listed for the index."""

def setUp(self):
self.tmpdir = tempfile.mkdtemp()
base = pathlib.Path(self.tmpdir)
self.report_folder = base / '_HTML' / 'Category'
self.report_folder.mkdir(parents=True)
for folder in ('data', 'media', '_HTML/media'):
(base / folder).mkdir(parents=True, exist_ok=True)
lavafuncs.initialize_lava(self.tmpdir, self.tmpdir, 'fs')
Context.set_output_params(types.SimpleNamespace(
media_folder=str(base / 'media'), html_media_folder=str(base / '_HTML' / 'media'),
data_folder=str(base / 'data')))
self.saved_limit = artifact_report.HTML_TABLE_ROW_LIMIT
artifact_report.set_html_row_limit(5)
del ilapfuncs.html_tables_held_back[:]

def tearDown(self):
artifact_report.set_html_row_limit(self.saved_limit)
del ilapfuncs.html_tables_held_back[:]
if lavafuncs.lava_db is not None:
try:
lavafuncs.lava_db.close()
except sqlite3.ProgrammingError:
pass
lavafuncs.lava_db = None
lavafuncs.lava_data = None
Context.clear()
Context.set_output_params(None)
shutil.rmtree(self.tmpdir, ignore_errors=True)

def _run_artifact(self, rows):
namespace = {
'__name__': 'fake_big_table',
'__artifacts_v2__': {'big_table': {
'name': 'Big Table', 'category': 'Category', 'description': 'rows',
'output_types': ['html', 'tsv', 'lava']}},
'_rows': _rows, 'HEADERS': HEADERS,
}
exec(f'def big_table(context):\n return HEADERS, _rows({rows}), "a/b.db"\n', namespace) # pylint: disable=exec-used
wrapped = ilapfuncs.artifact_processor(namespace['big_table'])
seeker = types.SimpleNamespace(file_infos={})
# The wrapper takes (files_found, report_folder, seeker, wrap_text) and, in the
# cores that have one, a timezone offset; hand it as many as it declares.
arguments = [[str(pathlib.Path(self.tmpdir) / 'data' / 'a' / 'b.db')], str(self.report_folder), seeker, False, 'UTC']
declared = len(inspect.signature(wrapped, follow_wrapped=False).parameters)
wrapped(*arguments[:declared])

def test_an_artifact_over_the_limit_keeps_its_tsv_and_lava_rows(self):
self._run_artifact(6)
with open(self.report_folder / 'Big Table.temphtml', encoding='utf8') as fh:
html = fh.read()
self.assertIn('This table has 6 rows, above the 5-row limit', html)
self.assertIn('_TSV Exports/Big Table.tsv', html)
self.assertNotIn('<td>value-0</td>', html)
with open(pathlib.Path(self.tmpdir) / '_TSV Exports' / 'Big Table.tsv', encoding='utf-8-sig') as fh:
tsv_rows = list(csv.reader(fh, delimiter='\t'))
self.assertEqual(len(tsv_rows), 7) # header plus six rows
tables = [r[0] for r in lavafuncs.lava_db.execute(
"SELECT name FROM sqlite_master WHERE type = 'table' AND name NOT LIKE '\\_%' ESCAPE '\\'")]
self.assertEqual(len(tables), 1, tables)
lava_rows = lavafuncs.lava_db.execute(f'SELECT COUNT(*) FROM "{tables[0]}"').fetchone()[0]
self.assertEqual(lava_rows, 6)
self.assertEqual(ilapfuncs.html_tables_held_back, [
{'category': 'Category', 'artifact_name': 'Big Table', 'page': 'Big_Table.html', 'rows': 6}])

def test_an_artifact_under_the_limit_is_not_listed(self):
self._run_artifact(5)
with open(self.report_folder / 'Big Table.temphtml', encoding='utf8') as fh:
html = fh.read()
self.assertEqual(html.count('<td>value-'), 5)
self.assertEqual(ilapfuncs.html_tables_held_back, [])


class IndexTabTests(unittest.TestCase):

def setUp(self):
self.tmpdir = tempfile.mkdtemp()
self.logs = pathlib.Path(self.tmpdir) / '_HTML' / '_Script_Logs'
self.logs.mkdir(parents=True)
(self.logs / 'DeviceInfo.html').write_text('device info<br>', encoding='utf8')
self.saved_limit = report.INDEX_TAB_EMBED_LIMIT
report.INDEX_TAB_EMBED_LIMIT = 2000
del ilapfuncs.html_tables_held_back[:]

def tearDown(self):
report.INDEX_TAB_EMBED_LIMIT = self.saved_limit
del ilapfuncs.html_tables_held_back[:]
shutil.rmtree(self.tmpdir, ignore_errors=True)

def _index(self, run_log, files_log):
(self.logs / 'Screen_Output.html').write_text(run_log, encoding='utf8')
(self.logs / 'ProcessedFilesLog.html').write_text(files_log, encoding='utf8')
# Cores without a LAVA-only tab take one parameter fewer.
extra = {'lava_only': False} if 'lava_only' in inspect.signature(report.create_index_html).parameters else {}
report.create_index_html(self.tmpdir, 1, '00:00:01', 'zip', '/cases/x.zip',
'<a class="nav-link" href="index.html">Report Home</a>', {}, '', **extra)
with open(pathlib.Path(self.tmpdir) / '_HTML' / 'index.html', encoding='utf8') as fh:
return fh.read()

@staticmethod
def _files_log(path_count):
paths = ''.join(f'<ul><li>private/var/mobile/file-{i:05d}.jpg</li></ul>' for i in range(path_count))
return ('Extraction/Path selected: /cases/x.zip<br><br>'
'<b>For firstArtifact artifact</b>'
f'<ul><li>{path_count} files for regex <i>*/mobile/*.jpg</i> located at:{paths}</li></ul>'
'<ul><li>No file found for regex <i>*/missing/*</i></li></ul>'
'<b>For secondArtifact artifact</b>'
'<ul><li>1 file for regex <i>*/one.db</i> located at:<ul><li>private/one.db</li></ul></li></ul>')

def test_small_logs_are_copied_into_their_tabs(self):
index = self._index('line one<br>\nline two<br>\n', self._files_log(3))
self.assertIn('line two<br>', index)
self.assertIn('file-00002.jpg', index)
self.assertNotIn('_Script_Logs/ProcessedFilesLog.html', index)

def test_a_large_processed_files_list_is_summarised_per_pattern(self):
index = self._index('short<br>\n', self._files_log(500))
self.assertIn('href="_Script_Logs/ProcessedFilesLog.html"', index)
self.assertIn('500 files for regex <i>*/mobile/*.jpg</i> located at:', index)
self.assertIn('file-00000.jpg', index)
self.assertIn(f'file-{report.INDEX_TAB_PATHS_PER_PATTERN - 1:05d}.jpg', index)
self.assertNotIn(f'file-{report.INDEX_TAB_PATHS_PER_PATTERN:05d}.jpg', index)
self.assertIn(f'{500 - report.INDEX_TAB_PATHS_PER_PATTERN} more in the full list', index)
self.assertIn('No file found for regex <i>*/missing/*</i>', index)
self.assertIn('<b>For secondArtifact artifact</b>', index)
self.assertIn('private/one.db', index)

def test_a_large_run_log_keeps_both_ends(self):
lines = ''.join(f'log line {i}<br>\n' for i in range(1000))
index = self._index(lines, self._files_log(1))
self.assertIn('href="_Script_Logs/Screen_Output.html"', index)
self.assertIn('log line 0<br>', index)
self.assertIn('log line 999<br>', index)
self.assertNotIn('log line 500<br>', index)
self.assertIn('lines not shown here', index)

def test_held_back_tables_are_listed_on_the_details_tab(self):
ilapfuncs.html_tables_held_back.append(
{'category': 'Chats', 'artifact_name': 'Big Table', 'page': 'Big_Table.html', 'rows': 123456})
index = self._index('short<br>\n', self._files_log(1))
self.assertIn('<a href="Big_Table.html">Big Table</a> (Chats): 123,456 rows', index)
self.assertIn(f'more than {artifact_report.HTML_TABLE_ROW_LIMIT:,} rows', index)


if __name__ == '__main__':
unittest.main()
55 changes: 54 additions & 1 deletion scripts/artifact_report.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,42 @@
from scripts.html_parts import *
from scripts.version_info import vleapp_version

# Rows above which write_artifact_data_table leaves the table off the page and writes a
# notice pointing at the LAVA database and the TSV export instead. Measured 2026-09-19 on
# a 16-column messages page in Chromium: 25,000 rows were usable 5.9 s after opening,
# 50,000 in 11.3 s and 100,000 in 25.8 s with the browser holding 1.7 GB for the page; at
# 150,000 the table script failed with "Maximum call stack size exceeded" inside jQuery,
# the loading spinner never cleared and the browser held 9 GB. The main script sets this
# from --html_row_limit; 0 turns the limit off.
HTML_TABLE_ROW_LIMIT = 50000

LAVA_DATABASE_LOCATION = ('the LAVA database (<code>_lava_artifacts.db</code> in the report '
'folder, opened with LAVA)')


def set_html_row_limit(limit):
"""Set the row limit for HTML tables; 0 turns it off."""
global HTML_TABLE_ROW_LIMIT # pylint: disable=global-statement
HTML_TABLE_ROW_LIMIT = max(0, int(limit))


def tsv_export_location(tsv_name):
"""Describe an artifact's TSV export for the held-back table notice."""
return f'the TSV export <code>_TSV Exports/{html.escape(tsv_name)}.tsv</code>'


def table_held_back_notice(num_entries, row_limit, full_data_locations=None):
"""The notice written in place of a table that exceeds the row limit."""
if full_data_locations:
where = ' and in '.join(full_data_locations)
else:
where = "the report's other outputs, such as the LAVA database and the TSV export"
return (f'<p class="note note-warning">This table has {num_entries:,} rows, above the '
f'{row_limit:,}-row limit for HTML report pages, so it is not shown here. The '
f'complete rows are in {where}. Run with <code>--html_row_limit 0</code> to write '
f'it anyway.</p>')


class ArtifactHtmlReport:

def __init__(self, artifact_name, artifact_category=''):
Expand Down Expand Up @@ -48,7 +84,9 @@ def write_artifact_data_table(
table_responsive=True,
table_style='',
table_id='dtBasicExample',
html_no_escape=[]
html_no_escape=[],
row_limit=None,
full_data_locations=None
):
''' Writes info about data, then writes the table to html file
Parameters
Expand All @@ -74,6 +112,14 @@ def write_artifact_data_table(
table_id : Specify an identifier string, which will be referenced in javascript

html_no_escape : if html_escape=True, list of columns not to escape

row_limit : Rows above which the table is left off the page and a notice
written instead; None uses HTML_TABLE_ROW_LIMIT, 0 means no limit

full_data_locations : HTML fragments naming where the complete rows are, for
the notice (see LAVA_DATABASE_LOCATION and tsv_export_location)

Returns True when the table was held back, False when it was written.
'''
if (not self.report_file):
raise ValueError('Output report file is closed/unavailable!')
Expand All @@ -90,6 +136,12 @@ def write_artifact_data_table(

self.report_file.write('<br />')

if row_limit is None:
row_limit = HTML_TABLE_ROW_LIMIT
if row_limit and num_entries > row_limit:
self.report_file.write(table_held_back_notice(num_entries, row_limit, full_data_locations))
return True

if table_responsive:
self.report_file.write("<div class='table-responsive'>")

Expand Down Expand Up @@ -121,6 +173,7 @@ def write_artifact_data_table(
self.report_file.write('</table>')
if table_responsive:
self.report_file.write("</div>")
return False

def add_section_heading(self, heading, size='h2'):
heading = html.escape(heading)
Expand Down
40 changes: 33 additions & 7 deletions scripts/ilapfuncs.py
Original file line number Diff line number Diff line change
Expand Up @@ -468,6 +468,24 @@ def get_data_list_with_media(media_header_info, data_list):

_reported_unsafe_report_names = set()

# Tables left off their HTML page for exceeding artifact_report.HTML_TABLE_ROW_LIMIT, listed
# on the index page so an examiner sees them without opening each artifact. Mutated in
# place only: report.py imports the list itself.
html_tables_held_back = []


def record_html_table_held_back(category, artifact_name, safe_artifact_name, rows):
"""Remember a table the HTML report held back, for the index page and the run log."""
html_tables_held_back.append({
'category': category,
'artifact_name': artifact_name,
# report.generate_report names the final page from the .temphtml file this way
'page': safe_artifact_name.replace(' ', '_') + '.html',
'rows': rows,
})
logfunc(f'{artifact_name}: {rows:,} rows, above the {artifact_report.HTML_TABLE_ROW_LIMIT:,}-row '
f'limit for HTML pages; the table is left off the page and stays in the other outputs')

def sanitize_report_name(name, kind='name'):
"""
Replaces path separators in an artifact name or category so it is usable as a file
Expand Down Expand Up @@ -561,8 +579,17 @@ def wrapper(files_found, report_folder, seeker, wrap_text):
report = artifact_report.ArtifactHtmlReport(artifact_name)
report.start_artifact_report(report_folder, safe_artifact_name, description)
report.add_script()
report.write_artifact_data_table(stripped_headers, html_data_list, source_path, html_no_escape=html_columns)
full_data_locations = []
if check_output_types('lava', output_types):
full_data_locations.append(artifact_report.LAVA_DATABASE_LOCATION)
if check_output_types('tsv', output_types):
full_data_locations.append(artifact_report.tsv_export_location(safe_artifact_name))
held_back = report.write_artifact_data_table(stripped_headers, html_data_list, source_path,
html_no_escape=html_columns,
full_data_locations=full_data_locations)
report.end_artifact_report()
if held_back:
record_html_table_held_back(category, artifact_name, safe_artifact_name, len(data_list))

if check_output_types('tsv', output_types):
tsv(report_folder, stripped_headers, txt_data_list if media_header_info else data_list, safe_artifact_name)
Expand Down Expand Up @@ -1168,12 +1195,11 @@ def write_lava_only_log():
lava_log.write(
"""
<p class="note alert-info mb-4">
The artifacts listed below are likely to return too much data to be viewed \
in a Web browser, so they have been stored in the <i>'_lava_artifacts.db'</i> \
SQLite database.<br>
They are not available from the side bar of the HTML report, but they can \
currently be viewed with any SQLite database viewer until we release <b>LAVA</b> \
(LEAPP Artifact Viewer App).<br></p>
The artifacts listed below are declared LAVA only because of the amount of data \
they return, so their rows are written to the <i>'_lava_artifacts.db'</i> \
SQLite database and they have no page in the side bar of the HTML report.<br>
Open the report folder in <b>LAVA</b> (LEAPP Artifact Viewer App) to review them; \
any SQLite database viewer can also read the database.<br></p>
"""
)
for category, artifacts in lava_only_artifacts.items():
Expand Down
Loading
Loading