diff --git a/admin/test/scripts/test_bits_jobs.py b/admin/test/scripts/test_bits_jobs.py new file mode 100644 index 00000000..92914a20 --- /dev/null +++ b/admin/test/scripts/test_bits_jobs.py @@ -0,0 +1,483 @@ +"""Pin the BITS queue reading in scripts/bits_qmgr.py and scripts/artifacts/windowsBitsJobs.py. + +Every record and page below is built by the test from the layouts the reader documents; no +value comes from a real device. Expected values are written out, never read back from the +code. +""" +import fnmatch +import pathlib +import sqlite3 +import struct +import sys +import tempfile +import unittest +import uuid +from datetime import datetime, timezone +from unittest.mock import patch + +REPO_ROOT = pathlib.Path(__file__).resolve().parents[3] +sys.path.insert(0, str(REPO_ROOT)) + +from scripts import bits_qmgr as bq # pylint: disable=wrong-import-position +from scripts.artifacts import windowsBitsJobs as artifact # pylint: disable=wrong-import-position + +# The six job markers, the file marker and the transfer marker, written out rather than taken +# from the reader. +JOB_MARKERS = [bytes.fromhex(value) for value in ( + 'a15609e143afc94292e66f9856eba7f6', '9f95d44c6470f24b84d7476a7e62699f', + 'f11926a93203bf4c9427898818958831', 'c133bcddfb5aaf4db8a12268b39d01ad', + 'd057568f2c013e4ead2cf4a5d7656faf', '5067419457031d46a4cc5dd9990706e4')] +FILE_MARKER = bytes.fromhex('e4cf9e5146d99743b73e268513051ab2') +TRANSFER = bytes.fromhex('36da56776f515a43acac44a248fff34d') +# What sits between a Files record's Id and its blob when the blob is stored in the record. +INLINE = bytes.fromhex('fe0001040001') + +JOB_ID = uuid.UUID('11111111-2222-3333-4444-555555555555') +OLD_JOB_ID = uuid.UUID('12121212-3434-5656-7878-909090909090') +FILE_ID = uuid.UUID('66666666-7777-8888-9999-aaaaaaaaaaaa') +OLD_FILE_ID = uuid.UUID('abababab-cdcd-efef-0101-232323232323') +OWNER = 'S-1-5-21-1111-2222-3333-1001' +# 2026-01-02 03:04:05 UTC, an hour later, and 90 days after that, as FILETIMEs. +CREATED = 134117966450000000 +MODIFIED = 134118002450000000 +EXPIRES = 134195762450000000 +TIMES = (CREATED, MODIFIED, MODIFIED, MODIFIED, EXPIRES) +WHEN = datetime(2026, 1, 2, 3, 4, 5, tzinfo=timezone.utc) +LATER = datetime(2026, 1, 2, 4, 4, 5, tzinfo=timezone.utc) +UNKNOWN_SIZE = 0xFFFFFFFFFFFFFFFF + + +def counted(text): + """A 32-bit UTF-16 code unit count, then the text and its terminating NUL.""" + raw = (text + '\x00').encode('utf-16-le') + return struct.pack('I', 42) + self.pages = { + 10: FakePage(False, [(0, branch(11)), (DELETED, branch(13)), (0, branch(12))]), + 11: FakePage(True, [(0x1, record_leaf(JOB_ID, job())), + (0x1 | DELETED, struct.pack('I', 0) + self.stored[:half]), + (0, struct.pack('I', half) + self.stored[half:]), + ], common=lv_id), + 30: FakePage(True, [(0x1, record_leaf(FILE_ID, file_record())), + (0x1, b'\x00\x00' + b'\x01\x02' + bytes(30)), + (0x1, record_leaf(OLD_FILE_ID, b'xx', column=257))]), + } + self.ese = FakeEse({'Jobs': ({'LongValues': {'lv': lv_catalog(20)}}, 10), + 'Files': ({'LongValues': {}}, 30)}, self.pages) + self.db = object.__new__(bq.QmgrDatabase) + self.db._db = self.ese # pylint: disable=protected-access + + def test_jobs_inline_separated_and_deleted(self): + self.assertEqual(self.db.records('Jobs'), [ + (JOB_ID.bytes_le, job(), False), + (None, None, True), + (OLD_JOB_ID.bytes_le, self.stored, False), + ]) + + def test_branch_flagged_deleted_not_followed(self): + self.db.records('Jobs') + self.assertNotIn(13, self.ese.read) + + def test_other_layouts_kept_for_counting(self): + self.assertEqual(self.db.records('Files'), [ + (FILE_ID.bytes_le, file_record(), False), + (None, None, False), + (OLD_FILE_ID.bytes_le, None, False), + ]) + + def test_long_value_piece_flagged_deleted_not_used(self): + tags = self.pages[20]._tags # pylint: disable=protected-access + tags[3] = (DELETED, tags[3][1]) + self.assertEqual(self.db.records('Jobs')[2], (OLD_JOB_ID.bytes_le, None, False)) + + def test_missing_long_value_piece(self): + del self.pages[20]._tags[3] # pylint: disable=protected-access + self.assertEqual(self.db.records('Jobs')[2], (OLD_JOB_ID.bytes_le, None, False)) + + def test_table_absent(self): + self.assertEqual(self.db.records('Other'), []) + + +class Context: + def __init__(self, root, files): + self.root, self.files = root, files + + def get_files_found(self): + return [str(f) for f in self.files] + + def get_relative_path(self, path): + return pathlib.Path(path).relative_to(self.root).as_posix() + + +class FakeQmgr: # pylint: disable=too-few-public-methods + tables = {} + + def __init__(self, path): + self.path = path + + def records(self, name): + return self.tables[name] + + +class ArtifactTest(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() # pylint: disable=consider-using-with + self.addCleanup(self.tmp.cleanup) + self.root = pathlib.Path(self.tmp.name) + self.logged = [] + for target, value in (('logfunc', self.logged.append), ('QmgrDatabase', FakeQmgr)): + patcher = patch.object(artifact, target, value) + patcher.start() + self.addCleanup(patcher.stop) + self.live_job = job(state=1) + self.old_job = job(name='Old Job', job_id=OLD_JOB_ID, state=4, file_ids=(OLD_FILE_ID,), + errors=(0x80072EE7,)) + FakeQmgr.tables = {'Jobs': [(JOB_ID.bytes_le, self.live_job, False), (None, None, True)], + 'Files': [(FILE_ID.bytes_le, file_record(), False)]} + folder = self.root/'ProgramData'/'Microsoft'/'Network'/'Downloader' + folder.mkdir(parents=True) + self.folder = folder + self.files = [folder/name for name in ('qmgr.db', 'edb.log', 'edbres00001.jrs', + 'qmgr.jfm', 'edb.chk')] + self.files[0].write_bytes(bytes(64) + self.live_job + bytes(8) + inline_file()) + self.files[1].write_bytes( + job(state=0) + bytes(8) + self.old_job + inline_file( + OLD_FILE_ID, remote='https://example.com/old.cab', total=UNKNOWN_SIZE) + + bytes(4) + file_record(remote='https://example.com/other.cab', + transferred=1 << 63) + JOB_MARKERS[1] + bytes(600)) + self.files[2].write_bytes(bytes(4096)) + self.files[3].write_bytes(job(name='Not Searched', job_id=uuid.UUID(int=7))) + self.files[4].write_bytes(bytes(16)) + + def run_both(self, files=None): + context = Context(self.root, files or self.files) + return (artifact.bitsJobs.__wrapped__(context), + artifact.bitsJobFiles.__wrapped__(context)) + + def test_job_rows(self): + (headers, rows, sources), _ = self.run_both() + self.assertEqual(headers[:2], (('Created (UTC)', 'datetime'), ('Modified (UTC)', 'datetime'))) + self.assertEqual([row[2:] for row in rows], [ + ('Font Download', '11111111-2222-3333-4444-555555555555', 'Connecting (1)', + 'Download (0)', 'Normal (2)', OWNER, '', '', '', 3, 1, + '66666666-7777-8888-9999-aaaaaaaaaaaa', '', 'Live', 'qmgr.db', 1), + ('Font Download', '11111111-2222-3333-4444-555555555555', 'Queued (0)', + 'Download (0)', 'Normal (2)', OWNER, '', '', '', 3, 1, + '66666666-7777-8888-9999-aaaaaaaaaaaa', '', 'Recovered', 'edb.log', 1), + ('Old Job', '12121212-3434-5656-7878-909090909090', 'Error (4)', 'Download (0)', + 'Normal (2)', OWNER, '', '', '', 3, 1, 'abababab-cdcd-efef-0101-232323232323', + '0x80072EE7', 'Recovered', 'edb.log', 1), + ]) + self.assertEqual({row[:2] for row in rows}, {(WHEN, LATER)}) + self.assertEqual(sources.splitlines(), [str(self.files[i]) for i in (1, 2, 0)]) + + def test_file_rows(self): + _, (headers, rows, _) = self.run_both() + self.assertEqual(headers[:5], ('Remote Name', 'Local Name', 'Temporary Name', + 'Bytes Transferred', 'Bytes Total')) + local = 'C:\\Users\\tester\\AppData\\Local\\Temp\\setup.exe' + temporary = 'C:\\Users\\tester\\AppData\\Local\\Temp\\BIT1A2B.tmp' + volume = '\\\\?\\Volume{01234567-89ab-cdef-0123-456789abcdef}\\' + self.assertEqual(rows, [ + ('https://example.com/old.cab', local, temporary, 1024, '', + 'abababab-cdcd-efef-0101-232323232323', '12121212-3434-5656-7878-909090909090', + 'Old Job', 'C:\\', volume, 'Recovered', 'edb.log', 1), + ('https://example.com/other.cab', local, temporary, '9223372036854775808', 4096, + '', '', '', 'C:\\', volume, 'Recovered', 'edb.log', 1), + ('https://example.com/setup.exe', local, temporary, 1024, 4096, + '66666666-7777-8888-9999-aaaaaaaaaaaa', '11111111-2222-3333-4444-555555555555', + 'Font Download', 'C:\\', volume, 'Live', 'qmgr.db', 1), + ]) + + def test_rows_fit_sqlite(self): + (jobs, files) = self.run_both() + with sqlite3.connect(':memory:') as db: + for (headers, rows, _) in (jobs, files): + db.execute(f'create table t{len(headers)} ({", ".join(f"c{i}" for i in range(len(headers)))})') + db.executemany(f'insert into t{len(headers)} values ({", ".join("?" * len(headers))})', + [tuple(str(v) if isinstance(v, datetime) else v for v in row) + for row in rows]) + + def test_run_log(self): + self.run_both() + prefix = 'BITS Jobs: ProgramData/Microsoft/Network/Downloader/' + self.assertEqual([line for line in self.logged if line.startswith('BITS Jobs:')], [ + prefix + 'qmgr.db: Jobs table records: 1 live, 1 flagged deleted, 0 not read', + prefix + 'qmgr.db: Files table records: 1 live, 0 flagged deleted, 0 not read', + prefix + 'edb.log: 2 job and 2 file records found', + prefix + 'qmgr.db: 1 job and 1 file records found', + ]) + + def test_tables_unreadable(self): + def fail(path): + raise Exception(f'bad page in {pathlib.Path(path).name}') # pylint: disable=broad-exception-raised + with patch.object(artifact, 'QmgrDatabase', fail): + (_, rows, _), _ = self.run_both() + self.assertEqual([row[15] for row in rows], ['Recovered'] * 3) + self.assertIn('BITS Jobs: could not read the tables of ' + 'ProgramData/Microsoft/Network/Downloader/qmgr.db: bad page in qmgr.db', + self.logged) + + def test_two_downloader_folders(self): + other = self.root/'Windows.old'/'ProgramData'/'Microsoft'/'Network'/'Downloader' + other.mkdir(parents=True) + (other/'edb.log').write_bytes(job(name='Older Windows', job_id=uuid.UUID(int=9))) + (_, rows, _), _ = self.run_both(self.files + [other/'edb.log']) + self.assertEqual({(row[2], row[16]) for row in rows}, { + ('Font Download', 'ProgramData/Microsoft/Network/Downloader/edb.log'), + ('Font Download', 'ProgramData/Microsoft/Network/Downloader/qmgr.db'), + ('Old Job', 'ProgramData/Microsoft/Network/Downloader/edb.log'), + ('Older Windows', 'Windows.old/ProgramData/Microsoft/Network/Downloader/edb.log'), + }) + + def test_copies_counted_across_files(self): + self.files[2].write_bytes(self.old_job) + (_, rows, _), _ = self.run_both() + old = [row for row in rows if row[2] == 'Old Job'] + self.assertEqual([(row[16], row[17]) for row in old], [('edb.log\nedbres00001.jrs', 2)]) + + def test_declared_paths(self): + for key in ('bitsJobs', 'bitsJobFiles'): + patterns = artifact.__artifacts_v2__[key]['paths'] + for name in ('qmgr.db', 'edb00001.log', 'edbres00002.jrs'): + path = f'x/ProgramData/Microsoft/Network/Downloader/{name}' + self.assertTrue(any(fnmatch.fnmatch(path, p) for p in patterns), (key, name)) + + +if __name__ == '__main__': + unittest.main() diff --git a/scripts/artifacts/windowsBitsJobs.py b/scripts/artifacts/windowsBitsJobs.py new file mode 100644 index 00000000..a633f6ad --- /dev/null +++ b/scripts/artifacts/windowsBitsJobs.py @@ -0,0 +1,394 @@ +"""Windows BITS job store (qmgr.db) for DLEAPP. + +Author: @AlexisBrignoni, Claude. + +Reads the Background Intelligent Transfer Service queue in +ProgramData\\Microsoft\\Network\\Downloader with scripts/bits_qmgr.py: the live records of +qmgr.db's Jobs and Files tables, and the job and file records still present in the +database file and its ESE log files, one row per distinct version of each. +""" + +import os +from collections import OrderedDict + +from scripts.bits_qmgr import QmgrDatabase, filetime, find_records, guid_text, parse_file, parse_job +from scripts.ilapfuncs import artifact_processor, logfunc + +__artifacts_v2__ = { + "bitsJobs": { + "name": "BITS Jobs", + "description": 'Jobs stored in the Windows BITS queue database (qmgr.db) and its ESE ' + 'log files, one row per distinct stored version.', + "author": "@AlexisBrignoni, Claude", + "creation_date": "2026-09-27", + "last_update_date": "2026-09-27", + "requirements": "none (vendored ESE reader)", + "category": "Windows", + "notes": 'Reads the Background Intelligent Transfer Service queue in ' + 'ProgramData\\Microsoft\\Network\\Downloader: qmgr.db, an ESE database, and the ' + '.log and .jrs files in the same folder, with scripts/bits_qmgr.py. Only this' + " ESE form is read; the qmgr0.dat and qmgr1.dat queue files ANSSI's README " + 'describes ' + '(https://github.com/ANSSI-FR/bits_parser/blob/bd3c79b0ccc9191ecc8209e9f0b836a2b16e0357/README.rst?plain=1#L32-L35)' + " are not. Rows come from two places. Live rows are the records of qmgr.db's " + "Jobs table. A record carrying ESE's deleted node flag fNDDeleted " + '(https://github.com/microsoft/Extensible-Storage-Engine/blob/7030fe7407615160e54d152e4ef704eede2fdd7e/dev/ese/src/inc/node.hxx#L248),' + " which ESE's own code treats as not there unless its version store still " + 'holds an update to it ' + '(https://github.com/microsoft/Extensible-Storage-Engine/blob/7030fe7407615160e54d152e4ef704eede2fdd7e/dev/ese/src/ese/node.cxx#L1049-L1079),' + ' is not a live row; it is counted in the run log instead. Recovered rows are' + ' job records found by searching qmgr.db and every .db, .log and .jrs file in' + ' the folder for the job marker GUIDs BitsParser lists ' + '(https://github.com/fireeye/BitsParser/blob/0a2b51eeec79c5e181d8fc526d5715552299c45f/BitsParser.py#L41-L47).' + ' They include jobs no longer in the table, and on AF-Case2 they included 15 ' + 'versions of the 2 live jobs, each with an earlier Modified time than the ' + 'live record. A row is one distinct version of a job record: copies whose ' + 'read fields are all the same are merged, Found In names the files they were ' + "found in and Copies counts the copies the search found, so a Live row's " + "table record is not counted. The record layout is read from ANSSI's " + 'MIT-licensed bits_parser, bits/structs.py at commit ' + 'bd3c79b0ccc9191ecc8209e9f0b836a2b16e0357 ' + '(https://github.com/ANSSI-FR/bits_parser/blob/bd3c79b0ccc9191ecc8209e9f0b836a2b16e0357/bits/structs.py),' + ' cited below as structs.py; no code is copied. A job record holds the type, ' + 'priority and state (L79-L97), a 32-bit value not reported (L98), the job ' + 'GUID (L99), the name, the description and two strings structs.py calls cmd ' + 'and args (L120-L123), the owner SID (L104), a 32-bit value structs.py reads ' + 'as the notify flags (L105-L114), a block it calls the access token that runs' + ' to the transfer marker GUID (L125, and BitsParser.py L36, ' + 'https://github.com/fireeye/BitsParser/blob/0a2b51eeec79c5e181d8fc526d5715552299c45f/BitsParser.py#L36),' + ' the file GUIDs, the errors and five FILETIMEs (L161-L183). State, Type and ' + 'Priority are named as Microsoft lists BG_JOB_STATE ' + '(https://github.com/MicrosoftDocs/sdk-api/blob/a4fd3f7efe2e3378a96c6fe5a6a9455eba9fa021/sdk-api-src/content/bits/ne-bits-bg_job_state.md?plain=1#L56-L97),' + ' BG_JOB_TYPE ' + '(https://github.com/MicrosoftDocs/sdk-api/blob/a4fd3f7efe2e3378a96c6fe5a6a9455eba9fa021/sdk-api-src/content/bits/ne-bits-bg_job_type.md?plain=1#L56-L70)' + ' and BG_JOB_PRIORITY ' + '(https://github.com/MicrosoftDocs/sdk-api/blob/a4fd3f7efe2e3378a96c6fe5a6a9455eba9fa021/sdk-api-src/content/bits/ne-bits-bg_job_priority.md?plain=1#L56-L70),' + ' numbered from 0 in that order as structs.py numbers them (L79-L97), with ' + 'the stored number in brackets. A match is not reported when its state, type ' + 'or priority is outside those lists, its job GUID is zero, its owner is not a' + ' SID, one of its strings is not terminated text, its file GUIDs are not ' + "followed by the transfer marker again, or another record's marker comes " + 'before its transfer marker (the rest of such a match would be read from ' + "another record's bytes). Job Name equals the jobTitle of event 3 (job " + 'created) in the Microsoft-Windows-Bits-Client/Operational log on all 342 ' + 'rows whose job has such an event on the four images tested. Owner SID: 7 of ' + 'those rows hold a well-known service SID, and event 3 names the same account' + ' on all 7; the other 335 hold account SIDs, which the event gives by name ' + 'and which were not compared. Created (UTC) and Modified (UTC) are the first ' + 'two FILETIMEs, which structs.py names ctime and mtime (L167-L168). Created ' + "fell within a second of the job's event 3 on all 342 rows. Modified was not " + 'earlier than Created on any row, and on 323 of the 342 it fell within a ' + "second of one of that job's BITS-Client events. The other three FILETIMEs " + 'are not reported; on every row tested the fifth was the fourth plus exactly ' + '90 days. structs.py gives an error entry as 25 bytes (L151-L158). On the ' + 'Windows 10 1709 image tested (build 16299) only 25-byte entries left the ' + 'times in range, and on the builds 17763, 19041 and 22621 tested only 21-byte' + ' ones did, so both sizes are tried and the one that leaves the first two ' + 'FILETIMEs between 1970 and 2100 is kept. When both sizes do or neither does,' + ' the times and errors are left blank; no tested row needed that. Error Codes' + ' is the 32-bit value four bytes into each entry, the upper half of the ' + '64-bit value structs.py calls code (L152), in hexadecimal. 83 rows carry ' + 'one; on 48 of the 61 whose job has a BITS-Client event with an hr value, it ' + 'equals one of those values. Notify Program and Notify Parameters are the ' + 'strings structs.py calls cmd and args (L122-L123), which BitsParser reports ' + 'as CommandExecuted and CommandArguments ' + '(https://github.com/fireeye/BitsParser/blob/0a2b51eeec79c5e181d8fc526d5715552299c45f/BitsParser.py#L432-L433).' + ' The columns are named for the program and parameters that SetNotifyCmdLine ' + 'sets, which BITS runs when the job enters the error or transferred state ' + '(https://github.com/MicrosoftDocs/sdk-api/blob/a4fd3f7efe2e3378a96c6fe5a6a9455eba9fa021/sdk-api-src/content/bits1_5/nf-bits1_5-ibackgroundcopyjob2-setnotifycmdline.md?plain=1#L53).' + ' Notify Program and Notify Parameters were empty on every row tested, so ' + 'that the two strings hold those values is not established here, and the two ' + 'columns are filled only in the unit tests. Notify Flags held 3 on every row ' + 'except 3 LoneWolf rows, which held 11. Both values are sums of the values ' + 'Microsoft lists for SetNotifyFlags ' + '(https://github.com/MicrosoftDocs/sdk-api/blob/a4fd3f7efe2e3378a96c6fe5a6a9455eba9fa021/sdk-api-src/content/bits/nf-bits-ibackgroundcopyjob-setnotifyflags.md?plain=1#L68-L102).' + ' On the four images tested, Type was Download (0) on every row. File Count ' + 'was 1 on every row except 3 LoneWolf rows, which held 11. Description was ' + 'empty on every row of AF-Case2 and Szechuan, and filled on 26 of the 142 ' + 'LoneWolf rows and 70 of the 263 PC-MUS-001 rows. Owner SID held one value on' + ' every row of AF-Case2 and on every row of PC-MUS-001. Record Status was ' + 'Recovered on every row of LoneWolf, PC-MUS-001 and Szechuan. On the four ' + 'images tested, the Jobs table held 2 live records on AF-Case2 and none on ' + 'LoneWolf, PC-MUS-001 and Szechuan, each of which held one record flagged ' + 'deleted with no column data left. dissect.esedb 3.18 read the same Jobs and ' + 'Files table records from all four databases, with the same record IDs and ' + "byte-identical blobs. Compared with FireEye's Apache-2.0 BitsParser at " + 'commit 0a2b51eeec79c5e181d8fc526d5715552299c45f carving the same files: ' + 'apart from 4 matches with a zero job GUID, it reported the same job GUIDs ' + 'with the same job names. Every pair of Created and Modified times it ' + 'reported is reported here except 21 copies it reported with no times and 2 ' + 'with times in the year 7462; 22 of those 23 belong to jobs with errors on ' + 'the three images whose error entries are 21 bytes, which BitsParser reads as' + ' 25, the size in structs.py (L151-L158). Of the 4 zero-GUID matches, 3 have ' + "a type, priority or state outside Microsoft's lists and 1 has times " + 'BitsParser read from bytes further on in the file, and BitsParser also ' + 'reported one LoneWolf copy whose owner string stops partway through the SID.' + ' None of those is reported here.', + "paths": ('*/ProgramData/Microsoft/Network/Downloader/*',), + "output_types": ["standard"], + "artifact_icon": "download", + "sample_data": { + "af_case2_win10": "Windows 10 1809 build 17763 | 58 rows", + "lonewolf_win10": "Windows 10 Education build 16299 | 142 rows", + "pc_mus_001_win11": "Windows 11 22H2 build 22621 | 263 rows", + "szechuan_win10": "Windows 10 2004 build 19041 | 26 rows", + "dleapp_macos_bigsur": ("macOS 11.2.1 build 20D74 | 0 rows (no member matches " + "the declared paths)"), + }, + }, + "bitsJobFiles": { + "name": "BITS Job Files", + "description": 'Files of Windows BITS jobs (remote URL, local path, sizes) stored in ' + 'the queue database (qmgr.db) and its ESE log files, one row per ' + 'distinct stored version.', + "author": "@AlexisBrignoni, Claude", + "creation_date": "2026-09-27", + "last_update_date": "2026-09-27", + "requirements": "none (vendored ESE reader)", + "category": "Windows", + "notes": 'The files of the jobs in BITS Jobs, from the same folder and the same two ' + "places: the records of qmgr.db's Files table, and the file records found by " + "searching qmgr.db and the folder's .db, .log and .jrs files for the file " + 'marker GUID BitsParser lists ' + '(https://github.com/fireeye/BitsParser/blob/0a2b51eeec79c5e181d8fc526d5715552299c45f/BitsParser.py#L40).' + ' Rows, Record Status, Found In and Copies work as in BITS Jobs. The layout ' + "is read from ANSSI's bits_parser (bits/structs.py at commit " + 'bd3c79b0ccc9191ecc8209e9f0b836a2b16e0357, ' + 'https://github.com/ANSSI-FR/bits_parser/blob/bd3c79b0ccc9191ecc8209e9f0b836a2b16e0357/bits/structs.py#L131-L148):' + ' the local file name, the remote name and a temporary file name, two 64-bit ' + 'numbers, one byte (not reported), and the drive and volume strings. A match ' + 'is not reported when its local or remote name is empty or one of its strings' + ' is not terminated text. structs.py names the two numbers download_size and ' + 'transfer_size (L132-L133). Which is the count transferred and which the ' + 'total was measured on the four images tested: of the 461 rows whose Remote ' + 'Name appears in events 59, 60 or 61 of the ' + 'Microsoft-Windows-Bits-Client/Operational log, the first number equals a ' + 'bytesTransferred value those events give for that URL on 445, and the second' + ' equals a bytesTotal value on 305 of the 310 where it is known, so they are ' + 'reported as Bytes Transferred and Bytes Total. Either number is left blank ' + 'where the record stores all bits set, the value the BITS headers define as ' + 'BG_SIZE_UNKNOWN (MinGW-w64 bits.h, ' + 'https://github.com/mingw-w64/mingw-w64/blob/4564ee4b5063097bf747af3a3f8270a28adff820/mingw-w64-headers/include/bits.h#L95)' + ' and Microsoft documents as the size when BITS cannot determine it ' + '(https://github.com/MicrosoftDocs/sdk-api/blob/a4fd3f7efe2e3378a96c6fe5a6a9455eba9fa021/sdk-api-src/content/bits/ns-bits-bg_file_progress.md?plain=1#L59-L61);' + ' on the images tested only Bytes Total held it, on 217 of the 650 rows. File' + " ID is the Id of the file's record in the Files table. A live record has " + "one, and so does a copy found with the table record's bytes around its blob;" + ' other copies have none. 181 of the 650 rows carry one. Job ID and Job Name ' + 'name each job whose list of file GUIDs includes the File ID, so a row ' + 'without a File ID is linked to no job; 3 rows with a File ID had no job ' + 'listing it. Remote Name is the URL the record stores. Of the (job, URL) ' + 'pairs in events 59, 60 and 61 whose job has a linked row, the event URL ' + "matched a linked row's Remote Name on 94 and differed on 2, both on " + 'LoneWolf, where the event URL was on another host and carried a ' + 'cms_redirect=yes parameter. On the four images tested, Drive was C:\\ on ' + 'every row except 3 LoneWolf rows, which held \\\\?\\C:\\, and Volume held one ' + 'value on every row of each image. Record Status was Recovered on every row ' + "of LoneWolf, PC-MUS-001 and Szechuan. Compared with FireEye's BitsParser at " + 'commit 0a2b51eeec79c5e181d8fc526d5715552299c45f carving the same files, ' + 'every URL it reported is reported here; 15 URLs are reported here and not by' + ' BitsParser, 13 of which appear in the BITS-Client events.', + "paths": ('*/ProgramData/Microsoft/Network/Downloader/*',), + "output_types": ["standard"], + "artifact_icon": "file-text", + "sample_data": { + "af_case2_win10": "Windows 10 1809 build 17763 | 13 rows", + "lonewolf_win10": "Windows 10 Education build 16299 | 212 rows", + "pc_mus_001_win11": "Windows 11 22H2 build 22621 | 392 rows", + "szechuan_win10": "Windows 10 2004 build 19041 | 33 rows", + "dleapp_macos_bigsur": ("macOS 11.2.1 build 20D74 | 0 rows (no member matches " + "the declared paths)"), + }, + }, +} + +# BG_JOB_STATE, BG_JOB_TYPE and BG_JOB_PRIORITY, in the order Microsoft's bits.h pages give +# them (see the notes). +_STATES = ('Queued', 'Connecting', 'Transferring', 'Suspended', 'Error', 'Transient Error', + 'Transferred', 'Acknowledged', 'Cancelled') +_TYPES = ('Download', 'Upload', 'Upload Reply') +_PRIORITIES = ('Foreground', 'High', 'Normal', 'Low') +_DATABASE = 'qmgr.db' +# BG_SIZE_UNKNOWN, defined as (UINT64)(-1) in bits.h (see the notes). +_SIZE_UNKNOWN = 0xFFFFFFFFFFFFFFFF +_SEARCHED = ('.db', '.log', '.jrs') +_JOB_FIELDS = ('marker', 'type', 'priority', 'state', 'unknown', 'job_id', 'name', 'description', + 'program', 'parameters', 'owner', 'flags') +_FILE_FIELDS = ('local_name', 'remote_name', 'temporary_name', 'size_1', 'size_2', 'drive', + 'volume') + + +def _named(value, names): + return f'{names[value]} ({value})' if value < len(names) else str(value) + + +def _size(value): + """A stored 64-bit size: '' for BG_SIZE_UNKNOWN, and text for any other value SQLite + cannot hold as an integer.""" + if value == _SIZE_UNKNOWN: + return '' + return value if value < 1 << 63 else str(value) + + +def _job_key(job): + return (tuple(job[field] for field in _JOB_FIELDS) + + (tuple(job['file_ids']), tuple(job['times']), tuple(job['errors']))) + + +def _file_key(record): + return tuple(record[field] for field in _FILE_FIELDS) + + +def _read(path): + with open(path, 'rb') as handle: + return handle.read() + + +class _Version: + """One distinct job or file record, with where its copies were found.""" + + def __init__(self, record): + self.record = record + self.live = False + self.record_ids = [] + self.places = [] + self.copies = 0 + + def add(self, place, copy=True): + if place not in self.places: + self.places.append(place) + self.copies += copy + + def add_id(self, record_id): + if record_id and record_id not in self.record_ids: + self.record_ids.append(record_id) + + def status(self): + return 'Live' if self.live else 'Recovered' + + +def _collect(context, label): + """[(job versions, file versions)] for each Downloader folder found.""" + folders = OrderedDict() + for path in sorted({str(f) for f in context.get_files_found()}): + if os.path.isfile(path): + folders.setdefault(os.path.dirname(path), []).append(path) + results = [] + for folder, paths in folders.items(): + jobs, files = OrderedDict(), OrderedDict() + prefix = '' if len(folders) == 1 else context.get_relative_path(folder) + '/' + database = next((p for p in paths if os.path.basename(p).lower() == _DATABASE), None) + if database: + _live(context, label, database, prefix, jobs, files) + for path in paths: + name = os.path.basename(path) + if not name.lower().endswith(_SEARCHED): + continue + found_jobs = found_files = 0 + try: + data = _read(path) + except OSError as exc: + logfunc(f'{label}: could not read {context.get_relative_path(path)}: {exc}') + continue + for _, kind, record in find_records(data): + if kind == 'job': + jobs.setdefault(_job_key(record), _Version(record)).add(prefix + name) + found_jobs += 1 + else: + version = files.setdefault(_file_key(record), _Version(record)) + version.add(prefix + name) + version.add_id(record['file_id']) + found_files += 1 + if found_jobs or found_files: + logfunc(f'{label}: {context.get_relative_path(path)}: {found_jobs:,} job and ' + f'{found_files:,} file records found') + results.append((jobs, files)) + return results + + +def _live(context, label, database, prefix, jobs, files): + """Add the records of the Jobs and Files tables to the versions.""" + relative = context.get_relative_path(database) + try: + db = QmgrDatabase(database) + tables = (('Jobs', db.records('Jobs'), parse_job, _job_key, jobs), + ('Files', db.records('Files'), parse_file, _file_key, files)) + except Exception as exc: # pylint: disable=broad-exception-caught + # The vendored ESE reader raises bare Exception for pages it cannot read. + logfunc(f'{label}: could not read the tables of {relative}: {exc}') + return + for table, records, parse, key, versions in tables: + counts = {'live': 0, 'flagged deleted': 0, 'not read': 0} + for record_id, blob, deleted in records: + if deleted: + # ESE does not show a record flagged deleted (see the notes). Any copy of its + # bytes still in the file is found by the search like any other record. + counts['flagged deleted'] += 1 + continue + record = parse(blob) if blob else None + if record is None: + counts['not read'] += 1 + continue + version = versions.setdefault(key(record), _Version(record)) + version.live = True + counts['live'] += 1 + version.add_id(record_id) + version.add(prefix + os.path.basename(database), copy=False) + logfunc(f'{label}: {relative}: {table} table records: ' + + ', '.join(f'{count:,} {what}' for what, count in counts.items())) + + +@artifact_processor +def bitsJobs(context): + data_headers = (('Created (UTC)', 'datetime'), ('Modified (UTC)', 'datetime'), 'Job Name', + 'Job ID', 'State', 'Type', 'Priority', 'Owner SID', 'Description', + 'Notify Program', 'Notify Parameters', 'Notify Flags', 'File Count', + 'File IDs', 'Error Codes', 'Record Status', 'Found In', 'Copies') + data_list = [] + for jobs, _ in _collect(context, 'BITS Jobs'): + for version in jobs.values(): + job = version.record + times = job['times'] + data_list.append(( + filetime(times[0]) if times else '', filetime(times[1]) if times else '', + job['name'], guid_text(job['job_id']), _named(job['state'], _STATES), + _named(job['type'], _TYPES), _named(job['priority'], _PRIORITIES), job['owner'], + job['description'], job['program'], job['parameters'], job['flags'], + len(job['file_ids']), '\n'.join(guid_text(i) for i in job['file_ids']), + '\n'.join(f'0x{code:08X}' for code in job['errors']), + version.status(), '\n'.join(version.places), version.copies)) + data_list.sort(key=lambda row: (str(row[0]), str(row[1]), row[3])) + return data_headers, data_list, _sources(context) + + +@artifact_processor +def bitsJobFiles(context): + data_headers = ('Remote Name', 'Local Name', 'Temporary Name', 'Bytes Transferred', + 'Bytes Total', 'File ID', 'Job ID', 'Job Name', 'Drive', 'Volume', + 'Record Status', 'Found In', 'Copies') + data_list = [] + for jobs, files in _collect(context, 'BITS Job Files'): + owners = {} + for version in jobs.values(): + for file_id in version.record['file_ids']: + owners.setdefault(file_id, OrderedDict())[ + (guid_text(version.record['job_id']), version.record['name'])] = None + for version in files.values(): + record = version.record + linked = OrderedDict() + for file_id in version.record_ids: + linked.update(owners.get(file_id, {})) + data_list.append(( + record['remote_name'], record['local_name'], record['temporary_name'], + _size(record['size_1']), _size(record['size_2']), + '\n'.join(guid_text(i) for i in version.record_ids), + '\n'.join(job_id for job_id, _ in linked), '\n'.join(name for _, name in linked), + record['drive'], record['volume'], version.status(), + '\n'.join(version.places), version.copies)) + data_list.sort(key=lambda row: (row[0], row[1], row[5])) + return data_headers, data_list, _sources(context) + + +def _sources(context): + return '\n'.join(sorted(str(f) for f in context.get_files_found() + if os.path.isfile(str(f)) and str(f).lower().endswith(_SEARCHED))) diff --git a/scripts/bits_qmgr.py b/scripts/bits_qmgr.py new file mode 100644 index 00000000..558b7bb1 --- /dev/null +++ b/scripts/bits_qmgr.py @@ -0,0 +1,312 @@ +"""Reader for the Windows BITS job store (qmgr.db and its ESE log files), for DLEAPP. + +Author: @AlexisBrignoni, Claude. + +The Background Intelligent Transfer Service keeps its queue in +ProgramData\\Microsoft\\Network\\Downloader\\qmgr.db, an ESE database with a Jobs table +and a Files table, each holding a GUID Id column and a Blob column. The record layouts +inside the blobs are read here from ANSSI's MIT licensed bits_parser, bits/structs.py at +bd3c79b0ccc9191ecc8209e9f0b836a2b16e0357 +(https://github.com/ANSSI-FR/bits_parser/blob/bd3c79b0ccc9191ecc8209e9f0b836a2b16e0357/bits/structs.py), +and FireEye's Apache-2.0 BitsParser at 0a2b51eeec79c5e181d8fc526d5715552299c45f +(https://github.com/fireeye/BitsParser/blob/0a2b51eeec79c5e181d8fc526d5715552299c45f/BitsParser.py), +which applies those structures to qmgr.db. No code is copied from either. + +A job blob begins with one of the job marker GUIDs BitsParser lists (L41-L47), then the job +type, priority and state, a 32-bit value, the job's GUID, the name, the description, the two +strings ANSSI calls cmd and args, and the owner SID, each a 32-bit count of UTF-16 code units +and the text, then a 32-bit value ANSSI reads as the notify flags and a block ANSSI calls the +access token, which runs to the transfer marker GUID (L36). After that marker come a 32-bit +count of the job's file GUIDs, the GUIDs, the marker again, the error count and the errors, +three 32-bit values, three FILETIMEs, 14 further bytes and two more FILETIMEs (structs.py +L78-L126 and L151-L183). ANSSI gives an error entry as 25 bytes; the Windows 10 1709 image +tested stores 25 bytes and the later builds tested store 21, so both sizes are tried and the +one that leaves the first two FILETIMEs in range is kept. When both do, or neither does, the +times and errors are left out. + +A file blob begins with the file marker GUID (L40), then the local file name, the remote +name and the temporary file name as counted UTF-16LE strings, two 64-bit numbers, one byte, +and the drive and volume strings (structs.py L131-L148). + +Records are read from the Jobs and Files tables with ESE's deleted flag. A Blob stored apart +from its record, flagged by the tagged-data flag 0x04, is read from the table's long value +tree, whose entries are keyed by the long value ID (big-endian) alone, holding the total +size, or followed by a big-endian byte offset, holding a piece of the value; BitsParser's ESE +module keys the tree the same way (ese/ese.py at the commit above, L486-L561). Records no +longer in the tables, and older copies, are found by searching the database file and its ESE +logs for the marker GUIDs. +""" + +import re +import struct +import uuid +from datetime import datetime, timedelta, timezone + +from scripts.vendor import impacket_ese + +JOB_MARKERS = tuple(bytes.fromhex(value) for value in ( + 'a15609e143afc94292e66f9856eba7f6', '9f95d44c6470f24b84d7476a7e62699f', + 'f11926a93203bf4c9427898818958831', 'c133bcddfb5aaf4db8a12268b39d01ad', + 'd057568f2c013e4ead2cf4a5d7656faf', '5067419457031d46a4cc5dd9990706e4')) +FILE_MARKER = bytes.fromhex('e4cf9e5146d99743b73e268513051ab2') +TRANSFER_MARKER = bytes.fromhex('36da56776f515a43acac44a248fff34d') +ERROR_SIZES = (21, 25) +_MARKER_PATTERN = re.compile(b'|'.join(re.escape(m) for m in JOB_MARKERS + (FILE_MARKER,))) +_SID = re.compile(r'^S-1-\d+(-\d+)*$') +# Largest value of BG_JOB_TYPE, BG_JOB_PRIORITY and BG_JOB_STATE (see the notes). +_LAST_TYPE, _LAST_PRIORITY, _LAST_STATE = 2, 3, 8 +_TAGGED_SEPARATED = 0x04 +_MAX_STRING = 32768 +_MAX_TOKEN = 65536 +_MAX_ITEMS = 4096 +# FILETIMEs from 1970-01-01 to 2100-01-01: a value outside means the fields around it were +# misread. +_FILETIME_MIN = 116444736000000000 +_FILETIME_MAX = 157469184000000000 +# ESE record layout of both tables: one fixed column (Id, a GUID), no variable columns. +_RECORD_HEADER = b'\x01\x7f' +# In a Files record, what sits between the Id and the blob when the blob is stored inline: +# the fixed-column null bitmap, the tagged-column entry (column 256 at offset 4) and the +# tagged-data flag byte. +_INLINE_FILE_PREFIX = b'\xfe\x00\x01\x04\x00\x01' + + +def filetime(value): + """A FILETIME as a UTC datetime, or '' for zero or an out-of-range value.""" + if not _FILETIME_MIN <= value <= _FILETIME_MAX: + return '' + return datetime(1601, 1, 1, tzinfo=timezone.utc) + timedelta(microseconds=value // 10) + + +def guid_text(raw): + return str(uuid.UUID(bytes_le=bytes(raw))) + + +def _counted_text(data, offset): + """(text, next offset) for a 32-bit UTF-16 code unit count and its text.""" + count, = struct.unpack_from(' _MAX_STRING or end > len(data): + raise ValueError('text length out of range') + text = data[offset + 4:end].decode('utf-16-le') + if count: + if not text.endswith('\x00'): + raise ValueError('text not terminated') + text = text[:-1] + if '\x00' in text or not all(char.isprintable() or char in '\t\r\n' for char in text): + raise ValueError('not text') + return text, end + + +def parse_job(data, offset=0): + """A job record starting at a job marker, as a dict, or None when it does not parse.""" + try: + return _parse_job(data, offset) + except (ValueError, struct.error, UnicodeDecodeError): + return None + + +def _parse_job(data, offset): + if data[offset:offset + 16] not in JOB_MARKERS: + raise ValueError('no job marker') + position = offset + 16 + job_type, priority, state, unknown = struct.unpack_from(' _LAST_TYPE or priority > _LAST_PRIORITY or state > _LAST_STATE: + raise ValueError('type, priority or state out of range') + job = {'marker': data[offset:offset + 16], 'type': job_type, 'priority': priority, + 'state': state, 'unknown': unknown, 'job_id': bytes(data[position:position + 16])} + if not any(job['job_id']): + raise ValueError('no job id') + position += 16 + for name in ('name', 'description', 'program', 'parameters', 'owner'): + job[name], position = _counted_text(data, position) + if not _SID.match(job['owner']): + raise ValueError('owner is not a SID') + job['flags'], = struct.unpack_from(' _MAX_ITEMS or position + 16 * count + 16 > len(data): + raise ValueError('file count out of range') + job['file_ids'] = [bytes(data[position + 16 * i:position + 16 * (i + 1)]) for i in range(count)] + position += 16 * count + if data[position:position + 16] != TRANSFER_MARKER: + raise ValueError('no second transfer marker') + position += 16 + job['times'], job['errors'], job['error_size'], job['end'] = _job_tail(data, position) + return job + + +def _job_tail(data, position): + """(five FILETIMEs, error codes, error entry size, end) after the second transfer marker. + + Times are () and the errors are not read when neither error entry size leaves the first + two FILETIMEs in range. + """ + count, = struct.unpack_from(' _MAX_ITEMS: + return (), [], None, position + fits = [] + for size in ERROR_SIZES: + start = position + 4 + size * count + try: + first = struct.unpack_from('= 22 and prefix == _INLINE_FILE_PREFIX else None) + yield offset, 'file', record + else: + record = parse_job(data, offset) + if record is not None: + yield offset, 'job', record + + +class QmgrDatabase: + """The live Jobs and Files records of one qmgr.db.""" + + def __init__(self, path): + self._db = impacket_ese.ESENT_DB(path) + + def _table(self, name): + """(catalog data, root page) of a table, or (None, None). + + openTable hands back a shared dict, so both values are read at once. + """ + cursor = self._db.openTable(name) + if cursor is None: + return None, None + return cursor['TableData'], cursor['FatherDataPageNumber'] + + def _leaves(self, page_number, seen=None): + """(page, tag flags, tag data) of every leaf tag under a tree's root page. + + Leaf tags flagged deleted are included, and the caller reads the flag; branch tags + flagged deleted are not followed. + """ + seen = set() if seen is None else seen + if page_number in seen: + return + seen.add(page_number) + page = self._db.getPage(page_number) + leaf = page.record['PageFlags'] & impacket_ese.FLAGS_LEAF + for number in page.iterDataTagNums(): + flags, data = page.getTag(number) + if leaf: + yield page, flags, data + elif not flags & impacket_ese.TAG_DEFUNCT: + child = impacket_ese.ESENT_BRANCH_ENTRY(flags, data)['ChildPageNumber'] + yield from self._leaves(child, seen) + + def _long_values(self, table): + """{key: value} of a table's long value tree, without the entries flagged deleted.""" + values = {} + for entry in table['LongValues'].values(): + header = impacket_ese.ESENT_DATA_DEFINITION_HEADER(entry['EntryData']) + root = impacket_ese.ESENT_CATALOG_DATA_DEFINITION_ENTRY( + entry['EntryData'][len(header):])['FatherDataPageNumber'] + for page, flags, data in self._leaves(root): + if flags & impacket_ese.TAG_DEFUNCT: + continue + position, common = 0, b'' + if flags & impacket_ese.TAG_COMMON: + size, = struct.unpack_from('I', len(pieces))) + if not piece: + return None + pieces.extend(piece) + return bytes(pieces[:total]) + + def records(self, name): + """(Id, blob, flagged deleted) of each record in the Jobs or Files table. + + A record flagged deleted carries ESE's fNDDeleted node flag, which ESE reads as not + present (see the notes). The Id is None when the record does not have the two-column + layout both tables use, and the blob is None when it cannot be read; both are kept so + they can be counted. + """ + table, root = self._table(name) + if table is None: + return [] + values = None + out = [] + for _, flags, data in self._leaves(root): + deleted = bool(flags & impacket_ese.TAG_DEFUNCT) + leaf = impacket_ese.ESENT_LEAF_ENTRY(flags, data)['EntryData'] + if leaf[:2] != _RECORD_HEADER or len(leaf) < 26: + out.append((None, None, deleted)) + continue + record_id = bytes(leaf[4:20]) + tagged, = struct.unpack_from('= len(leaf): + out.append((record_id, None, deleted)) + continue + item_flags, value = leaf[start], bytes(leaf[start + 1:]) + if item_flags & _TAGGED_SEPARATED: + values = self._long_values(table) if values is None else values + value = self._long_value(values, value[:4]) if len(value) >= 4 else None + out.append((record_id, value, deleted)) + return out