-
Notifications
You must be signed in to change notification settings - Fork 2
Expand file tree
/
Copy pathPukeko.py
More file actions
262 lines (218 loc) · 8.65 KB
/
Copy pathPukeko.py
File metadata and controls
262 lines (218 loc) · 8.65 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
#!/usr/bin/env python
# -*- coding: utf-8 -*-
import os
import sys
import shutil
import argparse
import glob
# Graceful imports — missing packages are reported at startup by check_dependencies()
try:
import magic
except ImportError:
magic = None
try:
from pyxtxt import xtxt as _xtxt
except ImportError:
_xtxt = None
try:
import whisper
except ImportError:
whisper = None
try:
import colorama
colorama.init()
except ImportError:
colorama = None
# Terminal color definitions
class fg:
BLACK = '\033[30m'
RED = '\033[31m'
GREEN = '\033[32m'
YELLOW = '\033[33m'
BLUE = '\033[34m'
MAGENTA = '\033[35m'
CYAN = '\033[36m'
WHITE = '\033[37m'
RESET = '\033[39m'
class bg:
BLACK = '\033[40m'
RED = '\033[41m'
GREEN = '\033[42m'
YELLOW = '\033[43m'
BLUE = '\033[44m'
MAGENTA = '\033[45m'
CYAN = '\033[46m'
WHITE = '\033[47m'
RESET = '\033[49m'
class style:
BRIGHT = '\033[1m'
DIM = '\033[2m'
NORMAL = '\033[22m'
RESET_ALL = '\033[0m'
HOT_WORDS = [
"password", "passwd", "passphrase", "passkey",
"admin", "administrator", "sysadmin",
"user", "login", "credentials", "auth", "2fa", "otp", "pin",
"key", "S/N", "api_key", "secret_key", "encryption_key", "client_secret",
"secret", "token", "bearer", "oauth",
"hash", "salt",
"ssn", "credit_card",
"confidential", "-----BEGIN",
]
# Handled by Whisper (audio and video)
AUDIO_VIDEO_EXTENSIONS = (
'.mp3', '.wav', '.ogg', '.flac', '.m4a', '.aac', '.wma',
'.mp4', '.avi', '.mkv', '.mov', '.wmv', '.flv', '.webm', '.m4v'
)
# Handled by pyxtxt (documents and images)
DOCUMENT_EXTENSIONS = (
'.csv', '.doc', '.docx', '.eml', '.epub', '.gif', '.htm', '.html',
'.jpeg', '.jpg', '.json', '.log', '.msg', '.odt',
'.pdf', '.png', '.pptx', '.ps', '.psv', '.rtf', '.tff', '.tif',
'.tiff', '.tsv', '.txt', '.xls', '.xlsx'
)
def check_dependencies():
"""Check all required Python packages and system tools. Exit if anything critical is missing."""
errors = []
warnings = []
if magic is None: errors.append(" python-magic → pip install python-magic")
if _xtxt is None: errors.append(" pyxtxt → pip install pyxtxt")
if whisper is None: errors.append(" openai-whisper → pip install openai-whisper")
if colorama is None: errors.append(" colorama → pip install colorama")
# System tools
if shutil.which('tesseract') is None:
warnings.append(" tesseract not found — image OCR (.jpg .png .gif .tif etc.) will be skipped")
if shutil.which('ffmpeg') is None:
warnings.append(" ffmpeg not found — audio/video transcription will not work")
if errors:
print("[!] Missing required Python packages:")
for e in errors:
print(e)
print("\n Install all at once:")
print(" pip install python-magic openai-whisper colorama pyxtxt\n")
sys.exit(1)
if warnings:
print("[!] Warning — missing system tools, some features will be limited:")
for w in warnings:
print(w)
print()
def extract_text(path, whisper_model):
"""Extract text from a file. Returns string or None if unsupported."""
if path.lower().endswith(AUDIO_VIDEO_EXTENSIONS):
if whisper_model is None:
return None
print(fg.MAGENTA, style.BRIGHT, " transcribing...", style.RESET_ALL, end='\r')
result = whisper_model.transcribe(path)
return result['text']
elif path.lower().endswith(DOCUMENT_EXTENSIONS):
if _xtxt is None:
return None
return _xtxt(path)
elif magic is not None and "text" in magic.from_file(path, mime=True):
with open(path, "r", encoding='UTF8') as f:
return f.read()
return None
def find_files(path):
"""Recursively yield all file paths under path."""
if os.path.isfile(path):
yield path
elif os.path.isdir(path):
for entry in glob.glob(os.path.join(path, "*")):
if os.path.isfile(entry):
yield entry
elif os.path.isdir(entry):
yield from find_files(entry)
def check_hotwords(text, filepath):
"""Print lines containing hotwords."""
for line_no, line in enumerate(text.splitlines()):
if any(word.lower() in line.lower() for word in HOT_WORDS):
print("\t", fg.CYAN, style.BRIGHT, line_no, ':', line, style.RESET_ALL)
def main():
parser = argparse.ArgumentParser(
description=(
"Pukeko — tailored wordlist generator for breach assessment.\n"
"Scans a file or directory, extracts every unique word from all supported\n"
"formats, and saves a deduplicated, sorted wordlist. Use the resulting\n"
"wordlist to assess whether passwords or sensitive data were exposed in a leak.\n"
),
epilog=(
"supported formats:\n"
" documents/images : .csv .doc .docx .eml .epub .gif .htm .html .jpeg .jpg\n"
" .json .log .msg .odt .pdf .png .pptx .ps .psv .rtf\n"
" .tff .tif .tiff .tsv .txt .xls .xlsx\n"
" audio/video : .mp3 .wav .ogg .flac .m4a .aac .wma\n"
" .mp4 .avi .mkv .mov .wmv .flv .webm .m4v\n"
" plain text : any file detected as plain text (scripts, configs, etc.)\n"
"\n"
"examples:\n"
" python Pukeko.py -input /leak/dump -output wordlist.txt\n"
" python Pukeko.py -input /leak/dump -output wordlist.txt -model medium\n"
" python Pukeko.py -input /leak/dump -output wordlist.txt -hotwords\n"
" python Pukeko.py -input /leak/dump -output wordlist.txt -min 6 -max 30\n"
" python Pukeko.py -input document.pdf -output wordlist.txt -print\n"
),
formatter_class=argparse.RawDescriptionHelpFormatter
)
parser.add_argument('-input', dest='input', help="path to a file or directory to scan", metavar='PATH')
parser.add_argument('-output', dest='output', help="path to the output wordlist file", metavar='FILE')
parser.add_argument('-print', dest='show', action='store_true', help="print the extracted text of each file to stdout")
parser.add_argument('-hotwords', dest='hotwords', action='store_true', help="highlight lines containing sensitive keywords (passwords, tokens, keys, etc.)")
parser.add_argument('-min', dest='min', default=4, type=int, help="minimum word length to include (default: 4)", metavar='N')
parser.add_argument('-max', dest='max', default=20, type=int, help="maximum word length to include (default: 20)", metavar='N')
parser.add_argument('-model', dest='model', default='small', choices=['tiny', 'base', 'small', 'medium', 'large'],
help="Whisper model for audio/video transcription: tiny/base/small(default)/medium/large")
args = parser.parse_args()
if len(sys.argv) < 2:
parser.print_help()
sys.exit(1)
# Run dependency check before doing anything else
check_dependencies()
output = args.output
# Load Whisper model once
whisper_model = None
if whisper is not None:
print(fg.BLUE, style.BRIGHT, "Loading Whisper model:", args.model, style.RESET_ALL)
whisper_model = whisper.load_model(args.model)
# Load existing wordlist into memory
wordset = set()
if os.path.exists(output):
with open(output, "r", encoding='UTF8') as f:
wordset = set(f.read().split())
print(fg.RED, style.BRIGHT, len(wordset), style.RESET_ALL,
"in", output, "before being", fg.BLUE, style.BRIGHT, "Pukekoed", style.RESET_ALL)
# Process all input files, accumulating words in memory
input_path = os.path.normpath(args.input)
for filepath in find_files(input_path):
try:
text = extract_text(filepath, whisper_model)
if text is None:
continue
if args.show:
print(text)
new_words = set(text.split())
truly_new = new_words - wordset
wordset.update(new_words)
print(fg.YELLOW, style.BRIGHT, "+", len(truly_new), style.RESET_ALL, filepath)
if args.hotwords:
check_hotwords(text, filepath)
# Flush new words to disk immediately so Ctrl+C won't lose them
to_append = [w for w in truly_new if args.min <= len(w) <= args.max]
if to_append:
with open(output, "a", encoding='UTF8') as f:
for word in to_append:
f.write(word + "\n")
except Exception as e:
print(" Could not read the file", filepath, ":", e)
# Re-read, sort, and rewrite for a clean final wordlist
existing = set()
if os.path.exists(output):
with open(output, "r", encoding='UTF8') as f:
existing = set(f.read().split())
filtered = sorted(existing, key=len)
with open(output, "w", encoding='UTF8') as f:
for word in filtered:
f.write(word + "\n")
print(fg.GREEN, style.BRIGHT, len(filtered), style.RESET_ALL,
"in", output, "after being", fg.BLUE, style.BRIGHT, "Pukekoed" + style.RESET_ALL)
if __name__ == '__main__':
main()