-
Notifications
You must be signed in to change notification settings - Fork 43
Expand file tree
/
Copy pathspoonmap.py
More file actions
executable file
·8111 lines (7232 loc) · 381 KB
/
Copy pathspoonmap.py
File metadata and controls
executable file
·8111 lines (7232 loc) · 381 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
# Author: Spoonman (Larry.Spohn@TrustedSec.com)
# QA and Personal Pythonian Consultant: Bandrel (Justin.Bollinger@TrustedSec.com)
import argparse
import bisect
import collections
import contextlib
import datetime
import glob as _glob
import hashlib
import ipaddress
import json
import os
from pathlib import Path
import re
import resource
import secrets
import shutil
import socket
import string
import subprocess
import sys
import tempfile
import termios
import threading
import time
import queue
from queue import Queue
import random
import urllib.error
import urllib.request
import xml.etree.ElementTree as etree
from importlib import metadata
_COLOR_INFO = '\x1b[38;5;51m' # electric cyan — "currently doing X"
_COLOR_PROGRESS = '\x1b[38;5;118m' # neon lime green — completion status / results
_COLOR_RESULT = '\x1b[38;5;226m' # electric yellow — output paths / final summary
_COLOR_ERROR = '\x1b[38;5;198m' # hot pink — errors and warnings
_COLOR_RESET = '\x1b[0m'
# Written over a zero-length -oX file after a *successful* masscan run so that
# "completed, found nothing" is distinguishable on disk from "killed before it
# wrote anything". Parses to a tree with zero <host> elements, so it reads as
# an empty result set everywhere and as usable output to the resume gates.
_EMPTY_RESULT_XML = '<?xml version="1.0"?>\n<nmaprun></nmaprun>\n'
def _raise_fd_limit():
"""Raise RLIMIT_NOFILE to 65535 (or the hard limit, whichever is lower).
Called as preexec_fn in every masscan subprocess so libpcap can open raw
sockets without hitting the default 1024-descriptor soft limit
("accept: Too many open files").
"""
try:
_, hard = resource.getrlimit(resource.RLIMIT_NOFILE)
resource.setrlimit(resource.RLIMIT_NOFILE, (min(65535, hard), hard))
except (ValueError, resource.error):
pass # already at hard limit or no permission; masscan will error on its own
def verify_python_version():
import sys
if sys.version_info[0] == 2:
print('Python 3.6+ is required')
quit(1)
elif sys.version_info[0] == 3 and sys.version_info[1] < 6:
print('Python 3.6+ is required')
quit(1)
def save_terminal_state():
"""Save the current terminal state"""
try:
return termios.tcgetattr(sys.stdin)
except (termios.error, OSError):
return None
def restore_terminal_state(state):
"""Restore terminal state and reset terminal"""
if state:
try:
termios.tcsetattr(sys.stdin, termios.TCSADRAIN, state)
except (termios.error, OSError):
pass
# Always try to reset terminal using stty as a fallback
try:
subprocess.run(['stty', 'sane'], check=False, stderr=subprocess.DEVNULL)
except (OSError, subprocess.SubprocessError):
pass
def _format_eta(seconds):
s = int(seconds)
if s < 60:
return f'~{s} second{"s" if s != 1 else ""}'
m = s // 60
if m < 60:
return f'~{m} minute{"s" if m != 1 else ""}'
h, rem_m = divmod(m, 60)
if rem_m == 0:
return f'~{h} hour{"s" if h != 1 else ""}'
return f'~{h} hour{"s" if h != 1 else ""} {rem_m} minute{"s" if rem_m != 1 else ""}'
def _print_completion_status(label, completed, total, start_time):
pct = '{:.0%}'.format(completed / total)
msg = f'\n{label} Completion Status: {pct}'
remaining = total - completed
if completed >= 2 and remaining > 0:
elapsed = time.time() - start_time
eta = (elapsed / completed) * remaining
msg += f' — ETA: {_format_eta(eta)}'
print(_COLOR_PROGRESS + msg + _COLOR_RESET)
def _count_hosts_in_file(filepath):
"""Return total IP address count for all entries in a target/exclusions file.
Each line may be a bare IP, a CIDR, or a hostname.
IPv4 CIDRs are expanded to their full address count via ipaddress.
Hostnames count as 1. Blank lines and # comments are skipped.
Returns None if the file cannot be opened.
Non-IPv4 entries count as 0, matching _parse_ranges(), which rejects them at
parse time because SpooNMAP scans IPv4 only. ipaddress.ip_network() happily
parses IPv6, so counting it at face value meant one ``::/0`` line contributed
2**128 hosts. This count drives the discovery port-list trim against
INTERNAL_DISCOVERY_STATE_CEILING and _calc_scan_wait(), so a single stray IPv6
line in ranges.txt cut an all-IPv4 scan's port list from 10 ports to 5 and
skewed the inter-scan wait — no crash and no lost output, but it silently
changed what got scanned. _parse_ranges() already names the offending line,
so no second warning is printed here.
"""
count = 0
try:
with open(filepath, 'r') as f:
for line in f:
line = line.strip()
if not line or line.startswith('#'):
continue
try:
net = ipaddress.ip_network(line, strict=False)
except ValueError:
count += 1 # hostname — resolves to one IP
else:
if net.version == 4:
count += net.num_addresses
except OSError:
return None
return count
def _ip_sort_key(value):
"""Total-order sort key for a list of IPv4 address strings that never raises.
The previous inline key, ``tuple(int(o) for o in x.split('.'))``, raises
ValueError for anything that is not four decimal octets — a hostname that
survived resolution, an IPv6 literal that leaked out of a parser, a
truncated line read back from a resume file. Those sorts run *after* a
completed masscan sweep, so a single odd entry discarded the whole sweep
with an opaque traceback. Sorting must not be able to lose scan results.
IPv4 addresses keep their existing numeric ordering; anything unparseable
sorts after them, ordered lexically among itself, and is still written out.
"""
try:
return (0, int(ipaddress.IPv4Address(value)), '')
except ValueError:
return (1, 0, str(value))
def _parse_target_ranges(filepath):
"""Parse IPs/CIDRs/ranges from a masscan-style target or exclude file.
Returns a list of inclusive ``(start_int, end_int)`` IPv4 bounds.
Handles three formats masscan accepts but ipaddress rejects:
- Inline comments: 10.0.0.0/8 # note
- Range notation: 10.0.0.1-10.0.0.254
- Netmask notation: 10.0.0.0 255.255.0.0
SpooNMAP is IPv4-only, so an IPv6 entry is reported by file and line
number and skipped here. ipaddress.ip_network() happily parses IPv6, so
without this check the bounds were stored and only blew up several hundred
lines later, in _build_discovery_target_file()'s summarize_address_range()
call, as an AddressValueError for a value >= 2**32 — and only when an
exclusions file happened to be configured, since the no-exclusions path
returns early. Naming the offending line beats that traceback.
Module level rather than nested inside _build_discovery_target_file()
because mass_scan() needs the same parse to decide whether a cached
live_hosts entry is still inside the operator's current scope. One parser
for "what does this target file actually cover", not two.
"""
ranges = []
try:
with open(filepath) as fh:
for lineno, line in enumerate(fh, 1):
# Strip inline comments before parsing
line = line.split('#')[0].strip()
if not line:
continue
bounds, error = _parse_range_line(line)
if error == 'ipv6':
print(_COLOR_ERROR
+ f'Warning: {filepath} line {lineno}: '
+ f'ignoring non-IPv4 target "{line}" '
+ '— SpooNMAP scans IPv4 only.'
+ _COLOR_RESET)
continue
# An unparseable line is skipped silently, as it always has been:
# masscan accepts forms this parser does not model, and this
# function also runs on the exclusions file.
if bounds is not None:
ranges.append(bounds)
except OSError:
pass
return ranges
def _parse_range_line(line):
"""Parse one masscan-style target line into inclusive IPv4 ``(start, end)``.
Returns a ``(bounds, error)`` pair: *bounds* is the tuple or None, and
*error* is None, ``'ipv6'`` (parsed, but not IPv4), or ``'invalid'``.
Split out of _parse_target_ranges() so that --target validates a value
against exactly the syntax a target *file* accepts. Two parsers would
drift, and the divergence would show up as a CLI target the tool accepts
at the command line and then silently fails to scan.
"""
# Standard CIDR or bare IP
try:
net = ipaddress.ip_network(line, strict=False)
except ValueError:
net = None
if net is not None:
if net.version != 4:
return None, 'ipv6'
return (int(net.network_address), int(net.broadcast_address)), None
# Range notation: A.B.C.D-E.F.G.H
if '-' in line:
parts = line.split('-', 1)
try:
start = int(ipaddress.IPv4Address(parts[0].strip()))
end = int(ipaddress.IPv4Address(parts[1].strip()))
if start <= end:
return (start, end), None
return None, 'invalid'
except ValueError:
pass
# Netmask notation: A.B.C.D M.M.M.M
parts = line.split()
if len(parts) == 2:
try:
net = ipaddress.ip_network(f'{parts[0]}/{parts[1]}', strict=False)
return (int(net.network_address), int(net.broadcast_address)), None
except ValueError:
pass
return None, 'invalid'
def _parse_target_arg(value):
"""Validate a ``--target`` value into a list of target lines.
Accepts a comma-separated list of anything a target file line may hold (bare
IP, CIDR, ``A-B`` range, ``addr netmask``). Raises ValueError naming the
offending token, so a typo is rejected before the scan starts rather than
producing a clean run over an empty target set — a false negative with a
report attached is this tool's worst outcome.
"""
tokens = [tok.strip() for tok in value.split(',')]
tokens = [tok for tok in tokens if tok]
if not tokens:
raise ValueError('--target was given no address')
for tok in tokens:
bounds, error = _parse_range_line(tok)
if error == 'ipv6':
raise ValueError(f'--target "{tok}" is not IPv4 '
'— SpooNMAP scans IPv4 only')
if bounds is None:
raise ValueError(f'--target "{tok}" is not a valid '
'IP, CIDR, range, or address/netmask')
return tokens
# Sentinel distinguishing "--cleanup given with no path" (read output_path
# from config.json) from "--cleanup not given at all" (None). A bare string
# default would collide with a legitimately empty operator-supplied value.
_CLEANUP_NO_PATH = object()
def _build_arg_parser():
"""Construct SpooNMAP's argument parser.
Split out of _parse_args() so the parser itself (help text, registered
flags) can be inspected directly in tests without going through
parse_args()/sys.exit().
"""
parser = argparse.ArgumentParser(
prog='spoonmap',
description='SpooNMAP: orchestrates masscan port discovery followed by '
'nmap service/script scanning for authorized penetration '
'tests. Run with no arguments for the interactive prompts, '
'or place a config.json alongside it to skip them.',
# A half-typed flag must be a usage error, not a guess: without this,
# argparse accepts any unambiguous prefix (--re for --resume, --targ
# for --target, --clean for --cleanup), which silently widens the
# accepted surface beyond what's documented for a tool that fires
# packets at a client's network under a signed scope.
allow_abbrev=False,
)
parser.add_argument(
'--version', '-v', '-V', action='store_true',
help='Print the installed SpooNMAP version and exit. Checked first: '
'if both --version and --check-update are given, --version wins.',
)
parser.add_argument(
'--check-update', action='store_true',
help='Check GitHub for a newer release and exit without scanning. '
'Ignored if --version is also given.',
)
parser.add_argument(
'--resume', action='store_true',
help='Resume a previous scan, reusing cached output that is still valid.',
)
parser.add_argument(
'--cleanup', nargs='?', const=_CLEANUP_NO_PATH, default=None,
metavar='DIR',
help='Remove prior scan output and exit. With no DIR, reads '
'output_path from config.json in the current directory.',
)
parser.add_argument(
'--target', metavar='SPEC',
help='Comma-separated target(s) to scan (IP, CIDR, A-B range, or '
'"address netmask") instead of prompting or using target_file.',
)
return parser
def _parse_args(argv=None):
"""Parse SpooNMAP's command line into a namespace.
Module-level (not inline in main()) so the reject paths stay under test:
main() is `# pragma: no cover`, and "an unrecognized flag stops the run
instead of falling through into a scan" -- along with a malformed
--cleanup/--target -- is exactly the behaviour that must not regress.
argparse itself handles --help (prints and exits 0) and any unknown
argument (prints usage to stderr and exits 2).
"""
return _build_arg_parser().parse_args(argv)
def _resolve_cli_target(value):
"""Return validated ``--target`` tokens from *value*, or None if absent.
*value* is the parsed ``--target`` string (argparse's ``args.target``),
or None if the flag was not given -- argparse itself now rejects a missing
or option-like value (e.g. ``--target --resume``) before this is ever
called. Exits non-zero on a value that fails validation. Lives outside
main() so the reject path is testable: main() is an interactive entry
point excluded from coverage, and "bad --target aborts the run" is
exactly the behaviour that must not regress into a clean scan over an
empty target set.
"""
if value is None:
return None
try:
return _parse_target_arg(value)
except ValueError as exc:
print(_COLOR_ERROR + f'ERROR: {exc}' + _COLOR_RESET)
sys.exit(1)
def _write_cli_target_file(tokens, output_path):
"""Write ``--target`` addresses to a scope file and return its path.
Deliberately does *not* touch ranges.txt. That file is the operator's
engagement scope input and is tracked in-repo as an empty template;
overwriting it from a convenience flag would destroy scope data and leave no
record of what was actually authorised. Written through _atomic_write() so
a partial write can never be read as a complete target list.
"""
path = os.path.join(output_path, 'cli_targets.txt')
_atomic_write(path, ''.join(f'{tok}\n' for tok in tokens))
return path
def _merge_ranges(ranges):
"""Coalesce overlapping/adjacent ``(start, end)`` bounds, sorted by start."""
out = []
for s, e in sorted(ranges):
if out and s <= out[-1][1] + 1:
out[-1] = (out[-1][0], max(out[-1][1], e))
else:
out.append((s, e))
return out
def _ip_in_ranges(value, merged_ranges):
"""Return True if IPv4 string *value* falls inside *merged_ranges*.
*merged_ranges* must come from _merge_ranges(): disjoint and sorted, so a
bisect finds the only range that could contain the address. Never raises —
anything that is not a plain IPv4 address (an unresolved hostname, an IPv6
literal, a truncated line read back from a resume file) reads as out of
range, so it is excluded rather than crashing the caller. This gates what
gets *scanned*, so failing closed is the only safe direction.
"""
try:
addr = int(ipaddress.IPv4Address(value))
except ValueError:
return False
idx = bisect.bisect_right(merged_ranges, (addr, float('inf'))) - 1
if idx < 0:
return False
start, end = merged_ranges[idx]
return start <= addr <= end
# Retained results are disclosed, never deleted; deletion is exclusively
# operator-initiated via the [d]elete prompt or --cleanup flag.
def _report_out_of_scope_retained(port_ips, scope_ranges, target_file,
examples=3):
"""Warn about retained hosts that fall outside the current target scope.
Cached ``live_hosts/portN.txt`` entries are unioned into this run's results
and are deliberately never deleted — losing a completed scan's output is this
tool's worst failure mode, so a narrowed ``ranges.txt`` does not prune them.
But ``all_live_hosts.txt`` and ``spoonmap_output.*`` feed engagement
deliverables, and a host outside the current scope sitting in those files is
a host the operator may report on or pivot to believing it was authorised for
this engagement. Rules-of-engagement violations start exactly there, and it
used to be entirely silent.
So: disclose, never delete, and never filter what gets written. This prints
and returns the offending set; it mutates nothing. Anything that is not a
plain IPv4 address counts as out of scope, because it cannot be shown to be
inside it (see _ip_in_ranges).
An empty *scope_ranges* means there is no parseable authorisation to compare
against, so nothing is claimed either way.
"""
if not scope_ranges:
return set()
stale = {ip
for ips in port_ips.values()
for ip in ips
if not _ip_in_ranges(ip, scope_ranges)}
if not stale:
return set()
shown = sorted(stale, key=_ip_sort_key)[:examples]
remainder = len(stale) - len(shown)
sample = ', '.join(shown) + (f' (+{remainder} more)' if remainder else '')
print(_COLOR_ERROR
+ f'Warning: {len(stale)} retained host(s) are OUTSIDE the current '
f'target scope and were NOT scanned this run: {sample}. These are '
'cached results from an earlier run against a wider scope. '
'They are kept in live_hosts/ and all_live_hosts.txt on purpose — '
'completed scan results are never deleted — but they are not in '
f'{target_file}, and nothing was sent to them this run. Confirm they '
'are in scope for this engagement before reporting on or pivoting to '
'them. To clean them, re-run and select [d]elete at the prompt, '
'or use --cleanup.'
+ _COLOR_RESET)
return stale
def _build_discovery_target_file(target_file, exclusions_file, disc):
"""Pre-subtract exclusions from target ranges and write a masscan-ready file.
Masscan builds its randomization permutation over the full target space
before applying --excludefile, so passing a 3M-IP target with 2.96M
excluded hosts causes it to iterate at ~72 effective pps instead of 1000.
Pre-computing the difference here gives masscan a file containing only the
~230k IPs it will actually probe, restoring full rate efficiency.
Returns (filtered_file_path, accurate_host_count). If no exclusions apply,
returns (target_file, raw_count) unchanged so callers omit --excludefile.
"""
def _subtract(targets, excls):
result = []
ei = 0
for ts, te in targets:
cur = ts
while ei < len(excls) and excls[ei][1] < cur:
ei += 1
j = ei
while j < len(excls) and excls[j][0] <= te:
es, ee = excls[j]
if es > cur:
result.append((cur, es - 1))
cur = max(cur, ee + 1)
if cur > te:
break
j += 1
if cur <= te:
result.append((cur, te))
return result
target_ranges = _parse_target_ranges(target_file)
if not target_ranges:
return target_file, 0
raw_count = sum(e - s + 1 for s, e in target_ranges)
if not exclusions_file or not os.path.exists(exclusions_file):
return target_file, raw_count
excl_ranges = _parse_target_ranges(exclusions_file)
if not excl_ranges:
return target_file, raw_count
remaining = _subtract(_merge_ranges(target_ranges), _merge_ranges(excl_ranges))
if not remaining:
filtered_file = os.path.join(disc, 'discovery_targets_filtered.txt')
open(filtered_file, 'w').close()
return filtered_file, 0
filtered_file = os.path.join(disc, 'discovery_targets_filtered.txt')
count = 0
with open(filtered_file, 'w') as fh:
for start, end in remaining:
for net in ipaddress.summarize_address_range(
ipaddress.IPv4Address(start), ipaddress.IPv4Address(end)):
fh.write(str(net) + '\n')
count += net.num_addresses
return filtered_file, count
def ascii_art(): # pragma: no cover -- cosmetic banner, no branches to verify
print(r'''
________ _____ _______ _________________
__ ___/______________________ | / /__ |/ /__ |__ __ \
_____ \___ __ \ __ \ __ \_ |/ /__ /|_/ /__ /| |_ /_/ /
____/ /__ /_/ / /_/ / /_/ / /| / _ / / / _ ___ | ____/
/____/ _ .___/\____/\____//_/ |_/ /_/ /_/ /_/ |_/_/
/_/
''')
def is_hostname(line):
"""
Determine if a line is a hostname (not an IP address or CIDR range)
Args:
line: The line to check
Returns:
True if the line appears to be a hostname, False if it's an IP/CIDR
"""
line = line.strip()
if not line or line.startswith('#'):
return False
# Check if it's a CIDR notation
if '/' in line:
return False
# Check if it's an IP address (simple regex)
ip_pattern = r'^(\d{1,3}\.){3}\d{1,3}$'
if re.match(ip_pattern, line):
return False
# If it contains letters or is a domain-like string, treat as hostname
return True
def resolve_hostname(hostname):
"""
Resolve a hostname to an IP address
Args:
hostname: The hostname to resolve
Returns:
IP address string, or None if resolution fails
"""
try:
ip = socket.gethostbyname(hostname.strip())
return ip
except (socket.gaierror, socket.herror, OSError) as e:
print(_COLOR_ERROR + f'Warning: Could not resolve hostname {hostname}: {e}' + _COLOR_RESET)
return None
def _extract_ssl_cert_hostnames(ssl_cert_output):
"""Extract hostnames from ssl-cert NSE script output (CN + SAN).
Returns a deduped, order-preserved list: the certificate's commonName
first, then each Subject Alternative Name DNS: entry in the order nmap
printed them. Wildcard names (e.g. '*.example.com') are returned like
any other name -- callers that feed a scan target must filter those out
themselves. Anchored to the 'Subject:' line specifically (not
'Issuer:'), since ssl-cert output carries a commonName for both and only
the subject's identifies the host being scanned.
"""
hostnames = []
seen = set()
cn_match = re.search(r'^Subject:.*?commonName=([^\s/,]+)', ssl_cert_output, re.MULTILINE)
if cn_match:
cn = cn_match.group(1).strip()
if cn and cn not in seen:
hostnames.append(cn)
seen.add(cn)
san_match = re.search(r'^Subject Alternative Name:\s*(.+)$', ssl_cert_output, re.MULTILINE)
if san_match:
for entry in san_match.group(1).split(','):
entry = entry.strip()
if entry.startswith('DNS:'):
name = entry[len('DNS:'):].strip()
if name and name not in seen:
hostnames.append(name)
seen.add(name)
return hostnames
def _write_if_changed(path, content):
"""Write *content* to *path* only if it differs from the current contents.
Preserves the file's mtime when the content is unchanged so mtime-based
resume freshness checks stay valid across re-runs (e.g. an unchanged
resolved_targets.txt must not appear newer than the discovery output it
produced). Returns True if the file was (re)written, False if left as-is.
"""
try:
with open(path) as fh:
if fh.read() == content:
return False
except (OSError, UnicodeDecodeError):
pass # missing/unreadable → (re)write below
with open(path, 'w') as fh:
fh.write(content)
return True
def _atomic_write(path, content):
"""Write *content* to *path* atomically (temp file in the same dir + os.replace).
The temp file is created in the same directory as *path* so os.replace()
never crosses a filesystem boundary and therefore stays atomic: readers
see either the old contents or the new ones, never a half-written file.
A failure part-way through leaves *path* untouched and removes the temp.
"""
directory = os.path.dirname(path) or '.'
fd, tmp_path = tempfile.mkstemp(dir=directory,
prefix='.' + os.path.basename(path) + '.',
suffix='.tmp')
try:
with os.fdopen(fd, 'w') as fh:
fh.write(content)
os.replace(tmp_path, path)
except BaseException:
with contextlib.suppress(OSError):
os.unlink(tmp_path)
raise
def _safe_mtime(path):
"""Return *path*'s mtime, or 0 if it cannot be read.
Closes the exists()/getmtime() race in the resume gates: a file removed by
another process between the two calls must read as stale (0) rather than
raising FileNotFoundError and aborting a scan mid-run.
"""
try:
return os.path.getmtime(path)
except OSError:
return 0
def _safe_size(path):
"""Return *path*'s size in bytes, or 0 if it is not a readable regular file.
The _safe_mtime() companion for size checks: the resume gates and
_parse_result_xml() both test existence and then stat, and a file removed in
between (a parallel --cleanup, an operator tidying up mid-run) must read as
unusable (0) rather than raising FileNotFoundError out of a scan. A
directory reads as 0 too, which keeps the isfile() checks these callers used
to make.
"""
try:
if not os.path.isfile(path):
return 0
return os.path.getsize(path)
except OSError:
return 0
def _target_entries(target_file):
"""Return the normalised set of target lines in *target_file*, or None.
Blank lines, comments and ordering are stripped, so the result tracks the
file's *line set* rather than its layout. Not its address set: `10.0.0.0/24`
and an explicit list of those 254 addresses are different entries here even
though they scan the same hosts. That mismatch only ever errs toward a
redundant re-scan, never toward skipping work.
None means the file could not be read at all. `errors='replace'` keeps an
undecodable byte from raising out of a resume gate, which would abort the run
long before masscan got a chance to reject the file itself.
"""
try:
with open(target_file, 'r', errors='replace') as fh:
return {
entry for entry in (line.strip() for line in fh)
if entry and not entry.startswith('#')
}
except OSError:
return None
def _coverage_record_path(output_file):
"""Sidecar path recording what *output_file* covered.
Deliberately not an .xml name: masscan_results/ is aggregated wholesale by
_aggregate_result_dir(), which lists the directory rather than globbing, and
_parse_result_xml() drops anything not ending in .xml. So this file is
invisible to every result consumer, exactly as portN.xml.failed is.
"""
return output_file + '.coverage'
def _exclusion_entries(exclusions_file):
"""Normalised exclusion set: empty when there is no exclusions file at all.
A falsy *exclusions_file* means "nothing was excluded", which is a real,
checkable state and not the same as an unreadable file (None) — the effective
scan target is `targets - exclusions`, so losing track of the second term
hides exactly as much as losing the first.
"""
if not exclusions_file:
return set()
return _target_entries(exclusions_file)
def _read_coverage_record(output_file):
"""Return ``{'targets': set, 'exclusions': set}`` for *output_file*, or None.
None covers absent, unreadable and malformed alike: all three mean "we cannot
say what this output covered", and the gate treats that as unusable rather
than guessing.
"""
try:
with open(_coverage_record_path(output_file), 'r', errors='replace') as fh:
data = json.load(fh)
return {'targets': set(data['targets']),
'exclusions': set(data['exclusions'])}
except (OSError, ValueError, TypeError, KeyError):
return None
def _discard_coverage_record(output_file):
"""Remove *output_file*'s coverage record, if any.
Called before a phase runs and on every path where a record cannot be written
truthfully. A record left over from an earlier run would be applied to
whatever output sits there now, and if that earlier run covered *more* it
would validate a later, narrower output — so absent (which the gate rejects)
is the only safe state. Failure to remove is ignored: there is nothing
better to do, and the gate still compares the record it finds.
"""
try:
os.remove(_coverage_record_path(output_file))
except OSError:
pass
def _stamp_target_coverage(output_file, target_file, exclusions_file):
"""Record what *output_file* actually covered: its targets and its exclusions.
Call from success paths only, exactly as _EMPTY_RESULT_XML is stamped: a
record written for a killed scan would assert coverage that never happened,
which is worse than having none at all.
Both halves live in **one** file written by a single _atomic_write, not two
files written in sequence. As two files, a KeyboardInterrupt between the
writes — routine during a scan, and a BaseException that an `except
Exception` handler does not catch — left a fresh target list beside a stale
exclusion list, and the gate accepted that pair as though the run had been
exclusion-free. One file makes "both halves describe the same run" a
filesystem guarantee rather than a comment. The lists are stored rather than
a digest of them because the gate compares by subset, not equality; see
_resume_cache_usable().
Neither an unreadable input nor a failed write raises on its own account: the
scan already succeeded, so unwinding here would discard real results, and the
only cost is that the next --resume redoes this phase. Every failure path,
including an interrupt, first deletes any existing record — see
_discard_coverage_record() for why leaving one is worse than having none.
"""
targets = _target_entries(target_file)
exclusions = _exclusion_entries(exclusions_file)
unreadable = target_file if targets is None else (
exclusions_file if exclusions is None else None)
if unreadable is not None:
_discard_coverage_record(output_file)
print(_COLOR_ERROR
+ f'Warning: could not read {os.path.basename(unreadable)} to record '
f'what {os.path.basename(output_file)} covered; a --resume will '
're-run this phase rather than trust it.'
+ _COLOR_RESET)
return
body = json.dumps({'targets': sorted(targets),
'exclusions': sorted(exclusions)}, indent=1) + '\n'
try:
_atomic_write(_coverage_record_path(output_file), body)
except Exception as e:
_discard_coverage_record(output_file)
print(_COLOR_ERROR
+ f'Warning: could not record the target set for '
f'{os.path.basename(output_file)} ({e}); a --resume will re-run '
'this phase rather than trust it.'
+ _COLOR_RESET)
except BaseException:
# An interrupt mid-write leaves _atomic_write's temp file unrenamed, so
# the *previous* record survives — next to output this run just replaced.
# Drop it, then let the interrupt continue unwinding.
_discard_coverage_record(output_file)
raise
def _resume_cache_usable(output_file, baseline_mtime, description, *,
target_file, exclusions_file, is_xml=True):
"""True when cached *output_file* may satisfy a resume gate.
Existence plus freshness is not enough. A scan that was killed, or that
died before writing anything, leaves a zero-length or truncated output file
behind; a gate that only checks existence treats that emptiness as
"already done" forever, silently under-scanning with no warning. So the
file must also hold usable content: XML outputs have to parse (via
_parse_result_xml, which rejects empty and unterminated files), while a
plain-text host list only has to be non-empty.
*baseline_mtime* keeps the separate, still-valid freshness concern: a target
list edited after the cache was written must force the work to be redone.
Pass 0 for gates with no target-file baseline.
*target_file* and *exclusions_file* are what this run would hand to
masscan/nmap as -iL and --excludefile. Both are keyword-only and neither has
a default, so omitting one is a TypeError rather than a silently skipped
check — an earlier optional-keyword form had already cost this at two of the
three call sites that left it out. *exclusions_file* may be None for a phase
that passes no --excludefile; *target_file* may not, because every phase
scans something.
An mtime cannot stand in for either check: it records *when* a file changed,
not *what the cached scan covered*, so switching host discovery off — which
widens the target from the discovery file to the whole range while rewriting
nothing — left every cached phase satisfying its gate and reported a narrow
scan as a complete wide one.
A phase's effective coverage is `targets - exclusions`, so the two halves are
compared in **opposite directions**:
- targets: usable when `current ⊆ stamped` — everything this run would scan
was already covered. Equality would reject a cache that covered strictly
more, and that is the common case, not a corner: the batch phase's target is
rebuilt every run from a probe that is deliberately not resume-gated, and
the iterative probe stops at the first port that finds hosts, so which
cached ports get folded in varies run to run. Under equality an ordinary
difference in probe luck re-scanned every completed batch against a
*narrower* target than the cache it discarded, i.e. --resume stopped
resuming.
- exclusions: usable when `current ⊇ stamped`. Excluding *more* now means a
strictly smaller scan set, which the cache still covers. Excluding
*fewer* — narrowing exclusions.txt because a host was cleared for testing —
brings something previously skipped into scope, and nothing else about the
run changes, so this is the only signal that it must be re-scanned.
Both comparisons are over normalised **lines**, not addresses. `10.0.0.0/24`
and an explicit list of its hosts are different entries even though they scan
the same range, and an exclusion removed from a range the targets never
covered still forces a re-scan. Deliberate: line comparison cannot
over-accept, only over-scan, whereas address-set arithmetic would import the
whole parsing surface (and its failure modes) into a resume gate.
"""
if not os.path.exists(output_file):
return False
if _safe_mtime(output_file) < baseline_mtime:
return False
# Both halves of "what would this run scan" have to be readable before the
# cache can be judged at all; an unreadable input is not evidence the cache
# is fine.
current_targets = _target_entries(target_file)
current_exclusions = _exclusion_entries(exclusions_file)
for label, value, name in (('target', current_targets, target_file),
('exclusions', current_exclusions, exclusions_file)):
if value is None:
print(_COLOR_INFO
+ f're-running {description}: this run\'s {label} file '
f'({os.path.basename(name)}) could not be read, so the '
'cache cannot be trusted' + _COLOR_RESET)
return False
record = _read_coverage_record(output_file)
if record is None:
print(_COLOR_INFO + f're-running {description}: cached result does not '
'record what it covered' + _COLOR_RESET)
return False
uncovered = current_targets - record['targets']
if uncovered:
examples = ', '.join(sorted(uncovered)[:3])
print(_COLOR_INFO
+ f're-running {description}: {len(uncovered)} target(s) in this run '
f'were not covered by the cached result (e.g. {examples})'
+ _COLOR_RESET)
return False
unexcluded = record['exclusions'] - current_exclusions
if unexcluded:
examples = ', '.join(sorted(unexcluded)[:3])
noun = 'entry' if len(unexcluded) == 1 else 'entries'
print(_COLOR_INFO
+ f're-running {description}: {len(unexcluded)} {noun} excluded when '
f'the cache was written {"is" if len(unexcluded) == 1 else "are"} '
f'now in scope (e.g. {examples})'
+ _COLOR_RESET)
return False
if is_xml:
usable = _parse_result_xml(output_file) is not None
else:
usable = _safe_size(output_file) > 0
if not usable:
print(_COLOR_INFO + f're-running {description}: cached result was empty '
'or unreadable' + _COLOR_RESET)
return usable
def preprocess_targets(target_file, output_path):
"""
Preprocess the target file to separate hostnames from IPs.
Creates a resolved-IP target file and a hostname mapping file.
Args:
target_file: Path to the original target file
output_path: Directory for output files
Returns:
Tuple of (resolved_target_file, ip_to_hostname_map)
"""
ip_to_hostname = {}
masscan_targets = []
print(_COLOR_INFO + 'Preprocessing target file...' + _COLOR_RESET)
with open(target_file, 'r') as f:
for line in f:
line = line.strip()
if not line or line.startswith('#'):
continue
if is_hostname(line):
# Resolve hostname to IP
print(f'Resolving hostname: {line}')
ip = resolve_hostname(line)
if ip:
print(f' {line} -> {ip}')
ip_to_hostname[ip] = line
masscan_targets.append(ip)
else:
print(f' Skipping {line} (resolution failed)')
else:
# It's already an IP or CIDR, add as-is
masscan_targets.append(line)
# Write resolved IP list (used by both nmap and masscan paths). Rewritten
# only when the content changed, so an unchanged target set preserves the
# file's mtime and resume correctly skips discovery already completed against
# it; a changed target set bumps the mtime and forces re-discovery.
os.makedirs(_disc(output_path), exist_ok=True)
masscan_file = os.path.join(_disc(output_path), 'resolved_targets.txt')
_write_if_changed(masscan_file, ''.join(f'{target}\n' for target in masscan_targets))
# Save IP-to-hostname mapping (also content-stable to avoid needless churn)
mapping_file = os.path.join(_disc(output_path), 'ip_hostname_map.json')
_write_if_changed(mapping_file, json.dumps(ip_to_hostname, indent=2))
print(_COLOR_INFO + f'Resolved {len(ip_to_hostname)} hostnames to IPs' + _COLOR_RESET)
print(_COLOR_INFO + f'Target file: {masscan_file}' + _COLOR_RESET)
return masscan_file, ip_to_hostname
def _merge_ssl_cert_hostnames(output_path, ip_to_hostname):
"""Fill gaps in ip_to_hostname from ssl-cert CN/SAN data in nse_results/.
Never overwrites an existing (operator-supplied) entry -- a name typed
into the target file always wins over a cert-derived guess. For an IP
with no prior entry, prefers the certificate's commonName; falls back to
the first non-wildcard Subject Alternative Name. A host whose only
names are wildcards gets no entry, since a wildcard is not a usable
scan target. Returns a new dict; persists it to
<output_path>/discovery/ip_hostname_map.json via _write_if_changed().
"""
nse_dir = f'{output_path}/nse_results'
merged = dict(ip_to_hostname)
if os.path.isdir(nse_dir):
for fname in sorted(os.listdir(nse_dir)):