From ffbc41f3caaf530055815a25cf382b826af978f7 Mon Sep 17 00:00:00 2001 From: Justin Bollinger Date: Tue, 17 Feb 2026 12:43:22 -0500 Subject: [PATCH 1/4] feat: add computer account filtering and NetNTLM dedup - Detect and optionally filter Windows computer accounts (username ending with $) from pwdump-format NTLM hash files (type 1000) - Detect and optionally deduplicate NetNTLMv1/v2 hashes (types 5500/5600) by username, keeping first occurrence - Add 10 tests covering both features Fixes #27 Fixes #28 Co-Authored-By: Claude Opus 4.6 --- hate_crack/attacks.py | 4 +- hate_crack/main.py | 176 ++++++++++++++++++++++++++----- tests/test_ntlm_preprocessing.py | 158 +++++++++++++++++++++++++++ 3 files changed, 307 insertions(+), 31 deletions(-) create mode 100644 tests/test_ntlm_preprocessing.py diff --git a/hate_crack/attacks.py b/hate_crack/attacks.py index d6ba8fd..420e23d 100644 --- a/hate_crack/attacks.py +++ b/hate_crack/attacks.py @@ -498,6 +498,4 @@ def ollama_attack(ctx: Any) -> None: "industry": industry, "location": location, } - ctx.hcatOllama( - ctx.hcatHashType, ctx.hcatHashFile, "target", target_info - ) + ctx.hcatOllama(ctx.hcatHashType, ctx.hcatHashFile, "target", target_info) diff --git a/hate_crack/main.py b/hate_crack/main.py index 39af63f..a4a67e6 100755 --- a/hate_crack/main.py +++ b/hate_crack/main.py @@ -113,7 +113,6 @@ def _resolve_config_destination(): return os.getcwd() - def _ensure_hashfile_in_cwd(hashfile_path): """Ensure hashfile path points to cwd to keep output files in execution dir.""" if not hashfile_path: @@ -155,9 +154,7 @@ else: _config_path = _resolve_config_path() if not _config_path: print("Initializing config.json from config.json.example") - src_config = os.path.abspath( - os.path.join(_package_path, "config.json.example") - ) + src_config = os.path.abspath(os.path.join(_package_path, "config.json.example")) config_dir = _resolve_config_destination() dst_config = os.path.abspath(os.path.join(config_dir, "config.json")) shutil.copy(src_config, dst_config) @@ -214,9 +211,7 @@ def ensure_binary(binary_path, build_dir=None, name=None): print( "\nRun 'make install' from the repository directory to install with assets:" ) - print( - " cd /path/to/hate_crack && make install" - ) + print(" cd /path/to/hate_crack && make install") quit(1) # Binary missing - need to build @@ -677,9 +672,7 @@ if not SKIP_INIT: # Verify hcstat2gen binary (optional, for LLM attacks) # Note: hcstat2gen is part of hashcat-utils, already in hate_crack repo - hcstat2gen_path = ( - hate_path + "/hashcat-utils/bin/" + hcatHcstat2genBin - ) + hcstat2gen_path = hate_path + "/hashcat-utils/bin/" + hcatHcstat2genBin try: ensure_binary( hcstat2gen_path, @@ -775,7 +768,8 @@ def usage(): def ascii_art(): from hate_crack import __version__ - print(r""" + print( + r""" ___ ___ __ _________ __ / | \_____ _/ |_ ____ \_ ___ \____________ ____ | | __ @@ -783,8 +777,11 @@ def ascii_art(): \ Y // __ \| | \ ___/ \ \____| | \// __ \\ \___| < \___|_ /(____ /__| \___ >____\______ /|__| (____ /\___ >__|_ \ \/ \/ \/_____/ \/ \/ \/ \/ - Version """ + __version__ + """ - """) + Version """ + + __version__ + + """ + """ + ) # File selector with tab autocomplete @@ -919,6 +916,66 @@ def _write_field_sorted_unique(input_path, output_path, field_index, delimiter=" return False +def _filter_computer_accounts(input_path, output_path, delimiter=":"): + """Filter out computer accounts (usernames ending with $) from a hash file. + + Reads the input file, removes lines where the first field (username) + ends with '$', and writes the remaining lines to output_path. + Returns the number of computer accounts removed. + """ + removed = 0 + kept_lines = [] + try: + with open(input_path, "r", errors="replace") as src: + for line in src: + stripped = line.rstrip("\n") + if not stripped: + continue + username = stripped.split(delimiter, 1)[0] + if username.endswith("$"): + removed += 1 + else: + kept_lines.append(stripped) + with open(output_path, "w") as dst: + for line in kept_lines: + dst.write(line + "\n") + except FileNotFoundError: + pass + return removed + + +def _dedup_netntlm_by_username(input_path, output_path, delimiter=":"): + """Deduplicate NetNTLM hashes by username, keeping the first occurrence. + + NetNTLM format: username::domain:challenge:response:blob + The username is the first field before the delimiter. + Returns a tuple of (total_lines, duplicates_removed). + """ + seen_usernames = set() + kept_lines = [] + duplicates = 0 + total = 0 + try: + with open(input_path, "r", errors="replace") as src: + for line in src: + stripped = line.rstrip("\n") + if not stripped: + continue + total += 1 + username = stripped.split(delimiter, 1)[0].lower() + if username in seen_usernames: + duplicates += 1 + else: + seen_usernames.add(username) + kept_lines.append(stripped) + with open(output_path, "w") as dst: + for line in kept_lines: + dst.write(line + "\n") + except FileNotFoundError: + pass + return total, duplicates + + def _run_hashcat_show(hash_type, hash_file, output_path): result = subprocess.run( [ @@ -1556,7 +1613,7 @@ def hcatOllama(hcatHashType, hcatHashFile, mode, context_data): wordlist_sample = "\n".join(lines) prompt = ( "Generate baseword to be used in a denylist for keeping users from setting their passwords with these basewords." - "Study the patterns, character choices, and structures. Focus on patterns like capitalization, leetspeak, suffixes, and common substitutions. Here are the sample passwords:\n" + "Study the patterns, character choices, and structures. Focus on patterns like capitalization, leetspeak, suffixes, and common substitutions. Here are the sample passwords:\n" f"{wordlist_sample}" ) elif mode == "target": @@ -1579,12 +1636,14 @@ def hcatOllama(hcatHashType, hcatHashFile, mode, context_data): # Step B: Call Ollama API to generate candidates print(f"Generating password candidates via Ollama ({ollamaModel})...") api_url = f"{ollamaUrl}/api/generate" - payload = json.dumps({ - "model": ollamaModel, - "prompt": prompt, - "stream": False, - "options": {"num_ctx": ollamaNumCtx}, - }).encode("utf-8") + payload = json.dumps( + { + "model": ollamaModel, + "prompt": prompt, + "stream": False, + "options": {"num_ctx": ollamaNumCtx}, + } + ).encode("utf-8") if debug_mode: print(f"[DEBUG] Ollama API URL: {api_url}") @@ -1612,7 +1671,9 @@ def hcatOllama(hcatHashType, hcatHashFile, mode, context_data): with urllib.request.urlopen(req, timeout=600) as resp: result = json.loads(resp.read().decode("utf-8")) if debug_mode: - print(f"[DEBUG] Ollama response (after pull): {json.dumps(result, indent=2)}") + print( + f"[DEBUG] Ollama response (after pull): {json.dumps(result, indent=2)}" + ) except Exception as retry_err: print(f"Error calling Ollama API after pull: {retry_err}") return @@ -1633,7 +1694,9 @@ def hcatOllama(hcatHashType, hcatHashFile, mode, context_data): response_text = result.get("response", "") if "I'm sorry, but I can't help with that" in response_text: - print("Error: Ollama refused the request. Try a different model or adjust your prompt.") + print( + "Error: Ollama refused the request. Try a different model or adjust your prompt." + ) return raw_lines = response_text.strip().split("\n") # Filter out blank lines and lines that look like numbering/explanation @@ -1664,7 +1727,9 @@ def hcatOllama(hcatHashType, hcatHashFile, mode, context_data): print(f"Generated {len(candidates)} password candidates -> {candidates_path}") if debug_mode: filtered_count = len(raw_lines) - len(candidates) - print(f"[DEBUG] Filtered out {filtered_count} lines from Ollama response ({len(raw_lines)} raw -> {len(candidates)} candidates)") + print( + f"[DEBUG] Filtered out {filtered_count} lines from Ollama response ({len(raw_lines)} raw -> {len(candidates)} candidates)" + ) # Step C: Run hashcat wordlist attack with LLM-generated candidates (no rules) print("Running wordlist attack with LLM-generated candidates...") @@ -1690,14 +1755,14 @@ def hcatOllama(hcatHashType, hcatHashFile, mode, context_data): return # Step D: Run hashcat with LLM candidates against every rule in the rules directory - rule_files = sorted( - f for f in os.listdir(rulesDirectory) if f != ".DS_Store" - ) + rule_files = sorted(f for f in os.listdir(rulesDirectory) if f != ".DS_Store") if not rule_files: print("No rule files found in rules directory. Skipping rule-based attacks.") return - print(f"\nRunning LLM candidates with {len(rule_files)} rule file(s) from {rulesDirectory}...") + print( + f"\nRunning LLM candidates with {len(rule_files)} rule file(s) from {rulesDirectory}..." + ) for rule in rule_files: rule_path = os.path.join(rulesDirectory, rule) print(f"\n\tRunning with rule: {rule}") @@ -1725,7 +1790,6 @@ def hcatOllama(hcatHashType, hcatHashFile, mode, context_data): return - # Middle fast Combinator Attack def hcatMiddleCombinator(hcatHashType, hcatHashFile): global hcatProcess @@ -3590,6 +3654,25 @@ def main(): if re.search(r"[a-f0-9A-F]{32}:[a-f0-9A-F]{32}:::", hcatHashFileLine): pwdump_format = True print("PWDUMP format detected...") + # Detect computer accounts (usernames ending with $) + computer_count = sum( + 1 + for line in open(hcatHashFile, "r", errors="replace") + if line.strip() and line.split(":", 1)[0].endswith("$") + ) + if computer_count > 0: + print( + f"Detected {computer_count} computer account(s)" + " (usernames ending with $)." + ) + filter_choice = ( + input("Would you like to ignore computer accounts? (Y) ") or "Y" + ) + if filter_choice.upper() == "Y": + filtered_path = f"{hcatHashFile}.filtered" + removed = _filter_computer_accounts(hcatHashFile, filtered_path) + print(f"Removed {removed} computer account(s).") + hcatHashFile = filtered_path print("Parsing NT hashes...") _write_field_sorted_unique(hcatHashFile, f"{hcatHashFile}.nt", 4) print("Parsing LM hashes...") @@ -3640,6 +3723,43 @@ def main(): else: print("unknown format....does it have usernames?") exit(1) + # Detect and optionally deduplicate NetNTLM hashes by username + if hcatHashType in ("5500", "5600"): + total, duplicates = _dedup_netntlm_by_username( + hcatHashFile, hcatHashFile + ".dedup" + ) + if duplicates > 0: + print( + f"Detected {duplicates} duplicate account(s) out of" + f" {total} total NetNTLM hashes." + ) + dedup_choice = ( + input( + "Would you like to ignore duplicate accounts" + " (keep first occurrence only)? (Y) " + ) + or "Y" + ) + if dedup_choice.upper() == "Y": + hcatHashFileOrig = hcatHashFile + hcatHashFile = hcatHashFile + ".dedup" + print( + f"Using deduplicated hash file with" + f" {total - duplicates} unique accounts." + ) + else: + # Remove the dedup file if user chose not to use it + try: + os.remove(hcatHashFile + ".dedup") + except OSError: + pass + else: + # No duplicates found, clean up + try: + os.remove(hcatHashFile + ".dedup") + except OSError: + pass + # Check POT File for Already Cracked Hashes if not os.path.isfile(hcatHashFile + ".out"): hcatOutput = open(hcatHashFile + ".out", "w+") diff --git a/tests/test_ntlm_preprocessing.py b/tests/test_ntlm_preprocessing.py new file mode 100644 index 0000000..9780b06 --- /dev/null +++ b/tests/test_ntlm_preprocessing.py @@ -0,0 +1,158 @@ +"""Tests for NTLM/NetNTLM hash preprocessing helpers (issues #27 and #28).""" + +import os +import sys +import importlib + +import pytest + + +@pytest.fixture +def main_module(monkeypatch): + """Load hate_crack.main with SKIP_INIT to access helper functions.""" + monkeypatch.setenv("HATE_CRACK_SKIP_INIT", "1") + if "hate_crack.main" in sys.modules: + mod = sys.modules["hate_crack.main"] + importlib.reload(mod) + return mod + import hate_crack.main as mod + + return mod + + +class TestFilterComputerAccounts: + """Issue #27 - filter accounts ending with $.""" + + def test_removes_computer_accounts(self, tmp_path, main_module): + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "COMPUTER1$:1002:aad3b435b51404ee:hash2:::\n" + "user2:1003:aad3b435b51404ee:hash3:::\n" + "WORKSTATION$:1004:aad3b435b51404ee:hash4:::\n" + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 2 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + assert all("$" not in line.split(":", 1)[0] for line in lines) + + def test_no_computer_accounts(self, tmp_path, main_module): + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "user2:1002:aad3b435b51404ee:hash2:::\n" + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 0 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + + def test_all_computer_accounts(self, tmp_path, main_module): + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "COMP1$:1001:aad3b435b51404ee:hash1:::\n" + "COMP2$:1002:aad3b435b51404ee:hash2:::\n" + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 2 + content = output_file.read_text().strip() + assert content == "" + + def test_missing_file(self, tmp_path, main_module): + removed = main_module._filter_computer_accounts( + str(tmp_path / "nonexistent.txt"), + str(tmp_path / "output.txt"), + ) + assert removed == 0 + + def test_empty_file(self, tmp_path, main_module): + hash_file = tmp_path / "hashes.txt" + hash_file.write_text("") + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 0 + + +class TestDedupNetntlmByUsername: + """Issue #28 - deduplicate NetNTLM hashes by username.""" + + def test_removes_duplicates(self, tmp_path, main_module): + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "user1::DOMAIN:challenge1:response1:blob1\n" + "user2::DOMAIN:challenge2:response2:blob2\n" + "user1::DOMAIN:challenge3:response3:blob3\n" + "user3::DOMAIN:challenge4:response4:blob4\n" + "user2::DOMAIN:challenge5:response5:blob5\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 5 + assert duplicates == 2 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 3 + # First occurrences should be kept + assert "challenge1" in lines[0] + assert "challenge2" in lines[1] + assert "challenge4" in lines[2] + + def test_case_insensitive_dedup(self, tmp_path, main_module): + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "User1::DOMAIN:challenge1:response1:blob1\n" + "USER1::DOMAIN:challenge2:response2:blob2\n" + "user1::DOMAIN:challenge3:response3:blob3\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 3 + assert duplicates == 2 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 1 + + def test_no_duplicates(self, tmp_path, main_module): + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "user1::DOMAIN:challenge1:response1:blob1\n" + "user2::DOMAIN:challenge2:response2:blob2\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 2 + assert duplicates == 0 + + def test_missing_file(self, tmp_path, main_module): + total, duplicates = main_module._dedup_netntlm_by_username( + str(tmp_path / "nonexistent.txt"), + str(tmp_path / "output.txt"), + ) + assert total == 0 + assert duplicates == 0 + + def test_empty_file(self, tmp_path, main_module): + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text("") + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 0 + assert duplicates == 0 From 3550838064986fa01236a347fea2270361d8932b Mon Sep 17 00:00:00 2001 From: Justin Bollinger Date: Tue, 17 Feb 2026 12:46:48 -0500 Subject: [PATCH 2/4] fix: address QA review issues for NTLM preprocessing (#27, #28) - Add type hints to _filter_computer_accounts and _dedup_netntlm_by_username - Fix unclosed file handle when reading hash file for format detection - Extract _count_computer_accounts helper to eliminate duplicate file reads - Stream _filter_computer_accounts output instead of collecting in memory - Only write .dedup file when duplicates actually exist - Add tests for _count_computer_accounts, malformed lines, and no-file-on-zero-dupes Co-Authored-By: Claude Opus 4.6 --- hate_crack/main.py | 59 +++++++++++++++++------------- tests/test_ntlm_preprocessing.py | 63 ++++++++++++++++++++++++++++++++ 2 files changed, 97 insertions(+), 25 deletions(-) diff --git a/hate_crack/main.py b/hate_crack/main.py index a4a67e6..8949b96 100755 --- a/hate_crack/main.py +++ b/hate_crack/main.py @@ -916,7 +916,23 @@ def _write_field_sorted_unique(input_path, output_path, field_index, delimiter=" return False -def _filter_computer_accounts(input_path, output_path, delimiter=":"): +def _count_computer_accounts(input_path: str, delimiter: str = ":") -> int: + """Count computer accounts (usernames ending with $) in a hash file.""" + count = 0 + try: + with open(input_path, "r", errors="replace") as src: + for line in src: + stripped = line.strip() + if stripped and stripped.split(delimiter, 1)[0].endswith("$"): + count += 1 + except FileNotFoundError: + pass + return count + + +def _filter_computer_accounts( + input_path: str, output_path: str, delimiter: str = ":" +) -> int: """Filter out computer accounts (usernames ending with $) from a hash file. Reads the input file, removes lines where the first field (username) @@ -924,9 +940,10 @@ def _filter_computer_accounts(input_path, output_path, delimiter=":"): Returns the number of computer accounts removed. """ removed = 0 - kept_lines = [] try: - with open(input_path, "r", errors="replace") as src: + with open(input_path, "r", errors="replace") as src, open( + output_path, "w" + ) as dst: for line in src: stripped = line.rstrip("\n") if not stripped: @@ -935,24 +952,24 @@ def _filter_computer_accounts(input_path, output_path, delimiter=":"): if username.endswith("$"): removed += 1 else: - kept_lines.append(stripped) - with open(output_path, "w") as dst: - for line in kept_lines: - dst.write(line + "\n") + dst.write(stripped + "\n") except FileNotFoundError: pass return removed -def _dedup_netntlm_by_username(input_path, output_path, delimiter=":"): +def _dedup_netntlm_by_username( + input_path: str, output_path: str, delimiter: str = ":" +) -> tuple[int, int]: """Deduplicate NetNTLM hashes by username, keeping the first occurrence. NetNTLM format: username::domain:challenge:response:blob The username is the first field before the delimiter. + Only writes output_path when duplicates are found. Returns a tuple of (total_lines, duplicates_removed). """ - seen_usernames = set() - kept_lines = [] + seen_usernames: set[str] = set() + kept_lines: list[str] = [] duplicates = 0 total = 0 try: @@ -968,9 +985,10 @@ def _dedup_netntlm_by_username(input_path, output_path, delimiter=":"): else: seen_usernames.add(username) kept_lines.append(stripped) - with open(output_path, "w") as dst: - for line in kept_lines: - dst.write(line + "\n") + if duplicates > 0: + with open(output_path, "w") as dst: + for line in kept_lines: + dst.write(line + "\n") except FileNotFoundError: pass return total, duplicates @@ -3650,16 +3668,13 @@ def main(): if hcatHashType == "1000": lmHashesFound = False pwdump_format = False - hcatHashFileLine = open(hcatHashFile, "r").readline().strip().lstrip("\ufeff") + with open(hcatHashFile, "r") as f: + hcatHashFileLine = f.readline().strip().lstrip("\ufeff") if re.search(r"[a-f0-9A-F]{32}:[a-f0-9A-F]{32}:::", hcatHashFileLine): pwdump_format = True print("PWDUMP format detected...") # Detect computer accounts (usernames ending with $) - computer_count = sum( - 1 - for line in open(hcatHashFile, "r", errors="replace") - if line.strip() and line.split(":", 1)[0].endswith("$") - ) + computer_count = _count_computer_accounts(hcatHashFile) if computer_count > 0: print( f"Detected {computer_count} computer account(s)" @@ -3753,12 +3768,6 @@ def main(): os.remove(hcatHashFile + ".dedup") except OSError: pass - else: - # No duplicates found, clean up - try: - os.remove(hcatHashFile + ".dedup") - except OSError: - pass # Check POT File for Already Cracked Hashes if not os.path.isfile(hcatHashFile + ".out"): diff --git a/tests/test_ntlm_preprocessing.py b/tests/test_ntlm_preprocessing.py index 9780b06..8f7f2e5 100644 --- a/tests/test_ntlm_preprocessing.py +++ b/tests/test_ntlm_preprocessing.py @@ -20,6 +20,36 @@ def main_module(monkeypatch): return mod +class TestCountComputerAccounts: + """Issue #27 - count computer accounts (helper for detection).""" + + def test_counts_computer_accounts(self, tmp_path, main_module): + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "COMPUTER1$:1002:aad3b435b51404ee:hash2:::\n" + "user2:1003:aad3b435b51404ee:hash3:::\n" + "WORKSTATION$:1004:aad3b435b51404ee:hash4:::\n" + ) + assert main_module._count_computer_accounts(str(hash_file)) == 2 + + def test_no_computer_accounts(self, tmp_path, main_module): + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "user2:1002:aad3b435b51404ee:hash2:::\n" + ) + assert main_module._count_computer_accounts(str(hash_file)) == 0 + + def test_missing_file(self, tmp_path, main_module): + assert main_module._count_computer_accounts(str(tmp_path / "nope.txt")) == 0 + + def test_empty_file(self, tmp_path, main_module): + hash_file = tmp_path / "hashes.txt" + hash_file.write_text("") + assert main_module._count_computer_accounts(str(hash_file)) == 0 + + class TestFilterComputerAccounts: """Issue #27 - filter accounts ending with $.""" @@ -84,6 +114,21 @@ class TestFilterComputerAccounts: ) assert removed == 0 + def test_malformed_lines(self, tmp_path, main_module): + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "malformed_line_without_dollar\n" + "COMP$:1002:aad3b435b51404ee:hash2:::\n" + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 1 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + class TestDedupNetntlmByUsername: """Issue #28 - deduplicate NetNTLM hashes by username.""" @@ -138,6 +183,8 @@ class TestDedupNetntlmByUsername: ) assert total == 2 assert duplicates == 0 + # Output file should NOT be created when no duplicates exist + assert not output_file.exists() def test_missing_file(self, tmp_path, main_module): total, duplicates = main_module._dedup_netntlm_by_username( @@ -156,3 +203,19 @@ class TestDedupNetntlmByUsername: ) assert total == 0 assert duplicates == 0 + assert not output_file.exists() + + def test_malformed_lines(self, tmp_path, main_module): + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "user1::DOMAIN:challenge1:response1:blob1\n" + "malformed_line_without_colons\n" + "user2::DOMAIN:challenge2:response2:blob2\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 3 + assert duplicates == 0 + assert not output_file.exists() From 31b6a2f6b1729987fa75e627f89d11a7c0a60919 Mon Sep 17 00:00:00 2001 From: Justin Bollinger Date: Tue, 17 Feb 2026 12:53:20 -0500 Subject: [PATCH 3/4] test: add comprehensive pwdump filter pipeline and edge case tests Add TestWriteFieldSortedUnique (7 tests) and TestPwdumpFilterPipeline (8 tests) covering the full filter -> extract NT/LM pipeline. Add edge case tests for unicode, BOM, long lines, permissions, and delimiter handling. Co-Authored-By: Claude Opus 4.6 --- tests/test_ntlm_preprocessing.py | 317 ++++++++++++++++++- tests/test_ntlm_preprocessing_edge_cases.py | 325 ++++++++++++++++++++ 2 files changed, 641 insertions(+), 1 deletion(-) create mode 100644 tests/test_ntlm_preprocessing_edge_cases.py diff --git a/tests/test_ntlm_preprocessing.py b/tests/test_ntlm_preprocessing.py index 8f7f2e5..938df80 100644 --- a/tests/test_ntlm_preprocessing.py +++ b/tests/test_ntlm_preprocessing.py @@ -1,6 +1,5 @@ """Tests for NTLM/NetNTLM hash preprocessing helpers (issues #27 and #28).""" -import os import sys import importlib @@ -219,3 +218,319 @@ class TestDedupNetntlmByUsername: assert total == 3 assert duplicates == 0 assert not output_file.exists() + + +class TestWriteFieldSortedUnique: + """Test _write_field_sorted_unique helper for extracting hash fields.""" + + def test_extracts_nt_hashes_field_4(self, tmp_path, main_module): + """Extract NT hashes (field 4) from pwdump format.""" + hash_file = tmp_path / "pwdump.txt" + hash_file.write_text( + "user1:500:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + "user2:501:aad3b435b51404eeaad3b435b51404ee:8846f7eaee8fb117ad06bdd830b7586c:::\n" + "user3:502:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + ) + output_file = tmp_path / "nt.txt" + result = main_module._write_field_sorted_unique( + str(hash_file), str(output_file), 4, ":" + ) + assert result is True + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 # Two unique NT hashes + assert "31d6cfe0d16ae931b73c59d7e0c089c0" in lines + assert "8846f7eaee8fb117ad06bdd830b7586c" in lines + + def test_extracts_lm_hashes_field_3(self, tmp_path, main_module): + """Extract LM hashes (field 3) from pwdump format.""" + hash_file = tmp_path / "pwdump.txt" + hash_file.write_text( + "user1:500:e52cac67419a9a224a3b108f3fa6cb6d:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + "user2:501:aad3b435b51404eeaad3b435b51404ee:8846f7eaee8fb117ad06bdd830b7586c:::\n" + "user3:502:e52cac67419a9a224a3b108f3fa6cb6d:abc123def456:::\n" + ) + output_file = tmp_path / "lm.txt" + result = main_module._write_field_sorted_unique( + str(hash_file), str(output_file), 3, ":" + ) + assert result is True + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 # Two unique LM hashes + assert "aad3b435b51404eeaad3b435b51404ee" in lines + assert "e52cac67419a9a224a3b108f3fa6cb6d" in lines + + def test_sorts_and_deduplicates(self, tmp_path, main_module): + """Verify sorting and deduplication.""" + hash_file = tmp_path / "pwdump.txt" + hash_file.write_text( + "user1:500:lm1:zzz:::\n" + "user2:501:lm2:aaa:::\n" + "user3:502:lm3:mmm:::\n" + "user4:503:lm4:aaa:::\n" # Duplicate NT hash + ) + output_file = tmp_path / "nt.txt" + main_module._write_field_sorted_unique(str(hash_file), str(output_file), 4, ":") + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 3 + assert lines == ["aaa", "mmm", "zzz"] # Sorted alphabetically + + def test_handles_missing_fields(self, tmp_path, main_module): + """Lines with fewer fields than requested should be skipped.""" + hash_file = tmp_path / "pwdump.txt" + hash_file.write_text( + "user1:500:lm1:nt1:::\n" + "malformed:500\n" # Only 2 fields + "user2:501:lm2:nt2:::\n" + ) + output_file = tmp_path / "nt.txt" + main_module._write_field_sorted_unique(str(hash_file), str(output_file), 4, ":") + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + assert "nt1" in lines + assert "nt2" in lines + + def test_missing_input_file(self, tmp_path, main_module): + """Should return False when input file doesn't exist.""" + result = main_module._write_field_sorted_unique( + str(tmp_path / "nonexistent.txt"), str(tmp_path / "output.txt"), 4, ":" + ) + assert result is False + + def test_empty_file(self, tmp_path, main_module): + """Empty input should create empty output.""" + hash_file = tmp_path / "empty.txt" + hash_file.write_text("") + output_file = tmp_path / "out.txt" + result = main_module._write_field_sorted_unique( + str(hash_file), str(output_file), 4, ":" + ) + assert result is True + assert output_file.read_text().strip() == "" + + +class TestPwdumpFilterPipeline: + """Full pipeline tests: filter -> extract NT/LM -> verify output.""" + + def test_full_pipeline_with_filtering(self, tmp_path, main_module): + """Test complete flow: filter computer accounts, extract NT/LM hashes.""" + # Step 1: Create pwdump file with mixed accounts + pwdump_file = tmp_path / "dump.txt" + pwdump_file.write_text( + "Administrator:500:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + "Guest:501:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + "DESKTOP-ABC$:1001:aad3b435b51404eeaad3b435b51404ee:8846f7eaee8fb117ad06bdd830b7586c:::\n" + "john:1002:e52cac67419a9a224a3b108f3fa6cb6d:5f4dcc3b5aa765d61d8327deb882cf99:::\n" + "WORKSTATION$:1003:aad3b435b51404eeaad3b435b51404ee:deadbeefcafebabe1234567890abcdef:::\n" + "alice:1004:aad3b435b51404eeaad3b435b51404ee:6cb75f652a9b52798eb6cf2201057c73:::\n" + ) + + # Step 2: Count computer accounts + count = main_module._count_computer_accounts(str(pwdump_file)) + assert count == 2 + + # Step 3: Filter computer accounts + filtered_file = tmp_path / "dump.txt.filtered" + removed = main_module._filter_computer_accounts( + str(pwdump_file), str(filtered_file) + ) + assert removed == 2 + + # Step 4: Verify filtered file preserves complete pwdump format + filtered_lines = filtered_file.read_text().strip().split("\n") + assert len(filtered_lines) == 4 + for line in filtered_lines: + # Each line should have pwdump format: user:uid:LM:NT::: + parts = line.split(":") + assert len(parts) == 7 # 7 fields total (6 colons) + assert not parts[0].endswith("$") # No computer accounts + + # Step 5: Extract NT hashes from filtered file + nt_file = tmp_path / "dump.txt.filtered.nt" + result = main_module._write_field_sorted_unique( + str(filtered_file), str(nt_file), 4, ":" + ) + assert result is True + + # Step 6: Verify NT hashes are correct and don't include computer account hashes + nt_hashes = nt_file.read_text().strip().split("\n") + assert len(nt_hashes) == 3 # Three unique NT hashes from non-computer accounts + assert "31d6cfe0d16ae931b73c59d7e0c089c0" in nt_hashes # Admin/Guest empty hash + assert "5f4dcc3b5aa765d61d8327deb882cf99" in nt_hashes # john's hash + assert "6cb75f652a9b52798eb6cf2201057c73" in nt_hashes # alice's hash + # Computer account hashes should NOT be present + assert "8846f7eaee8fb117ad06bdd830b7586c" not in nt_hashes + assert "deadbeefcafebabe1234567890abcdef" not in nt_hashes + + # Step 7: Extract LM hashes from filtered file + lm_file = tmp_path / "dump.txt.filtered.lm" + result = main_module._write_field_sorted_unique( + str(filtered_file), str(lm_file), 3, ":" + ) + assert result is True + + # Step 8: Verify LM hashes are correct + lm_hashes = lm_file.read_text().strip().split("\n") + assert len(lm_hashes) == 2 # Two unique LM hashes + assert ( + "aad3b435b51404eeaad3b435b51404ee" in lm_hashes + ) # Empty LM (Admin/Guest/alice) + assert "e52cac67419a9a224a3b108f3fa6cb6d" in lm_hashes # john's LM + + def test_pipeline_without_filtering(self, tmp_path, main_module): + """Test pipeline when no computer accounts exist.""" + pwdump_file = tmp_path / "dump.txt" + pwdump_file.write_text( + "Administrator:500:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + "john:1002:e52cac67419a9a224a3b108f3fa6cb6d:5f4dcc3b5aa765d61d8327deb882cf99:::\n" + "alice:1004:aad3b435b51404eeaad3b435b51404ee:6cb75f652a9b52798eb6cf2201057c73:::\n" + ) + + # Count should be zero + count = main_module._count_computer_accounts(str(pwdump_file)) + assert count == 0 + + # Extract NT directly from original file (simulate no filtering step) + nt_file = tmp_path / "dump.txt.nt" + main_module._write_field_sorted_unique(str(pwdump_file), str(nt_file), 4, ":") + + nt_hashes = nt_file.read_text().strip().split("\n") + assert len(nt_hashes) == 3 + assert "31d6cfe0d16ae931b73c59d7e0c089c0" in nt_hashes + assert "5f4dcc3b5aa765d61d8327deb882cf99" in nt_hashes + assert "6cb75f652a9b52798eb6cf2201057c73" in nt_hashes + + def test_pipeline_with_realistic_hashes(self, tmp_path, main_module): + """Test with realistic Active Directory pwdump data.""" + pwdump_file = tmp_path / "ad_dump.txt" + pwdump_file.write_text( + # Real format examples + "Administrator:500:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + "krbtgt:502:aad3b435b51404eeaad3b435b51404ee:d3c02561bba6ee4ad6cfd024ec8fda5d:::\n" + "CORP-DC01$:1000:aad3b435b51404eeaad3b435b51404ee:a1b2c3d4e5f6a7b8c9d0e1f2a3b4c5d6:::\n" + "CORP-WKS01$:1001:aad3b435b51404eeaad3b435b51404ee:1234567890abcdef1234567890abcdef:::\n" + "jdoe:1102:e52cac67419a9a224a3b108f3fa6cb6d:8846f7eaee8fb117ad06bdd830b7586c:::\n" + "CORP-SRV01$:1003:aad3b435b51404eeaad3b435b51404ee:fedcba0987654321fedcba0987654321:::\n" + "asmith:1103:aad3b435b51404eeaad3b435b51404ee:2b2ac52d43a3d5c5c5b5b5f5e5d5c5a5:::\n" + ) + + # Count computer accounts (domain controllers, workstations, servers) + count = main_module._count_computer_accounts(str(pwdump_file)) + assert count == 3 + + # Filter them out + filtered_file = tmp_path / "ad_dump.txt.filtered" + removed = main_module._filter_computer_accounts( + str(pwdump_file), str(filtered_file) + ) + assert removed == 3 + + # Extract NT hashes + nt_file = tmp_path / "ad_dump.txt.filtered.nt" + main_module._write_field_sorted_unique(str(filtered_file), str(nt_file), 4, ":") + + nt_hashes = nt_file.read_text().strip().split("\n") + assert len(nt_hashes) == 4 # Four unique NT hashes from user accounts + # Verify no computer account hashes leaked through + assert "a1b2c3d4e5f6a7b8c9d0e1f2a3b4c5d6" not in nt_hashes + assert "1234567890abcdef1234567890abcdef" not in nt_hashes + assert "fedcba0987654321fedcba0987654321" not in nt_hashes + + def test_pipeline_with_bom(self, tmp_path, main_module): + """Test that BOM character doesn't break filtering.""" + pwdump_file = tmp_path / "bom_dump.txt" + # Note: The main.py code strips BOM at line 3672 but _filter_computer_accounts + # doesn't handle it. This tests if that causes issues. + pwdump_file.write_text( + "\ufeffAdministrator:500:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + "COMPUTER$:1001:aad3b435b51404eeaad3b435b51404ee:8846f7eaee8fb117ad06bdd830b7586c:::\n" + "john:1002:e52cac67419a9a224a3b108f3fa6cb6d:5f4dcc3b5aa765d61d8327deb882cf99:::\n" + ) + + filtered_file = tmp_path / "bom_dump.txt.filtered" + removed = main_module._filter_computer_accounts( + str(pwdump_file), str(filtered_file) + ) + assert removed == 1 + + filtered_lines = filtered_file.read_text().strip().split("\n") + # BOM should be preserved in first line since filter doesn't strip it + assert len(filtered_lines) == 2 + # First username might have BOM attached + first_username = filtered_lines[0].split(":", 1)[0] + # BOM gets written through, so we check it's either with or without + assert "Administrator" in first_username + + def test_pipeline_preserves_all_fields(self, tmp_path, main_module): + """Verify filtered file maintains exact pwdump structure.""" + pwdump_file = tmp_path / "structure.txt" + pwdump_file.write_text( + "user1:500:LM1:NT1:extra1:extra2:extra3\n" + "COMP$:501:LM2:NT2:extra4:extra5:extra6\n" + "user2:502:LM3:NT3:extra7:extra8:extra9\n" + ) + + filtered_file = tmp_path / "structure.txt.filtered" + main_module._filter_computer_accounts(str(pwdump_file), str(filtered_file)) + + filtered_lines = filtered_file.read_text().strip().split("\n") + assert len(filtered_lines) == 2 + + # Each line should have exactly 7 fields + for line in filtered_lines: + assert line.count(":") == 6 + parts = line.split(":") + assert len(parts) == 7 + + def test_pipeline_empty_username_edge_case(self, tmp_path, main_module): + """Test handling of lines with empty username field.""" + pwdump_file = tmp_path / "empty_user.txt" + pwdump_file.write_text( + "user1:500:lm1:nt1:::\n" + ":501:lm2:nt2:::\n" # Empty username + "COMP$:502:lm3:nt3:::\n" + "user2:503:lm4:nt4:::\n" + ) + + count = main_module._count_computer_accounts(str(pwdump_file)) + assert count == 1 # Only COMP$ should count + + filtered_file = tmp_path / "empty_user.txt.filtered" + removed = main_module._filter_computer_accounts( + str(pwdump_file), str(filtered_file) + ) + assert removed == 1 + + filtered_lines = filtered_file.read_text().strip().split("\n") + assert len(filtered_lines) == 3 # Empty username line should be kept + + def test_pipeline_with_duplicate_hashes_across_accounts( + self, tmp_path, main_module + ): + """Verify deduplication works correctly when multiple users share hashes.""" + pwdump_file = tmp_path / "dup_hashes.txt" + pwdump_file.write_text( + "user1:500:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" + "user2:501:aad3b435b51404eeaad3b435b51404ee:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" # Same hash + "COMP$:502:aad3b435b51404eeaad3b435b51404ee:8846f7eaee8fb117ad06bdd830b7586c:::\n" + "user3:503:e52cac67419a9a224a3b108f3fa6cb6d:31d6cfe0d16ae931b73c59d7e0c089c0:::\n" # Same NT, different LM + ) + + filtered_file = tmp_path / "dup_hashes.txt.filtered" + main_module._filter_computer_accounts(str(pwdump_file), str(filtered_file)) + + nt_file = tmp_path / "dup_hashes.txt.filtered.nt" + main_module._write_field_sorted_unique(str(filtered_file), str(nt_file), 4, ":") + + nt_hashes = nt_file.read_text().strip().split("\n") + # Should have only 1 unique NT hash (31d6cfe0d16ae931b73c59d7e0c089c0) + assert len(nt_hashes) == 1 + assert nt_hashes[0] == "31d6cfe0d16ae931b73c59d7e0c089c0" + + lm_file = tmp_path / "dup_hashes.txt.filtered.lm" + main_module._write_field_sorted_unique(str(filtered_file), str(lm_file), 3, ":") + + lm_hashes = lm_file.read_text().strip().split("\n") + # Should have 2 unique LM hashes + assert len(lm_hashes) == 2 + assert "aad3b435b51404eeaad3b435b51404ee" in lm_hashes + assert "e52cac67419a9a224a3b108f3fa6cb6d" in lm_hashes diff --git a/tests/test_ntlm_preprocessing_edge_cases.py b/tests/test_ntlm_preprocessing_edge_cases.py new file mode 100644 index 0000000..b93b7f2 --- /dev/null +++ b/tests/test_ntlm_preprocessing_edge_cases.py @@ -0,0 +1,325 @@ +"""Additional edge case tests for NTLM preprocessing (issues #27 and #28).""" + +import os +import sys +import importlib +import pytest + + +@pytest.fixture +def main_module(monkeypatch): + """Load hate_crack.main with SKIP_INIT to access helper functions.""" + monkeypatch.setenv("HATE_CRACK_SKIP_INIT", "1") + if "hate_crack.main" in sys.modules: + mod = sys.modules["hate_crack.main"] + importlib.reload(mod) + return mod + import hate_crack.main as mod + + return mod + + +class TestFilterComputerAccountsEdgeCases: + """Additional edge cases for computer account filtering.""" + + def test_unicode_in_usernames(self, tmp_path, main_module): + """Test handling of Unicode characters in usernames.""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "Müller$:1002:aad3b435b51404ee:hash2:::\n" + "用户:1003:aad3b435b51404ee:hash3:::\n", + encoding="utf-8", + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 1 + lines = output_file.read_text(encoding="utf-8").strip().split("\n") + assert len(lines) == 2 + assert "Müller$" not in output_file.read_text(encoding="utf-8") + + def test_dollar_in_middle_of_username(self, tmp_path, main_module): + """Test that $ only at end triggers filtering.""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user$name:1001:aad3b435b51404ee:hash1:::\n" + "COMP1$:1002:aad3b435b51404ee:hash2:::\n" + "$startdollar:1003:aad3b435b51404ee:hash3:::\n", + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + # Only COMP1$ should be removed (ends with $) + assert removed == 1 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + assert "user$name" in lines[0] + assert "$startdollar" in lines[1] + + def test_empty_username(self, tmp_path, main_module): + """Test handling of lines with empty username field.""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + ":1001:aad3b435b51404ee:hash1:::\nuser1:1002:aad3b435b51404ee:hash2:::\n" + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 0 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + + def test_very_long_lines(self, tmp_path, main_module): + """Test handling of very long hash lines.""" + hash_file = tmp_path / "hashes.txt" + long_hash = "a" * 10000 + hash_file.write_text( + f"user1:1001:{long_hash}:hash1:::\nCOMP1$:1002:{long_hash}:hash2:::\n" + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 1 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 1 + assert "user1" in lines[0] + + def test_output_file_exists_overwrites(self, tmp_path, main_module): + """Test that output file is overwritten if it exists.""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "COMP1$:1002:aad3b435b51404ee:hash2:::\n" + ) + output_file = tmp_path / "filtered.txt" + output_file.write_text("OLD CONTENT THAT SHOULD BE REPLACED\n") + + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 1 + content = output_file.read_text() + assert "OLD CONTENT" not in content + assert "user1" in content + + def test_permission_denied_output(self, tmp_path, main_module): + """Test handling when output file cannot be written.""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "COMP1$:1002:aad3b435b51404ee:hash2:::\n" + ) + # Create a read-only directory for output + readonly_dir = tmp_path / "readonly" + readonly_dir.mkdir() + os.chmod(readonly_dir, 0o444) + + output_file = readonly_dir / "filtered.txt" + + # Should handle PermissionError gracefully + try: + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + # If it doesn't raise, should return 0 + assert removed == 0 + except PermissionError: + # This is also acceptable behavior + pass + finally: + # Cleanup - restore permissions + os.chmod(readonly_dir, 0o755) + + def test_blank_lines_preserved_count(self, tmp_path, main_module): + """Test that blank lines don't affect removed count.""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "\n" + "COMP1$:1002:aad3b435b51404ee:hash2:::\n" + "\n\n" + "user2:1003:aad3b435b51404ee:hash3:::\n" + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 1 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + + +class TestDedupNetntlmEdgeCases: + """Additional edge cases for NetNTLM deduplication.""" + + def test_unicode_usernames(self, tmp_path, main_module): + """Test deduplication with Unicode usernames.""" + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "用户1::DOMAIN:challenge1:response1:blob1\n" + "用户1::DOMAIN:challenge2:response2:blob2\n" + "Müller::DOMAIN:challenge3:response3:blob3\n", + encoding="utf-8", + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 3 + assert duplicates == 1 + lines = output_file.read_text(encoding="utf-8").strip().split("\n") + assert len(lines) == 2 + + def test_whitespace_in_usernames(self, tmp_path, main_module): + """Test handling of usernames with leading/trailing spaces.""" + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "user1::DOMAIN:challenge1:response1:blob1\n" + " user1::DOMAIN:challenge2:response2:blob2\n" + "user1 ::DOMAIN:challenge3:response3:blob3\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + # These are NOT duplicates because split() doesn't strip whitespace + # This is actually correct behavior - the usernames differ + assert total == 3 + assert duplicates == 0 + + def test_case_variations(self, tmp_path, main_module): + """Test various case combinations are deduplicated.""" + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "User1::DOMAIN:challenge1:response1:blob1\n" + "user1::DOMAIN:challenge2:response2:blob2\n" + "USER1::DOMAIN:challenge3:response3:blob3\n" + "uSeR1::DOMAIN:challenge4:response4:blob4\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 4 + assert duplicates == 3 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 1 + + def test_very_long_username(self, tmp_path, main_module): + """Test handling of very long usernames.""" + hash_file = tmp_path / "netntlm.txt" + long_user = "a" * 10000 + hash_file.write_text( + f"{long_user}::DOMAIN:challenge1:response1:blob1\n" + f"{long_user}::DOMAIN:challenge2:response2:blob2\n" + "user2::DOMAIN:challenge3:response3:blob3\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 3 + assert duplicates == 1 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + + def test_memory_efficiency_many_duplicates(self, tmp_path, main_module): + """Test memory handling with many duplicates.""" + hash_file = tmp_path / "netntlm.txt" + # Create 1000 lines, all duplicates of the same user + content = "" + for i in range(1000): + content += f"user1::DOMAIN:challenge{i}:response{i}:blob{i}\n" + hash_file.write_text(content) + + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 1000 + assert duplicates == 999 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 1 + + def test_special_characters_in_username(self, tmp_path, main_module): + """Test usernames with special characters.""" + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "user@domain::DOMAIN:challenge1:response1:blob1\n" + "user-name::DOMAIN:challenge2:response2:blob2\n" + "user.name::DOMAIN:challenge3:response3:blob3\n" + "user@domain::DOMAIN:challenge4:response4:blob4\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 4 + assert duplicates == 1 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 3 + + def test_only_colons(self, tmp_path, main_module): + """Test handling of malformed lines with only colons.""" + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "user1::DOMAIN:challenge1:response1:blob1\n" + ":::::\n" + "user2::DOMAIN:challenge2:response2:blob2\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + # Empty username from ::::: should be counted but not cause errors + assert total == 3 + + def test_preserved_line_order(self, tmp_path, main_module): + """Test that first occurrence is preserved, not last.""" + hash_file = tmp_path / "netntlm.txt" + hash_file.write_text( + "user1::DOMAIN:FIRST:response1:blob1\n" + "user2::DOMAIN:challenge2:response2:blob2\n" + "user1::DOMAIN:SECOND:response3:blob3\n" + "user3::DOMAIN:challenge4:response4:blob4\n" + "user1::DOMAIN:THIRD:response5:blob5\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 5 + assert duplicates == 2 + content = output_file.read_text() + # First occurrence should be kept + assert "FIRST" in content + assert "SECOND" not in content + assert "THIRD" not in content + + +class TestCountComputerAccountsEdgeCases: + """Additional edge cases for computer account counting.""" + + def test_mixed_delimiters(self, tmp_path, main_module): + """Test that only the specified delimiter is used.""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_text( + "user1:1001:aad3b435b51404ee:hash1:::\n" + "COMP$;1002;aad3b435b51404ee;hash2\n" # Different delimiter + ) + # Default delimiter is ":" + assert main_module._count_computer_accounts(str(hash_file)) == 0 + # Custom delimiter ";" - COMP$ is first field with ";" delimiter + assert main_module._count_computer_accounts(str(hash_file), ";") == 1 + + def test_single_field_line(self, tmp_path, main_module): + """Test lines with only one field (no delimiter).""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_text("COMP1$\nuser1:1001:aad3b435b51404ee:hash1:::\n") + # COMP1$ should still be counted (split returns single element) + assert main_module._count_computer_accounts(str(hash_file)) == 1 From 89bcc6d74aa9b40725a4f05abcfc45e2c94f5628 Mon Sep 17 00:00:00 2001 From: Justin Bollinger Date: Tue, 17 Feb 2026 12:59:09 -0500 Subject: [PATCH 4/4] fix: QA hardening for NTLM preprocessing - Catch PermissionError/OSError in file operations (not just FileNotFoundError) - Refactor _dedup_netntlm_by_username to two-pass streaming (memory safe) - Handle CRLF line endings in filter and dedup functions - Add KeyboardInterrupt handling with temp file cleanup during preprocessing - Track .filtered/.dedup temp files for cleanup on interruption - Add CRLF line ending tests for both filter and dedup Co-Authored-By: Claude Opus 4.6 --- hate_crack/main.py | 271 ++++++++++++++++++------------- tests/test_ntlm_preprocessing.py | 39 +++++ 2 files changed, 197 insertions(+), 113 deletions(-) diff --git a/hate_crack/main.py b/hate_crack/main.py index 8949b96..845719a 100755 --- a/hate_crack/main.py +++ b/hate_crack/main.py @@ -925,8 +925,9 @@ def _count_computer_accounts(input_path: str, delimiter: str = ":") -> int: stripped = line.strip() if stripped and stripped.split(delimiter, 1)[0].endswith("$"): count += 1 - except FileNotFoundError: - pass + except (FileNotFoundError, PermissionError, OSError) as e: + if not isinstance(e, FileNotFoundError): + print(f"Warning: Could not process {input_path}: {e}") return count @@ -941,11 +942,12 @@ def _filter_computer_accounts( """ removed = 0 try: - with open(input_path, "r", errors="replace") as src, open( - output_path, "w" - ) as dst: + with ( + open(input_path, "r", errors="replace") as src, + open(output_path, "w") as dst, + ): for line in src: - stripped = line.rstrip("\n") + stripped = line.rstrip("\r\n") if not stripped: continue username = stripped.split(delimiter, 1)[0] @@ -953,8 +955,9 @@ def _filter_computer_accounts( removed += 1 else: dst.write(stripped + "\n") - except FileNotFoundError: - pass + except (FileNotFoundError, PermissionError, OSError) as e: + if not isinstance(e, FileNotFoundError): + print(f"Warning: Could not process {input_path}: {e}") return removed @@ -967,15 +970,19 @@ def _dedup_netntlm_by_username( The username is the first field before the delimiter. Only writes output_path when duplicates are found. Returns a tuple of (total_lines, duplicates_removed). + + Uses a two-pass approach to avoid holding all lines in memory: + - Pass 1: scan to collect seen usernames and count duplicates + - Pass 2: stream non-duplicate lines directly to the output file """ seen_usernames: set[str] = set() - kept_lines: list[str] = [] duplicates = 0 total = 0 try: + # Pass 1: count totals and identify unique usernames with open(input_path, "r", errors="replace") as src: for line in src: - stripped = line.rstrip("\n") + stripped = line.rstrip("\r\n") if not stripped: continue total += 1 @@ -984,13 +991,25 @@ def _dedup_netntlm_by_username( duplicates += 1 else: seen_usernames.add(username) - kept_lines.append(stripped) + + # Pass 2: write non-duplicate lines directly to output (only if needed) if duplicates > 0: - with open(output_path, "w") as dst: - for line in kept_lines: - dst.write(line + "\n") - except FileNotFoundError: - pass + first_seen: set[str] = set() + with ( + open(input_path, "r", errors="replace") as src, + open(output_path, "w") as dst, + ): + for line in src: + stripped = line.rstrip("\r\n") + if not stripped: + continue + username = stripped.split(delimiter, 1)[0].lower() + if username not in first_seen: + first_seen.add(username) + dst.write(stripped + "\n") + except (FileNotFoundError, PermissionError, OSError) as e: + if not isinstance(e, FileNotFoundError): + print(f"Warning: Could not process {input_path}: {e}") return total, duplicates @@ -3665,109 +3684,135 @@ def main(): # Get Initial Input Hash Count # If LM or NT Mode Selected and pwdump Format Detected, Prompt For LM to NT Attack - if hcatHashType == "1000": - lmHashesFound = False - pwdump_format = False - with open(hcatHashFile, "r") as f: - hcatHashFileLine = f.readline().strip().lstrip("\ufeff") - if re.search(r"[a-f0-9A-F]{32}:[a-f0-9A-F]{32}:::", hcatHashFileLine): - pwdump_format = True - print("PWDUMP format detected...") - # Detect computer accounts (usernames ending with $) - computer_count = _count_computer_accounts(hcatHashFile) - if computer_count > 0: + # Track temp files created during preprocessing for cleanup on interruption + _preprocessing_temp_files: list[str] = [] + + def _cleanup_preprocessing_temps() -> None: + """Remove any temp files created during preprocessing.""" + for path in _preprocessing_temp_files: + try: + os.remove(path) + except OSError: + pass + + try: + if hcatHashType == "1000": + lmHashesFound = False + pwdump_format = False + with open(hcatHashFile, "r") as f: + hcatHashFileLine = f.readline().strip().lstrip("\ufeff") + if re.search(r"[a-f0-9A-F]{32}:[a-f0-9A-F]{32}:::", hcatHashFileLine): + pwdump_format = True + print("PWDUMP format detected...") + # Detect computer accounts (usernames ending with $) + computer_count = _count_computer_accounts(hcatHashFile) + if computer_count > 0: + print( + f"Detected {computer_count} computer account(s)" + " (usernames ending with $)." + ) + filter_choice = ( + input("Would you like to ignore computer accounts? (Y) ") or "Y" + ) + if filter_choice.upper() == "Y": + filtered_path = f"{hcatHashFile}.filtered" + _preprocessing_temp_files.append(filtered_path) + removed = _filter_computer_accounts(hcatHashFile, filtered_path) + print(f"Removed {removed} computer account(s).") + hcatHashFile = filtered_path + # Keep this file - remove from cleanup list + _preprocessing_temp_files.remove(filtered_path) + print("Parsing NT hashes...") + _write_field_sorted_unique(hcatHashFile, f"{hcatHashFile}.nt", 4) + print("Parsing LM hashes...") + _write_field_sorted_unique(hcatHashFile, f"{hcatHashFile}.lm", 3) + if ( + (lineCount(hcatHashFile + ".lm") == 1) + and ( + hcatHashFileLine.split(":")[2].lower() + != "aad3b435b51404eeaad3b435b51404ee" + ) + ) or (lineCount(hcatHashFile + ".lm") > 1): + lmHashesFound = True + lmChoice = ( + input( + "LM hashes identified. Would you like to brute force" + " the LM hashes first? (Y) " + ) + or "Y" + ) + if lmChoice.upper() == "Y": + hcatLMtoNT() + hcatHashFileOrig = hcatHashFile + hcatHashFile = hcatHashFile + ".nt" + elif re.search(r"^[a-f0-9A-F]{32}$", hcatHashFileLine): + pwdump_format = False + print("PWDUMP format was not detected...") + print("Hash only detected") + elif re.search(r"^.+:[a-f0-9A-F]{32}$", hcatHashFileLine): + pwdump_format = False + print("PWDUMP format was not detected...") + print("username with Hash detected") + _write_field_sorted_unique(hcatHashFile, f"{hcatHashFile}.nt", 2) + hcatHashFileOrig = hcatHashFile + hcatHashFile = hcatHashFile + ".nt" + elif re.search(r"^.+::.+:.+:[a-f0-9A-F]{64}:", hcatHashFileLine): + # NetNTLMv2 format: username::domain:server_challenge:ntproofstr:blob + # NetNTLMv2-ESS format is similar, with Enhanced Session Security + pwdump_format = False + # Try to detect if it's NetNTLMv2-ESS (has specific markers) + if re.search( + r"^.+::.+:.+:[a-f0-9A-F]{16}:[a-f0-9A-F]{32}:[a-f0-9A-F]+$", + hcatHashFileLine, + ): + print("NetNTLMv2-ESS format detected") + print("Note: Hash type should be 5600 for NetNTLMv2-ESS hashes") + else: + print("NetNTLMv2 format detected") + print("Note: Hash type should be 5500 for NetNTLMv2 hashes") + else: + print("unknown format....does it have usernames?") + exit(1) + # Detect and optionally deduplicate NetNTLM hashes by username + if hcatHashType in ("5500", "5600"): + dedup_path = hcatHashFile + ".dedup" + _preprocessing_temp_files.append(dedup_path) + total, duplicates = _dedup_netntlm_by_username(hcatHashFile, dedup_path) + if duplicates == 0: + # No dedup file was created, remove from cleanup list + _preprocessing_temp_files.remove(dedup_path) + else: print( - f"Detected {computer_count} computer account(s)" - " (usernames ending with $)." + f"Detected {duplicates} duplicate account(s) out of" + f" {total} total NetNTLM hashes." ) - filter_choice = ( - input("Would you like to ignore computer accounts? (Y) ") or "Y" - ) - if filter_choice.upper() == "Y": - filtered_path = f"{hcatHashFile}.filtered" - removed = _filter_computer_accounts(hcatHashFile, filtered_path) - print(f"Removed {removed} computer account(s).") - hcatHashFile = filtered_path - print("Parsing NT hashes...") - _write_field_sorted_unique(hcatHashFile, f"{hcatHashFile}.nt", 4) - print("Parsing LM hashes...") - _write_field_sorted_unique(hcatHashFile, f"{hcatHashFile}.lm", 3) - if ( - (lineCount(hcatHashFile + ".lm") == 1) - and ( - hcatHashFileLine.split(":")[2].lower() - != "aad3b435b51404eeaad3b435b51404ee" - ) - ) or (lineCount(hcatHashFile + ".lm") > 1): - lmHashesFound = True - lmChoice = ( + dedup_choice = ( input( - "LM hashes identified. Would you like to brute force the LM hashes first? (Y) " + "Would you like to ignore duplicate accounts" + " (keep first occurrence only)? (Y) " ) or "Y" ) - if lmChoice.upper() == "Y": - hcatLMtoNT() - hcatHashFileOrig = hcatHashFile - hcatHashFile = hcatHashFile + ".nt" - elif re.search(r"^[a-f0-9A-F]{32}$", hcatHashFileLine): - pwdump_format = False - print("PWDUMP format was not detected...") - print("Hash only detected") - elif re.search(r"^.+:[a-f0-9A-F]{32}$", hcatHashFileLine): - pwdump_format = False - print("PWDUMP format was not detected...") - print("username with Hash detected") - _write_field_sorted_unique(hcatHashFile, f"{hcatHashFile}.nt", 2) - hcatHashFileOrig = hcatHashFile - hcatHashFile = hcatHashFile + ".nt" - elif re.search(r"^.+::.+:.+:[a-f0-9A-F]{64}:", hcatHashFileLine): - # NetNTLMv2 format: username::domain:server_challenge:ntproofstr:blob - # NetNTLMv2-ESS format is similar, with Enhanced Session Security - pwdump_format = False - # Try to detect if it's NetNTLMv2-ESS (has specific markers) - if re.search( - r"^.+::.+:.+:[a-f0-9A-F]{16}:[a-f0-9A-F]{32}:[a-f0-9A-F]+$", - hcatHashFileLine, - ): - print("NetNTLMv2-ESS format detected") - print("Note: Hash type should be 5600 for NetNTLMv2-ESS hashes") - else: - print("NetNTLMv2 format detected") - print("Note: Hash type should be 5500 for NetNTLMv2 hashes") - else: - print("unknown format....does it have usernames?") - exit(1) - # Detect and optionally deduplicate NetNTLM hashes by username - if hcatHashType in ("5500", "5600"): - total, duplicates = _dedup_netntlm_by_username( - hcatHashFile, hcatHashFile + ".dedup" - ) - if duplicates > 0: - print( - f"Detected {duplicates} duplicate account(s) out of" - f" {total} total NetNTLM hashes." - ) - dedup_choice = ( - input( - "Would you like to ignore duplicate accounts" - " (keep first occurrence only)? (Y) " - ) - or "Y" - ) - if dedup_choice.upper() == "Y": - hcatHashFileOrig = hcatHashFile - hcatHashFile = hcatHashFile + ".dedup" - print( - f"Using deduplicated hash file with" - f" {total - duplicates} unique accounts." - ) - else: - # Remove the dedup file if user chose not to use it - try: - os.remove(hcatHashFile + ".dedup") - except OSError: - pass + if dedup_choice.upper() == "Y": + hcatHashFileOrig = hcatHashFile + hcatHashFile = dedup_path + # Keep this file - remove from cleanup list + _preprocessing_temp_files.remove(dedup_path) + print( + f"Using deduplicated hash file with" + f" {total - duplicates} unique accounts." + ) + else: + # Remove the dedup file if user chose not to use it + try: + os.remove(dedup_path) + except OSError: + pass + _preprocessing_temp_files.remove(dedup_path) + except KeyboardInterrupt: + print("\nPreprocessing interrupted. Cleaning up temp files...") + _cleanup_preprocessing_temps() + sys.exit(1) # Check POT File for Already Cracked Hashes if not os.path.isfile(hcatHashFile + ".out"): diff --git a/tests/test_ntlm_preprocessing.py b/tests/test_ntlm_preprocessing.py index 938df80..0cbdb97 100644 --- a/tests/test_ntlm_preprocessing.py +++ b/tests/test_ntlm_preprocessing.py @@ -128,6 +128,25 @@ class TestFilterComputerAccounts: lines = output_file.read_text().strip().split("\n") assert len(lines) == 2 + def test_crlf_line_endings(self, tmp_path, main_module): + """Test that CRLF (Windows) line endings are handled correctly.""" + hash_file = tmp_path / "hashes.txt" + hash_file.write_bytes( + b"user1:1001:aad3b435b51404ee:hash1:::\r\n" + b"COMPUTER1$:1002:aad3b435b51404ee:hash2:::\r\n" + b"user2:1003:aad3b435b51404ee:hash3:::\r\n" + ) + output_file = tmp_path / "filtered.txt" + removed = main_module._filter_computer_accounts( + str(hash_file), str(output_file) + ) + assert removed == 1 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + # Verify no stray \r in output + for line in lines: + assert "\r" not in line + class TestDedupNetntlmByUsername: """Issue #28 - deduplicate NetNTLM hashes by username.""" @@ -219,6 +238,26 @@ class TestDedupNetntlmByUsername: assert duplicates == 0 assert not output_file.exists() + def test_crlf_line_endings(self, tmp_path, main_module): + """Test that CRLF (Windows) line endings are handled correctly.""" + hash_file = tmp_path / "netntlm.txt" + hash_file.write_bytes( + b"user1::DOMAIN:challenge1:response1:blob1\r\n" + b"user2::DOMAIN:challenge2:response2:blob2\r\n" + b"user1::DOMAIN:challenge3:response3:blob3\r\n" + ) + output_file = tmp_path / "dedup.txt" + total, duplicates = main_module._dedup_netntlm_by_username( + str(hash_file), str(output_file) + ) + assert total == 3 + assert duplicates == 1 + lines = output_file.read_text().strip().split("\n") + assert len(lines) == 2 + # Verify no stray \r in output + for line in lines: + assert "\r" not in line + class TestWriteFieldSortedUnique: """Test _write_field_sorted_unique helper for extracting hash fields."""