diff --git a/.bandit b/.bandit index 0b01d12b..35460139 100644 --- a/.bandit +++ b/.bandit @@ -1,3 +1,3 @@ # FILE: .bandit [bandit] -skips: ['B607', 'B605'] \ No newline at end of file +skips: ['B607', 'B605'] diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index 05540e45..54ebb4e6 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -1,5 +1 @@ * @DefinetlyNotAI -wifi_stealer.py @ski-sketch -packet_sniffer.py @ski-sketch -bluetooth_details.py @ski-sketch -bluetooth_logger.py @ski-sketch \ No newline at end of file diff --git a/CREDITS.md b/.github/CREDITS.md similarity index 97% rename from CREDITS.md rename to .github/CREDITS.md index c66fcfec..3b25275d 100644 --- a/CREDITS.md +++ b/.github/CREDITS.md @@ -1,47 +1,47 @@ -# CREDITS - -This project is built on the shoulders of giants and inspired by the work of many talented individuals and -organizations. We acknowledge their contributions and are grateful for the knowledge and tools they have shared. - - - Contributor GitHub Profiles - - - - -## 👨‍💻 Coders Credits 👨‍💻 - -### Wifi-Stealer.py, bluetooth_details.py and bluetooth_logger.py by ski-sketch - -Created Wi-Fi Password Stealer using python -The sole creator of the code of wifi-stealer -Also created bluetooth_details.py and bluetooth_logger.py -which are used to get the details of the bluetooth devices -and log the details of the bluetooth devices respectively - -- [ski-sketch](https://github.com/ski-sketch) - -## 🛠️ Refactorers Credits 🛠️ - -Until Now, no one. Become a contributor and help us spread the word. - -## 🔨 Enhancers Credits 🔨 - -Until Now, no one. Become a contributor and help us spread the word. - -## 🐛 Bug bounty credits 🐛 - -### Found development bug - -Found and attempted fix of 2 bugs: Zipping name error - `--dev` flag loop - -- [ski-sketch](https://github.com/ski-sketch) - -# Acknowledgments - -This project would not be possible without the contributions and inspirations from the above-mentioned individuals and -organizations. We are deeply grateful for their work and the community that supports it. +# CREDITS + +This project is built on the shoulders of giants and inspired by the work of many talented individuals and +organizations. We acknowledge their contributions and are grateful for the knowledge and tools they have shared. + + + Contributor GitHub Profiles + + + + +## 👨‍💻 Coders Credits 👨‍💻 + +### Wifi-Stealer.py, bluetooth_details.py and bluetooth_logger.py by ski-sketch + +Created Wi-Fi Password Stealer using python +The sole creator of the code of wifi-stealer +Also created bluetooth_details.py and bluetooth_logger.py +which are used to get the details of the bluetooth devices +and log the details of the bluetooth devices respectively + +- [ski-sketch](https://github.com/ski-sketch) + +## 🛠️ Refactorers Credits 🛠️ + +Until Now, no one. Become a contributor and help us spread the word. + +## 🔨 Enhancers Credits 🔨 + +Until Now, no one. Become a contributor and help us spread the word. + +## 🐛 Bug bounty credits 🐛 + +### Found development bug + +Found and attempted fix of 2 bugs: Zipping name error - `--dev` flag loop + +- [ski-sketch](https://github.com/ski-sketch) + +# Acknowledgments + +This project would not be possible without the contributions and inspirations from the above-mentioned individuals and +organizations. We are deeply grateful for their work and the community that supports it. diff --git a/DCO.md b/.github/DCO.md similarity index 95% rename from DCO.md rename to .github/DCO.md index c3a29727..6a2e3c60 100644 --- a/DCO.md +++ b/.github/DCO.md @@ -32,5 +32,5 @@ By making a contribution to this project, I certify that: are public and that a record of the contribution (including all personal information I submit with it, including my sign-off) is maintained indefinitely and may be redistributed consistent with - this project or the open source license(s) involved. + this project or the open source license(s) involved. diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml index f813bca4..530686d9 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.yml +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -1,64 +1,78 @@ name: Report a bug -description: Tell us about a bug or issue you may have identified in Logicytics. -title: "Provide a general summary of the issue" +description: Report reproducible Logicytics v4 behavior without sharing private evidence. +title: "bug: " labels: [ "progress/Unreviewed" ] -assignees: "DefinetlyNotAI" +assignees: [ "DefinetlyNotAI" ] body: - type: checkboxes attributes: label: Prerequisites - description: Take a couple minutes to help our maintainers work faster. options: - - label: I have [searched](https://github.com/DefinetlyNotAI/Logicytics/issues?utf8=%E2%9C%93&q=is%3Aissue) for duplicate or closed issues. + - label: I searched existing issues and reproduced this on the latest supported v4 release. required: true - - label: I have read the [contributing guidelines](https://github.com/DefinetlyNotAI/Logicytics/blob/main/CONTRIBUTING.md). + - label: I read the contribution and security guidance and removed credentials, private evidence, and host identifiers. required: true - - label: I have checked that I am on the latest release, and have run the `--debug` flag and have made sure no external modification are responsible for this bug. + - label: I ran `python -m logicytics preflight` and the relevant command again with a sanitized configuration. required: true - type: textarea - id: what-happened + id: description attributes: - label: Describe the issue - description: Provide a summary of the issue and what you expected to happen, including specific steps to reproduce. + label: What happened? + description: Describe the actual result, expected result, and exact reproduction steps. + validations: + required: true + - type: input + id: version + attributes: + label: Logicytics version and commit + placeholder: "4.0.0, commit abc1234" + validations: + required: true + - type: input + id: environment + attributes: + label: Windows and Python versions + placeholder: "Windows 11 24H2; Python 3.11.9" + validations: + required: true + - type: dropdown + id: action + attributes: + label: Affected action + options: + - preflight or plan + - standard or balanced collection + - quick or thorough collection + - offline collection + - extensions or non-Python MODs + - performance collection + - direct collector execution + - debug, update, or developer action + - usage or semantic matching + - public Python API + - package, hash, or artifact reading + - other validations: required: true - type: textarea - id: d_log + id: command attributes: - label: Debugger Log - description: Include the log File generated by running `.\Logicytics --debug`, upload the File contents and paste it in. + label: Sanitized command and configuration + description: Include the command, profile/mode, approved capabilities, elevation state, and relevant non-secret settings. + render: powershell validations: required: true - type: textarea - id: b_log + id: diagnostics attributes: - label: Basic Log - description: If possible, delete the default log, and attempt to run `.\Logicytics` again with the steps you have highlighted, upload the new log here. + label: Redacted diagnostics + description: Paste only relevant redacted preflight, manifest status, collector failure, or debug details. Never attach collected evidence or secrets. validations: required: false - type: textarea id: extra attributes: - label: Anything else? - description: Include anything you deem important. + label: Additional context + description: Note whether optional WMIC, BitLocker, Sysinternals, PowerShell, or elevation capabilities were available. validations: required: false - - type: dropdown - id: flags_list - attributes: - label: What flags_list were you using to run Logicytics? - multiple: false - options: - - Threading - - Dev - - Default - - Modded - - Other - - N/A - - type: input - id: version - attributes: - label: What version of Logicytics are you using? - placeholder: "e.g., v2.1.0 or v1.4.0" - validations: - required: true diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index e62326c4..5368c4be 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -1,49 +1,31 @@ -## Pull Request Template +## Summary -### Prerequisites + - - +## Contract impact -- [ ] I have [searched](https://github.com/DefinetlyNotAI/Logicytics/pulls) for duplicate or closed issues. -- [ ] I have read the [contributing guidelines](https://github.com/DefinetlyNotAI/Logicytics/blob/main/CONTRIBUTING.md). -- [ ] I have followed the instructions in the [wiki](https://github.com/DefinetlyNotAI/Logicytics/wiki) about - contributions. -- [ ] I have updated the documentation accordingly, if required. -- [ ] I have tested my code with the `--dev` flag, if required. + -### PR Type +## Verification - - + -- [ ] Bug fix -- [ ] Deprecation Change -- [ ] New feature -- [ ] Refactoring -- [ ] Documentation - update -- [ ] ⚠️ Breaking change ⚠️ +- [ ] Relevant focused tests pass. +- [ ] `python -m unittest discover -v` passes. +- [ ] `python -m compileall -q logicytics core tests` passes. +- [ ] `python -m logicytics preflight` reports no invalid core collector. +- [ ] `git diff --check` passes. +- [ ] Live Windows integration tests pass when the change touches host behavior. -### Description +## Review checklist - +- [ ] I read [CONTRIBUTING.md](../CONTRIBUTING.md) and searched for duplicate work. +- [ ] This pull request contains one coherent theme and conventional commits. +- [ ] Collector metadata, cancellation, isolation, and artifact registration remain valid. +- [ ] Configuration, output, migration, release, and feature-status docs are updated when affected. +- [ ] No generated evidence, credentials, private data, caches, or unrelated changes are included. +- [ ] I agree to the [Developer Certificate of Origin](DCO.md) and repository license. -### Motivation and Context +## Related issues - - -### Credit - - - - - -### Issues Fixed - - + diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 554c5ed8..dd0fa13a 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -18,4 +18,4 @@ updates: interval: daily labels: - "type/Dependencies" - - "type/Github Actions" \ No newline at end of file + - "type/Github Actions" diff --git a/.github/script/validate-commit-msg.sh b/.github/script/validate-commit-msg.sh new file mode 100644 index 00000000..73aeada4 --- /dev/null +++ b/.github/script/validate-commit-msg.sh @@ -0,0 +1,55 @@ +#!/usr/bin/env sh + +set -eu + +commit_msg_file="$1" +first_line="$(head -n 1 "$commit_msg_file")" + +pattern='^((fixup|squash|amend|reword)! )?(build|chore|ci|docs|feat|fix|perf|refactor|revert|style|test)(\([A-Za-z0-9._/-]+\))?!?: .+$' + +if printf '%s\n' "$first_line" | grep -Eq "$pattern"; then + exit 0 +fi + +cat >&2 <<'EOF' + +Invalid commit message. + +Expected Conventional Commits format: + + : + (): + !: + ()!: + +Allowed types: + + build + chore + ci + docs + feat + fix + perf + refactor + revert + style + test + +Expected Conventional Commits format: + + : + (): + !: + ()!: + +Autosquash commits are also accepted: + + fixup! + squash! + amend! + reword! + +EOF + +exit 1 diff --git a/.github/workflows/greetings.yml b/.github/workflows/greetings.yml index 6363edf0..b6d651f9 100644 --- a/.github/workflows/greetings.yml +++ b/.github/workflows/greetings.yml @@ -1,6 +1,6 @@ name: Greetings -on: [pull_request_target, issues] +on: [ pull_request_target, issues ] permissions: contents: read diff --git a/.github/workflows/publish-wiki.yml b/.github/workflows/publish-wiki.yml new file mode 100644 index 00000000..a2468377 --- /dev/null +++ b/.github/workflows/publish-wiki.yml @@ -0,0 +1,75 @@ +name: Publish documentation wiki + +on: + push: + branches: [ main ] + paths: + - "docs/**" + - ".github/workflows/publish-wiki.yml" + workflow_dispatch: + +permissions: + contents: write + +concurrency: + group: documentation-wiki + cancel-in-progress: true + +jobs: + publish: + name: Publish docs to the repository wiki + runs-on: ubuntu-latest + steps: + - name: Harden the runner + uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + with: + egress-policy: audit + + - name: Checkout repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Checkout wiki repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + continue-on-error: true + with: + repository: ${{ github.repository }}.wiki + path: wiki + token: ${{ secrets.GITHUB_TOKEN }} + fetch-depth: 0 + + - name: Prepare wiki checkout + shell: bash + env: + REPOSITORY: ${{ github.repository }} + TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set -euo pipefail + if [ ! -d wiki/.git ]; then + rm -rf wiki + mkdir wiki + git -C wiki init -b master + git -C wiki remote add origin "https://x-access-token:${TOKEN}@github.com/${REPOSITORY}.wiki.git" + fi + + - name: Replace wiki pages with docs + shell: bash + run: | + set -euo pipefail + find wiki -mindepth 1 ! -path 'wiki/.git' ! -path 'wiki/.git/*' -exec rm -rf {} + + cp -R docs/. wiki/ + + - name: Commit and push wiki pages + shell: bash + run: | + set -euo pipefail + git -C wiki config user.name "github-actions[bot]" + git -C wiki config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git -C wiki add --all + if git -C wiki diff --cached --quiet; then + echo "Wiki is already current." + exit 0 + fi + git -C wiki commit -m "docs: publish repository documentation" + git -C wiki push origin HEAD:master diff --git a/.gitignore b/.gitignore index a25f2a40..ec85fc0a 100644 --- a/.gitignore +++ b/.gitignore @@ -319,6 +319,8 @@ $RECYCLE.BIN/ *.pyc /CODE/SysInternal_Suite/.sys.ignore /ACCESS/ +/output/ +/tools/ /CODE/vulnscan/tools/NN features/ /CODE/logicytics/User_History.json.gz /CODE/logicytics/User_History.json @@ -328,4 +330,6 @@ $RECYCLE.BIN/ /CODE/SysInternal_Suite/pslist.exe /CODE/SysInternal_Suite/PsLoggedon.exe /CODE/SysInternal_Suite/psloglist.exe -.idea/ \ No newline at end of file +.idea/ +.cache/ +.temp/ diff --git a/.mailmap b/.mailmap index f5b83505..3e9b544f 100644 --- a/.mailmap +++ b/.mailmap @@ -1 +1 @@ -Shahm Najeeb \ No newline at end of file +Shahm Najeeb diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 5c46e656..836af996 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -3,12 +3,20 @@ repos: rev: v8.16.3 hooks: - id: gitleaks + - repo: https://github.com/pre-commit/pre-commit-hooks rev: v4.4.0 hooks: - id: end-of-file-fixer - id: trailing-whitespace - - repo: https://github.com/pylint-dev/pylint - rev: v2.17.2 + + - # noinspection YAMLSchemaValidation + repo: local hooks: - - id: pylint + - # noinspection YAMLSchemaValidation + id: conventional-commit-message + name: Validate conventional commit message + entry: .github/script/validate-commit-msg.sh + language: script + stages: + - commit-msg diff --git a/CODE/Logicytics.py b/CODE/Logicytics.py deleted file mode 100644 index 3149d645..00000000 --- a/CODE/Logicytics.py +++ /dev/null @@ -1,522 +0,0 @@ -from __future__ import annotations - -import gc -import os -import subprocess -import sys -from concurrent.futures import ThreadPoolExecutor, as_completed -from datetime import datetime - -from prettytable import PrettyTable - -from logicytics import ( - Log, - execute, - check, - get, - file_management, - flag, - DEBUG, - DELETE_LOGS, - config, -) - -# Initialization -log = Log({"log_level": DEBUG, "delete_log": DELETE_LOGS}) -ACTION, SUB_ACTION = None, None -MAX_WORKERS = config.getint( - "Settings", "max_workers", fallback=min(32, (os.cpu_count() or 1) + 4) -) -log.debug(f"MAX_WORKERS: {MAX_WORKERS}") - - -class ExecuteScript: - def __init__(self): - self.execution_list = self.__generate_execution_list() - - @staticmethod - def __safe_remove(file_name: str, file_list: list[str] | set[str]) -> list[str]: - file_set = set(file_list) - if file_name in file_set: - file_set.remove(file_name) - else: - log.critical( - f"The file {file_name} should exist in this directory - But was not found!" - ) - return list(file_set) - - @staticmethod - def __safe_append(file_name: str, file_list: list[str] | set[str]) -> list[str]: - file_set = set(file_list) - if os.path.exists(file_name): - file_set.add(file_name) - else: - log.critical(f"Missing required file: {file_name}") - return list(file_set) - - def __generate_execution_list(self) -> list[str]: - """ - Generate an execution list of scripts based on the specified action. - - This function dynamically creates a list of scripts to be executed by filtering and selecting - scripts based on the global ACTION variable. It supports different execution modes: - - 'minimal': A predefined set of lightweight scripts - - 'nopy': PowerShell and script-based scripts without Python - - 'modded': Includes scripts from the MODS directory - - 'depth': Comprehensive script execution with data mining and logging scripts - - 'vulnscan_ai': Vulnerability scanning script only - - Returns: - list[str]: A list of script file paths to be executed, filtered and modified based on the current action. - - Raises: - ValueError: Implicitly if a script file cannot be removed from the initial list. - - Notes: - - Removes sensitive or unnecessary scripts from the initial file list - - Logs the final execution list for debugging purposes - - Warns users about potential long execution times for certain actions - """ - execution_list = get.list_of_files( - ".", - only_extensions=(".py", ".exe", ".ps1", ".bat"), - exclude_files=["Logicytics.py"], - exclude_dirs=["logicytics", "SysInternal_Suite"], - ) - files_to_remove = { - "sensitive_data_miner.py", - "dir_list.py", - "tree.ps1", - "vulnscan.py", - "event_log.py", - } - execution_list = [ - file for file in execution_list if file not in files_to_remove - ] - - if ACTION == "minimal": - execution_list = [ - "cmd_commands.py", - "registry.py", - "tasklist.py", - "wmic.py", - "netadapter.ps1", - "property_scraper.ps1", - "window_feature_miner.ps1", - "event_log.py", - ] - - elif ACTION == "nopy": - execution_list = [ - "browser_miner.ps1", - "netadapter.ps1", - "property_scraper.ps1", - "window_feature_miner.ps1", - "tree.ps1", - ] - - elif ACTION == "modded": - # Add all files in MODS to execution list - execution_list = get.list_of_files( - "../MODS", - only_extensions=(".py", ".exe", ".ps1", ".bat"), - append_file_list=execution_list, - exclude_files=["Logicytics.py"], - exclude_dirs=["logicytics", "SysInternal_Suite"], - ) - - elif ACTION == "depth": - log.warning( - "This flag will use clunky and huge scripts, and so may take a long time, but reap great rewards." - ) - files_to_append = { - "sensitive_data_miner.py", - "dir_list.py", - "tree.ps1", - "event_log.py", - } - for file in files_to_append: - execution_list = self.__safe_append(file, execution_list) - log.warning("This flag will use threading!") - - elif ACTION == "vulnscan_ai": - # Only vulnscan detector - if os.path.exists("vulnscan.py"): - execution_list = ["vulnscan.py"] - else: - log.critical("Vulnscan is missing...") - exit(1) - - if len(execution_list) == 0: - log.critical( - "Nothing is in the execution list.. This is due to faulty code or corrupted Logicytics files!" - ) - exit(1) - - log.debug(f"Execution list length: {len(execution_list)}") - log.debug(f"The following will be executed: {execution_list}") - return execution_list - - @staticmethod - def __script_handler(script: str) -> tuple[str, Exception | None]: - """ - Executes a single script and logs the result, capturing any exceptions that occur during execution. - - Parameters: - script (str): The path to the script to be executed - """ - log.debug(f"Executing {script}") - try: - log.execution(execute.script(script)) - log.info(f"{script} executed successfully") - return script, None - except Exception as err: - log.error(f"Error executing {script}: {err}") - return script, err - - def handler(self): - """Executes the scripts in the execution list based on the action.""" - log.info("Starting Logicytics...") - - if ACTION == "threaded" or ACTION == "depth": - self.__threaded() - elif ACTION == "performance_check": - self.__performance() - else: - self.__default() - - def __threaded(self): - """Executes scripts in parallel using threading.""" - log.debug("Using threading") - with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor: - futures = { - executor.submit(self.__script_handler, script): script - for script in self.execution_list - } - - for future in as_completed(futures): - script = futures[future] - try: - result, error = future.result() - if error: - log.error(f"Failed to execute {script}: {error}") - else: - log.debug(f"Completed {script}") - except Exception as e: - log.error(f"Thread crashed while executing {script}: {e}") - - def __default(self): - """Executes scripts sequentially.""" - try: - for script in self.execution_list: - result, error = self.__script_handler(script) - if error: - log.error(f"Failed to execute {script}") - else: - log.debug(f"Completed {script}") - except UnicodeDecodeError as e: - log.error(f"Error in script execution (Unicode): {e}") - except Exception as e: - log.error(f"Error in script execution: {e}") - - def __performance(self): - """Checks performance of each script.""" - if DEBUG.lower() != "debug": - log.warning("Advised to turn on DEBUG logging!!") - - execution_times = [] - - for file in range(len(self.execution_list)): - gc.collect() - start_time = datetime.now() - log.execution(execute.script(self.execution_list[file])) - end_time = datetime.now() - elapsed_time = end_time - start_time - execution_times.append((self.execution_list[file], elapsed_time)) - log.info(f"{self.execution_list[file]} executed in {elapsed_time}") - - table = PrettyTable() - table.field_names = ["Script", "Execution Time"] - for script, elapsed_time in execution_times: - table.add_row([script, elapsed_time]) - - try: - with open( - f"../ACCESS/LOGS/PERFORMANCE/Performance_Summary_" - f"{datetime.now().strftime('%Y-%m-%d_%H-%M-%S')}.txt", - "w", - ) as f: - f.write(table.get_string()) - f.write("\nNote: This test only measures execution time.\n") - log.info( - "Performance check complete! Performance log found in ACCESS/LOGS/PERFORMANCE" - ) - except Exception as e: - log.error(f"Error writing performance log: {e}") - - -class SpecialAction: - @staticmethod - def update() -> tuple[str, str]: - """ - Updates the repository by pulling the latest changes from the remote repository. - - This function navigates to the parent directory, pulls the latest changes using Git, - and then returns to the current working directory. - - Returns: - str: The output from the git pull command. - """ - # Check if git command is available - try: - if ( - subprocess.run(["git", "--version"], capture_output=True).returncode - != 0 - ): - return "Git is not installed or not available in the PATH.", "error" - except FileNotFoundError: - return "Git is not installed or not available in the PATH.", "error" - - # Check if the project is a git repository - try: - if not os.path.exists(os.path.join(os.getcwd(), "../.git")): - return ( - "This project is not a git repository. The update flag uses git.", - "error", - ) - except Exception as e: - return f"Error checking for git repository: {e}", "error" - - current_dir = os.getcwd() - parent_dir = os.path.dirname(current_dir) - os.chdir(parent_dir) - output = subprocess.run(["git", "pull"], capture_output=True).stdout.decode() - os.chdir(current_dir) - return output, "info" - - @staticmethod - def execute_new_window(file_path: str): - """ - Execute a Python script in a new command prompt window. - - This function launches the specified Python script in a separate command prompt window, waits for its completion, and then exits the current process. - - Parameters: - file_path (str): The relative path to the Python script to be executed, - which will be resolved relative to the current script's directory. - - Side Effects: - - Opens a new command prompt window - - Runs the specified Python script - - Terminates the current process after script execution - - Raises: - FileNotFoundError: If the specified script path does not exist - subprocess.SubprocessError: If there are issues launching the subprocess - """ - sr_current_dir = os.path.dirname(os.path.abspath(__file__)) - sr_script_path = os.path.join(sr_current_dir, file_path) - sr_process = subprocess.Popen( - ["cmd.exe", "/c", "start", sys.executable, sr_script_path] - ) - sr_process.wait() - exit(0) - - -def get_flags(): - """ - Retrieves action and sub-action flags from the Flag module and sets global variables. - - This function extracts the current action and sub-action from the Flag module, setting global - ACTION and SUB_ACTION variables. It logs the retrieved values for debugging and tracing purposes. - - No parameters. - - Side effects: - - Sets global variables ACTION and SUB_ACTION - - Logs debug information about current action and sub-action - """ - global ACTION, SUB_ACTION - # Get flags_list - ACTION, SUB_ACTION = flag.data() - log.debug(f"Action: {ACTION}") - log.debug(f"Sub-Action: {SUB_ACTION}") - - -def handle_special_actions(): - """ - Handles special actions based on the current action flag. - - This function performs specific actions depending on the global `ACTION` variable: - - For "debug": Opens the debug menu by executing '_debug.py' - - For "dev": Opens the developer menu by executing '_dev.py' - - For "update": Updates the repository using Health.update() method - - For "restore": Displays a warning and opens the backup location - - For "backup": Creates backups of the CODE and MODS directories - - Side Effects: - - Logs informational, debug, warning, or error messages - - May execute external Python scripts - - May open file locations - - May terminate the program after completing special actions - - Raises: - SystemExit: Exits the program after completing certain special actions - """ - # Special actions -> Quit - if ACTION == "debug": - log.info("Opening debug menu...") - SpecialAction.execute_new_window("_debug.py") - - messages = check.sys_internal_zip() - if messages: - # If there are messages, log them with debug - log.debug(messages) - - if ACTION == "dev": - log.info("Opening developer menu...") - SpecialAction.execute_new_window("_dev.py") - - if ACTION == "update": - log.info("Updating...") - message, log_type = SpecialAction.update() - log.string(message, log_type) - if log_type == "info": - log.info("Update complete!") - else: - log.error("Update failed!") - input("Press Enter to exit...") - exit(0) - - if ACTION == "usage": - flag.Match.generate_summary_and_graph() - input("Press Enter to exit...") - exit(1) - - -def check_privileges(): - """ - Checks if the script is running with administrative privileges and handles UAC (User Account Control) settings. - - This function verifies if the script has admin privileges. If not, it either logs a warning (in debug mode) or - prompts the user to run the script with admin privileges and exits. It also checks if UAC is enabled and logs - warnings accordingly. - - Raises: - SystemExit: If the script is not running with admin privileges and not in debug mode. - - Notes: - - Requires the `Check` module with `admin()` and `uac()` methods - - Depends on global `DEBUG` configuration variable - - Logs warnings or critical messages based on privilege and UAC status - """ - if not check.admin(): - if DEBUG == "DEBUG": - log.warning( - "Running in debug mode, continuing without admin privileges - This may cause issues" - ) - else: - log.critical( - "Please run this script with admin privileges - To ignore this message, run with DEBUG in config" - ) - input("Press Enter to exit...") - exit(1) - - if check.uac(): - log.warning( - "UAC is enabled, this may cause issues - Please disable UAC if possible" - ) - - -class ZIP: - @classmethod - def files(cls): - """Zips generated files based on the action.""" - if ACTION == "modded": - cls.__and_log("..\\MODS", "MODS") - cls.__and_log(".", "CODE") - - @staticmethod - def __and_log(directory: str, name: str): - log.debug( - f"Zipping directory '{directory}' with name '{name}' under action '{ACTION}'" - ) - # noinspection PyUnreachableCode - zip_values = file_management.Zip.and_hash( - directory, - name, - ACTION - if ACTION is not None - else f"ERROR_NO_ACTION_SPECIFIED_{datetime.now().isoformat()}", - ) - if isinstance(zip_values, str): - log.error(zip_values) - else: - zip_loc, hash_loc = zip_values - log.info(zip_loc) - log.debug(hash_loc) - - -def handle_sub_action(): - """ - Handles sub-actions based on the provided sub_action flag. - - This function checks the value of the `sub_action` variable and performs - corresponding sub-actions such as shutting down or rebooting the system. - """ - log.info("Completed successfully!") - log.newline() - # Handle sub actions for all actions except performance check - if ACTION != "performance_check": - if SUB_ACTION == "shutdown": - subprocess.call("shutdown /s /t 3", shell=False) - elif SUB_ACTION == "reboot": - subprocess.call("shutdown /r /t 3", shell=False) - - -@log.function -def Logicytics(): - """ - Orchestrates the complete Logicytics workflow, managing script execution, system actions, and user interactions. - - This function serves as the primary entry point for the Logicytics utility, coordinating a series of system-level operations: - - Retrieves command-line configuration flags - - Processes special actions - - Verifies system privileges - - Executes targeted scripts - - Compresses generated output files - - Handles final system sub-actions - - Provides a graceful exit mechanism - - Performs actions sequentially without returning a value, designed to be the main execution flow of the Logicytics utility. - """ - # Get flags_list and configs - get_flags() - # Check for special actions - handle_special_actions() - # Check for privileges and errors - check_privileges() - # Execute scripts - ExecuteScript().handler() - # Zip generated files - ZIP.files() - # Finish with sub actions - handle_sub_action() - # Finish - input("Press Enter to exit...") - - -if __name__ == "__main__": - try: - Logicytics() - except KeyboardInterrupt: - log.warning( - "Force shutdown detected! Some temporary files might be left behind." - ) - log.warning("Next time, let the program finish naturally for complete cleanup.") - # Emergency cleanup - zip generated files - ZIP.files() - exit(0) -else: - log.error("This script cannot be imported!") - exit(1) diff --git a/CODE/SysInternal_Suite/SysInternal_Suite.zip b/CODE/SysInternal_Suite/SysInternal_Suite.zip deleted file mode 100644 index 4d310946..00000000 Binary files a/CODE/SysInternal_Suite/SysInternal_Suite.zip and /dev/null differ diff --git a/CODE/_debug.py b/CODE/_debug.py deleted file mode 100644 index 87cc8a14..00000000 --- a/CODE/_debug.py +++ /dev/null @@ -1,278 +0,0 @@ -from __future__ import annotations - -import configparser -import os -import platform -import sys -import time - -import psutil -import requests - -from logicytics import Log, DEBUG, VERSION, check, config, get - -log_path = os.path.join( - os.path.dirname(os.path.dirname(os.path.abspath(__file__))), - "ACCESS\\LOGS\\DEBUG\\DEBUG.log", -) -log = Log( - { - "log_level": DEBUG, - "filename": log_path, - "truncate_message": False, - "delete_log": True, - } -) -url = config.get("System Settings", "config_url") - - -class VersionManager: - @staticmethod - def parse_version(version: str) -> tuple[int, int, int | str, str]: - """ - Parses a version string into a tuple (major, minor, patch, type). - """ - try: - if version.startswith("snapshot-"): - parts = version.split("-")[1].split(".") - major, minor = map(int, parts[:2]) - patch = parts[2] if len(parts) > 2 else "0" - return major, minor, patch, "snapshot" - else: - return tuple(map(int, version.split("."))) + ("release",) - except Exception as e: - log.error(f"Failed to parse version: {e}") - return 0, 0, 0, "error" - - -class FileManager: - @staticmethod - def check_required_files(directory: str, required_files: list[str]): - """ - Checks if all required files are present in the directory and its subdirectories. - """ - try: - log.debug(f"Checking directory: {directory}") - if not os.path.exists(directory): - log.error(f"Directory {directory} does not exist.") - return - - # Use get.list_of_files to retrieve files, excluding specified files, dirs, and extensions - actual_files = get.list_of_files( - directory, - exclude_files=[ - "logicytics/User_History.json.gz", - "logicytics/User_History.json", - ], - exclude_dirs=["SysInternal_Suite"], - exclude_extensions=[".pyc"], - ) - actual_files = [ - f.replace("\\", "/").replace('"', "") for f in actual_files - ] # Normalize paths - - log.debug(f"Actual files found: {actual_files}") - # Strip quotes and normalize paths for comparison - normalized_required_files = [ - required_file.strip() - .replace("\\", "/") - .replace('"', "") # Remove quotes and normalize paths - for required_file in required_files - ] - - # Compare files - missing_files, extra_files = FileManager.compare_files( - actual_files, normalized_required_files - ) - - if missing_files: - log.error(f"Missing files: {', '.join(missing_files)}") - if extra_files: - log.warning(f"Extra files found: {', '.join(extra_files)}") - if not missing_files and not extra_files: - log.info("All required files are present.") - except Exception as e: - log.error(f"Unexpected error during file check: {e}") - - @staticmethod - def compare_files( - actual_files: list[str], required_files: list[str] - ) -> tuple[list[str], list[str]]: - """ - Compares actual and required files, returning missing and extra files. - """ - missing_files = [file for file in required_files if file not in actual_files] - extra_files = [file for file in actual_files if file not in required_files] - return missing_files, extra_files - - -class SysInternalManager: - @staticmethod - def check_binaries(path: str): - """ - Checks the SysInternal Binaries in the given directory. - """ - try: - if not os.path.exists(path): - raise FileNotFoundError("Directory does not exist") - - contents = os.listdir(path) - log.debug(str(contents)) - - has_zip = any(file.endswith(".zip") for file in contents) - has_exe = any(file.endswith(".exe") for file in contents) - - if any(file.endswith(".ignore") for file in contents): - log.warning("A `.sys.ignore` file was found - Ignoring") - elif has_zip and not has_exe: - log.error("Only zip files - Missing EXEs due to no `ignore` file") - elif has_zip and has_exe: - log.info("Both zip and exe files - All good") - else: - log.error( - "SysInternal Binaries Not Found: Missing Files - Corruption detected" - ) - except Exception as e: - log.error(f"Unexpected error: {e}") - - -class SystemInfoManager: - @staticmethod - def cpu_info() -> tuple[str, str, str]: - """ - Retrieves CPU details. - """ - return ( - f"CPU Architecture: {platform.machine()}", - f"CPU Vendor ID: {platform.system()}", - f"CPU Model: {platform.release()} {platform.version()}", - ) - - @staticmethod - def python_version(): - """ - Checks the current Python version against recommended version ranges and logs the result. - """ - version = sys.version.split()[0] - MIN_VERSION = (3, 11) - MAX_VERSION = (3, 14) - try: - major, minor = map(int, version.split(".")[:2]) - if MIN_VERSION <= (major, minor) < MAX_VERSION: - if (major, minor) == MIN_VERSION: - log.info( - f"Python Version: {version} - Perfect (mainly tested on 3.11.x)" - ) - else: - log.info(f"Python Version: {version} - Supported") - elif (major, minor) < MIN_VERSION: - log.warning(f"Python Version: {version} - Recommended: 3.11.x") - else: - log.error(f"Python Version: {version} - Incompatible") - except Exception as e: - log.error(f"Failed to parse Python Version: {e}") - - -class ConfigManager: - @staticmethod - def get_online_config() -> dict | None: - """ - Retrieves configuration data from a remote repository. - """ - try: - _config = configparser.ConfigParser() - _config.read_string(requests.get(url, timeout=15).text) - return _config - except requests.exceptions.RequestException as e: - log.error(f"Connection error: {e}") - return None - - -class HealthCheck: - @staticmethod - def check_versions(local_version: str, remote_version: str): - """ - Compares local and remote versions. - """ - local_version_tuple = VersionManager.parse_version(local_version) - remote_version_tuple = VersionManager.parse_version(remote_version) - - if "error" in local_version_tuple or "error" in remote_version_tuple: - log.error("Version parsing error.") - return - - try: - if "snapshot" in local_version_tuple or "snapshot" in remote_version_tuple: - log.warning("Snapshot versions are unstable.") - - if local_version_tuple == remote_version_tuple: - log.info(f"Version is up to date. Your Version: {local_version}") - elif local_version_tuple > remote_version_tuple: - log.warning( - "Version is ahead of the repository. " - f"Your Version: {local_version}, " - f"Repository Version: {remote_version}" - ) - else: - log.error( - "Version is behind the repository. " - f"Your Version: {local_version}, Repository Version: {remote_version}" - ) - except Exception as e: - log.error(f"Version comparison error: {e}") - - -@log.function -def debug(): - """ - Executes a comprehensive system debug routine, performing various checks and logging system information. - """ - # Online Configuration Check - _config = ConfigManager.get_online_config() - if _config: - HealthCheck.check_versions(VERSION, _config["System Settings"]["version"]) - - # File Integrity Check - required_files = _config["System Settings"].get("files", "").split(",") - FileManager.check_required_files(".", required_files) - - # SysInternal Binaries Check - SysInternalManager.check_binaries("SysInternal_Suite") - - # System Checks - log.info("Admin privileges found") if check.admin() else log.warning( - "Admin privileges not found" - ) - log.info("UAC enabled") if check.uac() else log.warning("UAC disabled") - log.info(f"Execution path: {psutil.__file__}") - log.info(f"Global execution path: {sys.executable}") - log.info(f"Local execution path: {sys.prefix}") - log.info( - "Running in a virtual environment" - if sys.prefix != sys.base_prefix - else "Not running in a virtual environment" - ) - log.info( - "Execution policy is unrestricted" - if check.execution_policy() - else "Execution policy is restricted" - ) - - # Python Version Check - SystemInfoManager.python_version() - - # CPU Info - for info in SystemInfoManager.cpu_info(): - log.info(info) - - # Final Debug Status - log.info(f"Log Level: {DEBUG}") - - -if __name__ == "__main__": - try: - debug() - except Exception as err: - log.error(f"Failed to execute debug routine: {err}") - time.sleep(0.5) - input("Press Enter to exit...") diff --git a/CODE/_dev.py b/CODE/_dev.py deleted file mode 100644 index b9863125..00000000 --- a/CODE/_dev.py +++ /dev/null @@ -1,223 +0,0 @@ -from __future__ import annotations - -import os -import re -import subprocess - -import configobj - -from logicytics import log, get, CURRENT_FILES, VERSION - - -def color_print(text, color="reset", is_input=False) -> None | str: - colors = { - "reset": "\033[0m", - "red": "\033[31m", - "green": "\033[32m", - "yellow": "\033[33m", - "cyan": "\033[36m", - } - - color_code = colors.get(color.lower(), colors["reset"]) - if is_input: - return input(f"{color_code}{text}{colors['reset']}") - print(f"{color_code}{text}{colors['reset']}") - return None - - -def _update_ini_file(filename: str, new_data: list | str, key: str) -> None: - """ - Updates an INI file with a new array of current files or version. - Args: - filename (str): The path to the INI file to be updated. - new_data (list | str): The list of current files or the new version to be written to the INI file. - key (str): The key in the INI file to be updated. - Returns: - None - """ - try: - config = configobj.ConfigObj( - filename, encoding="utf-8", write_empty_values=True - ) - - if key == "files": - config["System Settings"][key] = ", ".join(new_data) - elif key == "version": - config["System Settings"][key] = new_data - else: - color_print(f"[!] Invalid key: {key}", "yellow") - return - - config.write() - except FileNotFoundError: - color_print("[x] INI file not found", "red") - except configobj.ConfigObjError as e: - color_print(f"[x] Parsing INI file failed: {e}", "red") - except Exception as e: - color_print(f"[x] {e}", "red") - - -def _prompt_user( - question: str, file_to_open: str = None, special: bool = False -) -> bool: - """ - Prompts the user with a yes/no question and optionally opens a file. - - Parameters: - question (str): The question to be presented to the user. - file_to_open (str, optional): Path to a file that will be opened if the user does not respond affirmatively. - special (bool, optional): Flag to suppress the default reminder message when the user responds negatively. - - Returns: - bool: True if the user responds with 'yes' or 'Y', False otherwise. - - Raises: - Exception: Logs any unexpected errors during user interaction. - - Notes: - - Uses subprocess to open files on Windows systems - - Case-insensitive input handling for 'yes' responses - - Provides optional file opening and reminder messaging - """ - try: - answer = color_print(f"[?] {question} (y)es or (n)o:- ", "cyan", is_input=True) - if not (answer.lower() == "yes" or answer.lower() == "y"): - if file_to_open: - subprocess.run(["start", file_to_open], shell=True) - if not special: - color_print( - "[x] Please ensure you fix the issues/problem and try again with the checklist.", - "red", - ) - return False - return True - except Exception as e: - color_print(f"[x] {e}", "red") - return None - - -def _perform_checks() -> bool: - """ - Performs a series of user prompts for various checks. - - Returns: - bool: True if all checks are confirmed by the user, False otherwise. - """ - checks = [ - ("Have you read the required contributing guidelines?", "..\\CONTRIBUTING.md"), - ("Have you made files you don't want to be run start with '_'?", "."), - ("Have you added the file to CODE dir?", "."), - ("Have you added docstrings and comments?", "..\\CONTRIBUTING.md"), - ("Is each file containing around 1 main feature?", "..\\CONTRIBUTING.md"), - ] - - for question, file_to_open in checks: - if not _prompt_user(question, file_to_open): - return False - return True - - -def _handle_file_operations() -> None: - """ - Handles file operations and logging for added, removed, and normal files. - """ - EXCLUDE_FILES = [ - "logicytics\\User_History.json.gz", - "logicytics\\User_History.json", - ] - files = get.list_of_files( - ".", - exclude_files=EXCLUDE_FILES, - exclude_dirs=["SysInternal_Suite"], - exclude_extensions=[".pyc"], - ) - added_files, removed_files, normal_files = [], [], [] - clean_files_list = [file.replace('"', "") for file in CURRENT_FILES] - - files_set = set(os.path.abspath(f) for f in files) - clean_files_set = set(os.path.abspath(f) for f in clean_files_list) - - for file in files_set: - if file in clean_files_set and file not in EXCLUDE_FILES: - normal_files.append(file) - elif file not in clean_files_set and file not in EXCLUDE_FILES: - added_files.append(file) - - for file in clean_files_set: - if file not in files_set and file not in EXCLUDE_FILES: - removed_files.append(file) - - print("\n".join([f"\033[92m+ {file}\033[0m" for file in added_files])) # Green + - print("\n".join([f"\033[91m- {file}\033[0m" for file in removed_files])) # Red - - print("\n".join([f"* {file}" for file in normal_files])) - - if not _prompt_user("Does the list above include your added files?"): - color_print("[x] Something went wrong! Please contact support.", "red") - return - - max_attempts = 10 - attempts = 0 - _update_ini_file("config.ini", files, "files") - - while True: - version = color_print( - f"[?] Enter the new version of the project (Old version is {VERSION}): ", - "cyan", - is_input=True, - ) - - if re.match(r"^(0|[1-9]\d*)\.(0|[1-9]\d*)\.(0|[1-9]\d*)$", version): - _update_ini_file("config.ini", version, "version") - break - attempts += 1 - if attempts >= max_attempts: - color_print( - "[x] Maximum attempts reached. Please run the script again.", "red" - ) - exit() - else: - color_print( - "[!] Please enter a valid version number (e.g., 1.2.3)", "yellow" - ) - color_print(f"[!] {max_attempts - attempts} attempts remaining", "yellow") - - color_print( - "\n[-] Great Job! Please tick the box in the GitHub PR request for completing steps in --dev", - "green", - ) - - -@log.function -def dev_checks() -> None: - """ - Performs comprehensive developer checks to ensure code quality and project guidelines compliance. - - This function guides developers through a series of predefined checks, validates file additions, - and updates project configuration. It performs the following key steps: - - Verify adherence to contributing guidelines - - Check file naming conventions - - Validate file placement - - Confirm docstring and comment coverage - - Assess feature modularity - - Categorize and display file changes - - Update project configuration file - - Raises: - None: Returns None if any check fails or an error occurs during the process. - - Side Effects: - - Creates necessary directories - - Prompts user for multiple confirmations - - Prints file change lists with color coding - - Updates configuration file with current files and version - - Logs warnings or errors during the process - """ - if not _perform_checks(): - return - _handle_file_operations() - - -if __name__ == "__main__": - dev_checks() - # Wait for the user to press Enter to exit the program - input("\n[*] Press Enter to exit the program... ") diff --git a/CODE/bluetooth_details.py b/CODE/bluetooth_details.py deleted file mode 100644 index fb28dcdb..00000000 --- a/CODE/bluetooth_details.py +++ /dev/null @@ -1,144 +0,0 @@ -from __future__ import annotations - -import json -import subprocess -from typing import TextIO - -from logicytics import log - - -@log.function -def get_bluetooth_device_details(): - """ - Retrieves and logs detailed information about Bluetooth devices on the system. - - Executes a PowerShell query to collect Bluetooth device details and writes the information to a text file. - The function performs the following key actions: - - Logs the start of the device information retrieval process - - Queries Bluetooth devices using an internal helper function - - Writes device details to 'Bluetooth Info.txt' if devices are found - - Returns: - None: No return value; results are written to a file and logged - """ - log.info("Fetching detailed info for Bluetooth devices...") - devices = _query_bluetooth_devices() - if devices: - _write_device_info_to_file(devices, "Bluetooth Info.txt") - - -def _query_bluetooth_devices() -> bool | list[dict[str, str]]: - """ - Queries the system for Bluetooth devices using PowerShell commands. - - Executes a PowerShell command to retrieve detailed information about Bluetooth devices connected to the system. - The function handles potential errors during command execution and JSON parsing, providing fallback values - for device information. - - Returns: - bool | list[dict[str, str]]: A list of device information dictionaries or False if an error occurs. - Each dictionary contains details such as Name, Device ID, Description, Manufacturer, Status, and PNP Device ID. - - Raises: - No direct exceptions are raised. Errors are logged and the function returns False. - - Example: - devices = _query_bluetooth_devices() - if devices: - for device in devices: - print(device['Name']) - """ - try: - # Run PowerShell command to get Bluetooth devices - command = ( - "Get-PnpDevice | Where-Object { $_.FriendlyName -like '*Bluetooth*' } | " - "Select-Object FriendlyName, DeviceID, Description, Manufacturer, Status, PnpDeviceID | " - "ConvertTo-Json -Depth 3" - ) - result = subprocess.run(["powershell", "-Command", command], - capture_output=True, text=True, check=True) - devices = json.loads(result.stdout) - except subprocess.CalledProcessError as e: - log.error(f"Failed to query Bluetooth devices with command '{command}': {e}") - return False - except json.JSONDecodeError as e: - log.error(f"Failed to parse device information: {e}") - return False - - if isinstance(devices, dict): - devices = [devices] # Handle single result case - - device_info_list = [] - for device in devices: - FALLBACK_MSG = 'Unknown (Fallback due to failed Get request)' - device_info = { - 'Name': device.get('FriendlyName', FALLBACK_MSG), - 'Device ID': device.get('DeviceID', FALLBACK_MSG), - 'Description': device.get('Description', FALLBACK_MSG), - 'Manufacturer': device.get('Manufacturer', FALLBACK_MSG), - 'Status': device.get('Status', FALLBACK_MSG), - 'PNP Device ID': device.get('PnpDeviceID', FALLBACK_MSG) - } - log.debug(f"Retrieved device: {device_info['Name']}") - device_info_list.append(device_info) - - return device_info_list - - -def _write_device_info_to_file(devices: list[dict[str, str]], filename: str): - """ - Writes the details of Bluetooth devices to a specified file. - - Args: - devices (list): A list of dictionaries containing Bluetooth device information. - filename (str): The path and name of the file where device details will be written. - - Raises: - IOError: If there is an error opening or writing to the specified file. - OSError: If there are file system related issues during file writing. - - Notes: - - Uses UTF-8 encoding for file writing - - Logs an error if file writing fails - - Calls _write_single_device_info() for each device in the list - """ - try: - with open(filename, "w", encoding="UTF-8") as file: - for device_info in devices: - _write_single_device_info(file, device_info) - log.info(f"Successfully wrote device details to '{filename}'") - except Exception as e: - log.error(f"Failed to write device information to file: {e}") - - -def _write_single_device_info(file: TextIO, device_info: dict[str, str]): - """ - Writes detailed information for a single Bluetooth device to the specified file. - - Parameters: - file (TextIO): An open file object to which device information will be written. - device_info (dict): A dictionary containing key-value pairs of Bluetooth device attributes. - - Writes the device name followed by all other device attributes, with each device's information separated by a blank line. Uses `.get()` method to provide a fallback 'Unknown' value if the device name is missing. - - Example: - If device_info is {'Name': 'Wireless Headset', 'Address': '00:11:22:33:44:55', 'Connected': 'True'} - The file will contain: - Name: Wireless Headset - Address: 00:11:22:33:44:55 - Connected: True - - If no name is provided: - Name: Unknown - Address: 00:11:22:33:44:55 - Connected: True - """ - file.write(f"Name: {device_info.get('Name', 'Unknown')}\n") - for key, value in device_info.items(): - if key != 'Name': - file.write(f" {key}: {value}\n") - file.write("\n") # Separate devices with a blank line - - -if __name__ == "__main__": - get_bluetooth_device_details() diff --git a/CODE/bluetooth_logger.py b/CODE/bluetooth_logger.py deleted file mode 100644 index c6d18819..00000000 --- a/CODE/bluetooth_logger.py +++ /dev/null @@ -1,179 +0,0 @@ -import datetime -import re -import subprocess -from typing import LiteralString - -from logicytics import log - - -# Utility function to log data to a file -def save_to_file(filename: str, section_title: str, data: str): - """ - Appends data to a file with a section title. - - Args: - filename (str): Path to the file where data will be written. Must be a valid file path. - section_title (str): Title describing the section being added to the file. - data (str or list): Content to be written. Accepts either a single string or a list of strings. - - Raises: - IOError: If the file cannot be opened or written to due to permission or path issues. - Exception: For any unexpected errors during file writing. - - Notes: - - Uses UTF-8 encoding for file writing - - Adds decorative section separators around the content - - Automatically handles single string or list of strings input - - Logs any errors encountered during file writing - """ - try: - with open(filename, 'a', encoding='utf-8') as file: - file.write(f"\n{'=' * 50}\n{section_title}\n{'=' * 50}\n") - file.write(f"{data}\n" if isinstance(data, str) else "\n".join(data) + "\n") - file.write(f"{'=' * 50}\n") - except Exception as err: - log.error(f"Error writing to file {filename}: {err}") - - -# Utility function to run PowerShell commands -def run_powershell_command(command: str) -> None | list[LiteralString]: - """ - Runs a PowerShell command and returns the output as a list of lines. - - Args: - command (str): The PowerShell command to execute. - - Returns: - list: A list of strings representing each line of the command output. - Returns an empty list if the command execution fails or an exception occurs. - - Raises: - subprocess.CalledProcessError: If the PowerShell command returns a non-zero exit status. - Exception: For any unexpected errors during command execution. - - Notes: - - Uses subprocess.run() with capture_output=True to capture command output - - Logs errors for failed commands or exceptions - - Splits command output into lines for easier processing - """ - try: - result = subprocess.run(["powershell", "-Command", command], capture_output=True, text=True) - if result.returncode != 0: - log.error(f"PowerShell command failed with return code {result.returncode}") - return [] - return result.stdout.splitlines() - except Exception as err: - log.error(f"Error running PowerShell command: {err}") - return [] - - -# Unified parsing function for PowerShell output -def parse_output(lines: list[LiteralString], regex: str, group_names: list[str]): - """ - Parses the output lines using the provided regex and group names. - - Parameters: - lines (list): A list of strings representing command output lines. - regex (str): Regular expression pattern to match each line. - group_names (list): List of group names to extract from matched regex. - - Returns: - list: Dictionaries containing extracted group names and their values. - - Raises: - Exception: If parsing the output encounters an unexpected error. - - Notes: - - Skips lines that do not match the provided regex pattern - - Logs debug messages for unrecognized lines - - Logs error if parsing fails completely - """ - results = [] - try: - for line in lines: - match = re.match(regex, line) - if match: - results.append({name: match.group(name) for name in group_names}) - else: - log.debug(f"Skipping unrecognized line: {line}") - return results - except Exception as err: - log.error(f"Parsing output failed: {err}") - - -# Function to get paired Bluetooth devices -def get_paired_bluetooth_devices() -> list[str]: - """ - Retrieves a list of paired Bluetooth devices with their names and MAC addresses. - - This function executes a PowerShell command to fetch Bluetooth devices with an "OK" status, - parses the output to extract device details, and attempts to retrieve MAC addresses from device IDs. - - Returns: - list: A list of formatted strings containing device names and MAC addresses. - Each string follows the format "Name: , MAC: ". - - Raises: - Exception: If there are issues running the PowerShell command or parsing the output. - """ - command = ( - 'Get-PnpDevice -Class Bluetooth | Where-Object { $_.Status -eq "OK" } | Select-Object Name, DeviceID' - ) - output = run_powershell_command(command) - log.debug(f"Raw PowerShell output for paired devices:\n{output}") - - devices = parse_output( - output, - regex=r"^(?P.+?)\s+(?P.+)$", - group_names=["Name", "DeviceID"] - ) - - # Extract MAC addresses - for device in devices: - mac_match = re.search(r"BLUETOOTHDEVICE_(?P[A-F0-9]{12})", device["DeviceID"], re.IGNORECASE) - device["MAC"] = mac_match.group("MAC") if mac_match else "Address Not Found" - - return [f"Name: {device['Name']}, MAC: {device['MAC']}" for device in devices] - - -# Function to log all Bluetooth data -@log.function -def log_bluetooth(): - """ - Logs comprehensive Bluetooth data including paired devices and system event logs. - - This function performs the following actions: - - Captures the current timestamp - - Retrieves and logs paired Bluetooth devices - - Collects Bluetooth connection/disconnection event logs - - Captures Bluetooth file transfer logs - - Saves all collected data to 'bluetooth_data.txt' - - The function uses internal utility functions to run PowerShell commands, parse outputs, and save results to a file. It provides a systematic approach to logging Bluetooth-related system information. - - Logs are saved with descriptive section titles, making the output easily readable and organized. If no data is found for a specific section, a default "No logs found" message is recorded. - - Note: - - Requires administrative or sufficient system permissions to access Windows event logs - - Logs are appended to the file, allowing historical tracking of Bluetooth events - """ - log.info("Starting Bluetooth data logging...") - filename = "bluetooth_data.txt" - timestamp = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") - save_to_file(filename, "Bluetooth Data Collection - Timestamp", timestamp) - - # Collect and log paired devices - log.info(f"Collecting paired devices...") - paired_devices = get_paired_bluetooth_devices() - section_title = "Paired Bluetooth Devices" - save_to_file(filename, section_title, paired_devices or ["No paired Bluetooth devices found."]) - log.debug(f"{section_title}: {paired_devices}") - - log.info("Finished Bluetooth data logging.") - - -if __name__ == "__main__": - try: - log_bluetooth() - except Exception as e: - log.error(f"Failed to log Bluetooth data: {e}") diff --git a/CODE/browser_miner.ps1 b/CODE/browser_miner.ps1 deleted file mode 100644 index f0be76b0..00000000 --- a/CODE/browser_miner.ps1 +++ /dev/null @@ -1,89 +0,0 @@ -# Define the list of source paths with placeholders -$sourcePaths = @( - "C:\Users\{}\AppData\Local\Microsoft\Edge\User Data\Default\Network", - "C:\Users\{}\AppData\Local\Google\Chrome\User Data\Default\Network", - "C:\Users\{}\AppData\Roaming\Mozilla\Firefox\Profiles", - "C:\Users\{}\AppData\Roaming\Opera Software\Opera Stable\Network", - "C:\Users\{}\AppData\Roaming\Opera Software\Opera GX Stable\Network", - 'C:\\Windows\\System32\\config', - 'C:\\Windows\\System32\\GroupPolicy', - 'C:\\Windows\\System32\\GroupPolicyUsers', - 'C:\\Windows\\System32\\winevt\\Logs' -) - -# Define the list of identifiers for renaming -$identifiers = @( - "Edge", - "Chrome", - "Firefox", - "OperaStable", - "OperaGXStable", - "SAM", - "SystemConfig", - "GroupPolicy", - "GroupPolicyUsers", - "WindowsEventLogs" -) - -# Get the current user's name -$currentUser = $env:USERNAME - -# Define the base directory for the destination -$baseDirectory = "Browser_Data" - -# Function to check if a path exists and is accessible -function Test-PathAndAccess($path) -{ - return Test-Path $path -PathType Container -ErrorAction SilentlyContinue -} - -# Loop through each source path -foreach ($sourcePath in $sourcePaths) -{ - # Replace the placeholder with the current user's name - $fullSourcePath = $sourcePath -replace '\{\}', $currentUser - - # Enhanced error checking for source path existence and accessibility - if (-not (Test-PathAndAccess $fullSourcePath)) - { - Write-Host "WARNING: Source path $fullSourcePath does not exist or cannot be accessed." - continue - } - - - # Extract the identifier from the source path using the corresponding Index from the $identifiers array - try - { - $index = [Array]::IndexOf($identifiers, $sourcePath.Split('\')[-1].Split('\\')[-1]) - $identifier = $identifiers[$index] - } - catch - { - Write-Host "ERROR: Failed to extract identifier from source path $fullSourcePath." - continue - } - - - # Define the destination path - $destinationPath = Join-Path -Path $baseDirectory -ChildPath "USER_$identifier" - - # Enhanced error checking for destination directory existence - if (-not (Test-PathAndAccess $destinationPath)) - { - New-Item -ItemType Directory -Path $destinationPath -Force | Out-Null - } - - # Attempt to copy the folder to the DATA directory and rename it - try - { - Copy-Item -Path $fullSourcePath -Destination $destinationPath -Recurse -Force -ErrorAction SilentlyContinue - # Print the success message to the console - Write-Host "INFO: Successfully copied $fullSourcePath to $destinationPath" - } - catch - { - # Detailed error handling - Write-Host "ERROR: An error occurred while copying $fullSourcePath to $destinationPath : $_" - exit - } -} diff --git a/CODE/cmd_commands.py b/CODE/cmd_commands.py deleted file mode 100644 index 3739dca7..00000000 --- a/CODE/cmd_commands.py +++ /dev/null @@ -1,31 +0,0 @@ -from logicytics import log, execute - - -@log.function -def command(file: str, commands: str, message: str, encoding: str = "UTF-8") -> None: - """ - Executes a command and writes the output to a file. - - Args: - file (str): The name of the file to write the command output to. - commands (str): The command to be executed. - message (str): A message to be logged. - encoding (str): The encoding to write the file in. - - Returns: - None - """ - log.info(f"Executing {message}") - try: - output = execute.command(commands) - with open(file, "w", encoding=encoding) as f: - f.write(output) - log.info(f"{message} Successful - {file}") - except Exception as e: - log.error(f"Error while getting {message}: {e}") - - -if __name__ == "__main__": - command("Drivers.txt", "driverquery /v", "Driver Query") - command("SysInfo.txt", "systeminfo", "System Info") - command("GPResult.txt", "GPResult /r", "GPResult", "windows-1252") diff --git a/CODE/config.ini b/CODE/config.ini deleted file mode 100644 index 8395bd55..00000000 --- a/CODE/config.ini +++ /dev/null @@ -1,114 +0,0 @@ -######################################################## -# The following settings are for Logicytics as a whole # -######################################################## - -[Settings] -# Would you like to enable debug mode? -# This will print out more information to the console, with prefix DEBUG -# This will not be logged however, and is useful for developers - This is different than the DEBUGGER itself -log_using_debug = false - -# Would you like for new logs to be created every execution? -# Or would you like to append to the same log file? -delete_old_logs = false - -# When using threading mode, you have the option to decide how many threads to use (workers) -# Uncomment and change the value to use a maximum amount of threads, -# otherwise keep it commented if you don't need a maximum limit -; max_workers = 10 - -# Logicytics will save preferences and history in a file, -# This is used by Flag.py, to suggest better flags -# Would you like this to happen? -# This is recommended, as it will improve the suggestions - Data will never be shared -save_preferences = true - -[System Settings] -# Do not play with these settings unless you know what you are doing -# Dev Mode allows a safe way to modify these settings!! -version = 3.6.0 -files = "bluetooth_details.py, bluetooth_logger.py, browser_miner.ps1, cmd_commands.py, config.ini, dir_list.py, dump_memory.py, encrypted_drive_audit.py, event_log.py, Logicytics.py, log_miner.py, media_backup.py, netadapter.ps1, network_psutil.py, packet_sniffer.py, property_scraper.ps1, registry.py, sensitive_data_miner.py, ssh_miner.py, sys_internal.py, tasklist.py, tree.ps1, usb_history.py, vulnscan.py, wifi_stealer.py, window_feature_miner.ps1, wmic.py, logicytics\Checks.py, logicytics\Config.py, logicytics\Execute.py, logicytics\FileManagement.py, logicytics\Flag.py, logicytics\Get.py, logicytics\Logger.py, logicytics\User_History.json.gz, vulnscan\Model_SenseMacro.4n1.pth" -# If you forked the project, change the USERNAME to your own to use your own fork as update material, -# I dont advise doing this however -config_url = https://raw.githubusercontent.com/DefinetlyNotAI/Logicytics/main/CODE/config.ini - -######################################################## -# The following settings are for specific modules # -######################################################## - -[Flag Settings] -# The minimum accuracy to suggest a flag, -# This is a percentage, and must be a float -# The default is 30.0, and is what we advise -# If the accuracy is below this, the flag will move to the next suggestion process -# The process is: difflib, then model, then history suggestions -# Make sure to keep between 0.0 and 100.0 -accuracy_min = 30.0 - -# This is the model to use to suggest flags, -# I advise to keep it as all-MiniLM-L6-v2 -# This is the best model for this task, and is lightweight -# The model MUST be a Sentence Transformer model -model_to_use = all-MiniLM-L6-v2 - -# Finally, should debug mode be enabled for the flag module? -# This will print out more information to the console, -# This is for the model itself, and is based on tqdm, it shows extra info on batches -# As well as more information on behind the scenes -model_debug = false - -################################################### - -[DumpMemory Settings] -# If the file size generated exceeds this limit, -# the file will be truncated with a message -# Put 0 to disable the limit - Limit is in MiB - int -file_size_limit = 0 -# Safety margin to check, it multiplies with the size limit -# This makes sure that after the file is created, there is still -# disk space left for other tasks, -# Make sure its above 1 or else it will fail -# Put 1 to disable the limit - Limit is in MiB - float -file_size_safety = 1.5 - -################################################### - -[NetWorkPsutil Settings] -# Total time this will take will be `sample_count * interval` - -# Number of samples to take for feature `measure network bandwidth usage` -# This is an integer, and should be 1 and above -sample_count = 5 -# Time between samples in seconds for feature `measure network bandwidth usage` -# This is a float, and should be above 0 -interval = 1.5 - -################################################### - -[PacketSniffer Settings] -# The interface to sniff packets on, keep it as WiFi for most cases -# Autocorrects between WiFi and Wi-Fi -interface = WiFi -# The number of packets to sniff, -# Must be greater than or equal to 1 - int -packet_count = 5000 -# The time to timeout the sniffing process only, -# Must be greater than or equal to 5 - int -timeout = 10 -# The maximum retry time for the whole process, -# Must be greater than or equal to 10 and timeout - int -max_retry_time = 30 - -################################################### - -[VulnScan Settings] -# Max characters of text from each file to analyze. Set an integer or None to disable truncation. -text_char_limit = None -# Max workers to be used, either integer or use "auto" to make it decide the best value -max_workers = auto -# Sensitivity threshold (0.0–1.0) for the model to flag content as sensitive -threshold = 0.6 -# Paths for required files -model = vulnscan/Model_SenseMacro.4n1.pth - -################################################## diff --git a/CODE/dir_list.py b/CODE/dir_list.py deleted file mode 100644 index c4495b79..00000000 --- a/CODE/dir_list.py +++ /dev/null @@ -1,70 +0,0 @@ -import os -from concurrent.futures import ThreadPoolExecutor - -from logicytics import log, execute - - -def run_command_threaded(directory: str, file: str, message: str, encoding: str = "UTF-8") -> None: - """ - Executes a PowerShell command to recursively list directory contents and writes the output to a specified file. - - Args: - directory (str): The target directory path to list contents from. - file (str): The output file path where directory contents will be appended. - message (str): A descriptive message for logging the operation. - encoding (str, optional): File writing encoding. Defaults to "UTF-8". - - Raises: - Exception: If command execution or file writing fails. - - Notes: - - Uses PowerShell's Get-ChildItem with recursive flag - - Appends output to the specified file - - Logs operation start and result/error - """ - log.info(f"Executing {message} for {directory}") - try: - safe_directory = directory.replace('"', '`"') # Escape quotes - command = f'powershell -NoProfile -Command "Get-ChildItem \\""{safe_directory}\\"" -Recurse"' - output = execute.command(command) - open(file, "a", encoding=encoding).write(output) - log.info(f"{message} Successful for {directory} - {file}") - except Exception as e: - log.error(f"Error while getting {message} for {directory}: {e}") - - -@log.function -def command_threaded(base_directory: str, file: str, message: str, encoding: str = "UTF-8") -> None: - """ - Concurrently lists contents of subdirectories within a base directory using thread pooling. - - Args: - base_directory (str): Root directory to explore and list subdirectories from. - file (str): Output file path to write directory listing results. - message (str): Descriptive logging message for the operation. - encoding (str, optional): File writing character encoding. Defaults to "UTF-8". - - Raises: - Exception: Logs and captures any errors during thread pool execution. - - Notes: - - Uses ThreadPoolExecutor for parallel directory content listing - - Processes each subdirectory concurrently - - Writes results to the specified file - - Handles potential errors during thread execution - """ - try: - with ThreadPoolExecutor(max_workers=min(32, os.cpu_count() * 4)) as executor: - subdirectories = [os.path.join(base_directory, d) for d in os.listdir(base_directory) if - os.path.isdir(os.path.join(base_directory, d))] - futures = [executor.submit(run_command_threaded, subdir, file, message, encoding) for subdir in - subdirectories] - for future in futures: - future.result() - except Exception as e: - log.error(f"Thread Pool Error: {e}") - - -if __name__ == "__main__": - log.warning("Running dir_list.py - This is very slow - We will use threading to speed it up") - command_threaded("C:\\", "Dir_Root.txt", "Root Directory Listing") diff --git a/CODE/dump_memory.py b/CODE/dump_memory.py deleted file mode 100644 index 360a27d8..00000000 --- a/CODE/dump_memory.py +++ /dev/null @@ -1,173 +0,0 @@ -import os -import platform -import struct -from datetime import datetime - -import psutil - -from logicytics import log, config - -# Constants from config with validation -LIMIT_FILE_SIZE = config.getint("DumpMemory Settings", "file_size_limit") # MiB -SAFETY_MARGIN = config.getfloat("DumpMemory Settings", "file_size_safety") # MiB -DUMP_DIR = config.get("DumpMemory Settings", "dump_directory", fallback="memory_dumps") - -if SAFETY_MARGIN < 1: - log.critical("Invalid Safety Margin Inputted - Cannot proceed with dump memory") - exit(1) - - -def capture_ram_snapshot(): - """ - Captures and logs the current system memory statistics to a file. - - Retrieves detailed information about RAM and swap memory usage using psutil. - Writes memory statistics in gigabytes to 'Ram_Snapshot.txt', including: - - Total RAM - - Used RAM - - Available RAM - - Total Swap memory - - Used Swap memory - - Free Swap memory - - Percentage of RAM used - - Logs the process and handles potential file writing errors. - - Raises: - IOError: If unable to write to the output file - Exception: For any unexpected errors during memory snapshot capture - """ - - def memory_helper(mem_var, flavor_text: str, use_free_rather_than_available: bool = False): - file.write(f"Total {flavor_text}: {mem_var.total / (1024 ** 3):.2f} GB\n") - file.write(f"Used {flavor_text}: {mem_var.used / (1024 ** 3):.2f} GB\n") - if use_free_rather_than_available: - file.write(f"Available {flavor_text}: {mem_var.free / (1024 ** 3):.2f} GB\n") - else: - file.write(f"Available {flavor_text}: {mem_var.available / (1024 ** 3):.2f} GB\n") - file.write(f"{flavor_text} Percent Usage: {mem_var.percent:.2f}%\n") - - log.info("Capturing RAM Snapshot...") - try: - memory = psutil.virtual_memory() - swap = psutil.swap_memory() - with open(os.path.join(DUMP_DIR, "Ram_Snapshot.txt"), "w", encoding="utf-8") as file: - memory_helper(memory, "RAM") - memory_helper(swap, "Swap Memory", use_free_rather_than_available=True) - except Exception as e: - log.error(f"Failed to capture RAM snapshot: {e}") - log.info("RAM Snapshot saved to Ram_Snapshot.txt") - - -def gather_system_info(): - """ - Gathers detailed system information and saves it to a file. - """ - log.info("Gathering system information...") - try: - sys_info = { - 'Architecture': platform.architecture(), - 'System': platform.system(), - 'Machine': platform.machine(), - 'Processor': platform.processor(), - 'Page Size (bytes)': struct.calcsize("P"), - 'CPU Count': psutil.cpu_count(), - 'CPU Frequency': psutil.cpu_freq().current if psutil.cpu_freq() else 'Unavailable', - 'Boot Time': datetime.fromtimestamp(psutil.boot_time()).strftime('%Y-%m-%d %H:%M:%S'), - } - except Exception as e: - log.error(f"Error gathering system information: {e}") - sys_info = {'Error': 'Failed to gather system information'} - try: - with open(os.path.join(DUMP_DIR, "SystemRam_Info.txt"), "w", encoding="utf-8") as file: - for key, value in sys_info.items(): - file.write(f"{key}: {value}\n") - except Exception as e: - log.error(f"Error writing system info to file: {e}") - log.info("System Information saved to SystemRam_Info.txt") - - -# Memory Dump -def memory_dump(): - """ - Performs a memory dump of the current process and saves it to a file. - """ - log.info("Creating basic memory dump scan...") - pid = os.getpid() - - try: - process = psutil.Process(pid) - dump_path = os.path.join(DUMP_DIR, "Ram_Dump.txt") - with open(dump_path, "wb") as dump_file: - total_size = 0 - - # Disk space safety check - required_space = LIMIT_FILE_SIZE * 1024 * 1024 * SAFETY_MARGIN - free_space = psutil.disk_usage(DUMP_DIR).free - if free_space < required_space: - log.error(f"Not enough disk space. Need at least {required_space / (1024 ** 2):.2f} MiB") - return - - for mem_region in process.memory_maps(grouped=False): - if 'r' not in mem_region.perms: - continue - - try: - start, end = (int(addr, 16) for addr in mem_region.addr.split('-')) \ - if '-' in mem_region.addr else (int(mem_region.addr, 16), - int(mem_region.addr, 16) + mem_region.rss) - except Exception as e: - log.warning(f"Invalid address format '{mem_region.addr}': {e}") - continue - - region_metadata = { - ' Start Address': hex(start), - ' End Address': hex(end), - ' Region Size (bytes)': end - start, - ' RSS (bytes)': mem_region.rss, - ' Permissions': mem_region.perms, - ' Path': mem_region.path, - ' Index': mem_region.index, - } - - try: - metadata_str = "Memory Region Metadata:\n" + "\n".join( - f"{key}: {value}" for key, value in region_metadata.items()) + "\n\n" - metadata_bytes = metadata_str.encode() - if (total_size + len(metadata_bytes) > LIMIT_FILE_SIZE * 1024 * 1024) and (LIMIT_FILE_SIZE != 0): - dump_file.write(f"Truncated: file exceeded {LIMIT_FILE_SIZE} MiB limit.\n".encode()) - break - dump_file.write(metadata_bytes) - total_size += len(metadata_bytes) - except Exception as e: - log.error(f"Error writing memory region metadata: {e}") - - except psutil.Error as e: - log.error(f"Error accessing process memory: {e}") - except Exception as e: - log.error(f"General memory dump error: {e}") - - log.info("Memory scan saved to Ram_Dump.txt") - - -# Main function to run all tasks -@log.function -def main(): - """ - Executes all memory diagnostics and collection routines. - """ - try: - os.makedirs(DUMP_DIR, exist_ok=True) - except Exception as e: - log.critical(f"Failed to create dump directory '{DUMP_DIR}': {e}") - return - - log.info("Starting system memory collection tasks...") - capture_ram_snapshot() - gather_system_info() - memory_dump() - log.info("All tasks completed [dump_memory.py].") - - -if __name__ == "__main__": - main() diff --git a/CODE/encrypted_drive_audit.py b/CODE/encrypted_drive_audit.py deleted file mode 100644 index 290281bf..00000000 --- a/CODE/encrypted_drive_audit.py +++ /dev/null @@ -1,106 +0,0 @@ -import datetime -import getpass -import os -import platform -import shutil -import subprocess -from pathlib import Path - -from logicytics import check, log - - -def now_iso(): - return datetime.datetime.now().astimezone().isoformat() - - -def run_cmd(cmd): - log.debug(f"Running command: {cmd}") - try: - proc = subprocess.run(cmd, capture_output=True, text=True, timeout=30) - if proc.returncode == 0: - log.debug(f"Command succeeded: {cmd}") - else: - log.warning(f"Command returned {proc.returncode}: {cmd}") - return proc.stdout.strip(), proc.stderr.strip(), proc.returncode - except FileNotFoundError: - log.error(f"Command not found: {cmd[0]}") - return "", "not found", 127 - except subprocess.TimeoutExpired: - log.error(f"Command timed out: {cmd}") - return "", "timeout", 124 - - -def have(cmd_name): - exists = shutil.which(cmd_name) is not None - log.debug(f"Check if '{cmd_name}' exists: {exists}") - return exists - - -def get_mountvol_output(): - log.info("Gathering mounted volumes via mountvol") - out, err, _ = run_cmd(["mountvol"]) - if not out: - return err - lines = out.splitlines() - filtered = [] - keep = False - for line in lines: - if line.strip().startswith("\\\\?\\Volume"): - keep = True - if keep: - filtered.append(line) - return "\n".join(filtered) - - -def main(): - script_dir = Path(__file__).resolve().parent - report_path = script_dir / "win_encrypted_volume_report.txt" - log.info(f"Starting encrypted volume analysis, report will be saved to {report_path}") - - with report_path.open("w", encoding="utf-8") as f: - f.write("=" * 80 + "\n") - f.write("Windows Encrypted Volume Report\n") - f.write("=" * 80 + "\n") - f.write(f"Generated at: {now_iso()}\n") - f.write(f"User: {getpass.getuser()}\n") - f.write(f"IsAdmin: {check.admin()}\n") - f.write(f"Hostname: {platform.node()}\n") - f.write(f"Version: {platform.platform()}\n\n") - - # Logical drives - log.info("Gathering logical volumes via wmic") - f.write("Logical Volumes (wmic):\n") - out, err, _ = run_cmd(["wmic", "logicaldisk", "get", - "DeviceID,DriveType,FileSystem,FreeSpace,Size,VolumeName"]) - f.write(out + "\n" + err + "\n\n") - - # Mounted volumes - f.write("Mounted Volumes (mountvol):\n") - f.write(get_mountvol_output() + "\n\n") - - # BitLocker status - f.write("=" * 80 + "\nBitLocker Status\n" + "=" * 80 + "\n") - if have("manage-bde"): - log.info("Checking BitLocker status with manage-bde") - for letter in "ABCDEFGHIJKLMNOPQRSTUVWXYZ": - path = f"{letter}:" - if os.path.exists(f"{path}\\"): - out, err, _ = run_cmd(["manage-bde", "-status", path]) - f.write(f"Drive {path}:\n{out}\n{err}\n\n") - else: - log.warning("manage-bde not found") - - if have("powershell"): - log.info("Checking BitLocker status with PowerShell") - f.write("PowerShell Get-BitLockerVolume:\n") - ps_cmd = r"Get-BitLockerVolume | Format-List *" - out, err, _ = run_cmd(["powershell", "-NoProfile", "-Command", ps_cmd]) - f.write(out + "\n" + err + "\n\n") - else: - log.warning("PowerShell not available") - - log.info(f"Report successfully saved to {report_path}") - - -if __name__ == "__main__": - main() diff --git a/CODE/event_log.py b/CODE/event_log.py deleted file mode 100644 index a27d561c..00000000 --- a/CODE/event_log.py +++ /dev/null @@ -1,83 +0,0 @@ -import os -import shutil -import threading - -import wmi # Import the wmi library - -from logicytics import log - - -@log.function -def parse_event_logs(log_type: str, output_file: str): - """ - Parses Windows event logs of a specified type and writes them to an output file using WMI. - - Args: - log_type (str): The type of event log to parse (e.g., 'Security', 'Application', 'System'). - output_file (str): The file path where the parsed event logs will be written. - - Raises: - wmi.x_wmi: If there is a WMI-specific error during event log retrieval. - Exception: If there is a general error during file operations or log parsing. - - Notes: - - Requires administrative privileges to access Windows event logs. - - Retrieves all events for the specified log type using a WMI query. - - Writes event details including category, timestamp, source, event ID, type, and data. - - Logs informational and debug messages during the parsing process. - """ - log.info(f"Parsing {log_type} events (Windows Events) and writing to {output_file}, this may take a while...") - try: - # Initialize WMI connection - c = wmi.WMI() - - # Query based on log_type ('Security', 'Application', or 'System') - query = f"SELECT * FROM Win32_NTLogEvent WHERE Logfile = '{log_type}'" - log.debug(f"Executing WMI query: {query}") - - # Open the output file for writing - with open(output_file, 'w') as f: - events = c.query(query) - f.write(f"Total records: {len(events)}\n\n") - log.debug(f"Number of events retrieved: {len(events)}") - for event in events: - event_data = { - 'Event Category': event.Category, - 'Time Generated': event.TimeGenerated, - 'Source Name': event.SourceName, - 'Event ID': event.EventCode, - 'Event Type': event.Type, - 'Event Data': event.InsertionStrings - } - f.write(str(event_data) + '\n\n') - - log.info(f"{log_type} events (Windows Events) have been written to {output_file}") - except wmi.x_wmi as err: - log.error(f"Error opening or reading the event log: {err}") - except Exception as err: - log.error(f"Fatal issue: {err}") - - -if __name__ == "__main__": - try: - if os.path.exists('event_logs'): - shutil.rmtree('event_logs') - os.mkdir('event_logs') - except Exception as e: - log.error(f"Fatal issue: {e}") - exit(1) - - threads = [] - threads_items = [('Security', 'event_logs/Security_events.txt'), - ('Application', 'event_logs/App_events.txt'), - ('System', 'event_logs/System_events.txt')] - - for log_type_main, output_file_main in threads_items: - thread = threading.Thread(target=parse_event_logs, args=(log_type_main, output_file_main)) - thread.daemon = True # Don't hang if main thread exits - threads.append(thread) - thread.start() - for thread in threads: - thread.join(timeout=600) # Wait max 10 minutes per thread - if thread.is_alive(): - log.error(f"Thread for {thread.name} timed out (10 minutes)!") diff --git a/CODE/log_miner.py b/CODE/log_miner.py deleted file mode 100644 index eff74f47..00000000 --- a/CODE/log_miner.py +++ /dev/null @@ -1,54 +0,0 @@ -import subprocess - -from logicytics import log - - -@log.function -def backup_windows_logs(): - """ - Backs up Windows system logs to a CSV file using PowerShell. - - This function retrieves system logs and exports them to a CSV file named 'Logs_backup.csv'. - It uses PowerShell's Get-EventLog cmdlet to collect system logs and Export-Csv to save them. - - The function handles potential errors during log backup and logs the operation's outcome. - If the backup fails, an error message is logged without raising an exception. - - Returns: - None - - Raises: - No explicit exceptions are raised; errors are logged instead. - - Example: - When called, this function will create a 'Logs_backup.csv' file - containing all system event log entries. - """ - try: - log_type = "System" - backup_file = "Logs_backup.csv" - # Construct the PowerShell command as a single string - cmd = f'Get-EventLog -LogName "{log_type}" | Export-Csv -Path "{backup_file}" -NoTypeInformation' - - # Use subprocess.Popen to Execute the PowerShell command - process = subprocess.Popen( - ["powershell.exe", "-Command", cmd], - stdin=subprocess.PIPE, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - universal_newlines=True, - ) - stdout, stderr = process.communicate(input=cmd) - - if process.returncode != 0: - log.error(f"Failed to backup logs: {stderr.strip()}") - - log.info(f"Windows logs backed up to {backup_file}") - except Exception as e: - log.error(f"Failed to backup logs: {str(e)}") - - log.info("Log Miner completed.") - - -if __name__ == "__main__": - backup_windows_logs() diff --git a/CODE/logicytics/Checks.py b/CODE/logicytics/Checks.py deleted file mode 100644 index e3ad3c9d..00000000 --- a/CODE/logicytics/Checks.py +++ /dev/null @@ -1,92 +0,0 @@ -from __future__ import annotations - -import ctypes -import os.path -import subprocess -import zipfile - -from logicytics.Execute import Execute - - -class Check: - @staticmethod - def admin() -> bool: - """ - Check if the current user has administrative privileges. - - Returns: - bool: True if the user is an admin, False otherwise. - """ - try: - return ctypes.windll.shell32.IsUserAnAdmin() - except AttributeError: - return False - - @staticmethod - def execution_policy() -> bool: - """ - Check if PowerShell execution policy is set to unrestricted. - - Returns: - bool: True if execution policy is unrestricted, False otherwise. - - Note: - This method requires PowerShell to be available on the system. - """ - try: - result = subprocess.run( - ["powershell", "-Command", "Get-ExecutionPolicy"], - capture_output=True, - text=True, - timeout=5 # Don't hang forever - ) - return result.returncode == 0 and result.stdout.strip().lower() == "unrestricted" - except (subprocess.TimeoutExpired, subprocess.SubprocessError) as e: - exit(f"Failed to check execution policy: {e}") - - @staticmethod - def uac() -> bool: - """ - Check if User Account Control (UAC) is enabled on the system. - - This function runs a PowerShell command to retrieve the value of the EnableLUA registry key, - which indicates whether UAC is enabled. It then returns True if UAC is enabled, False otherwise. - - Returns: - bool: True if UAC is enabled, False otherwise. - """ - value = Execute.command( - r"powershell (Get-ItemProperty HKLM:\SOFTWARE\Microsoft\Windows\CurrentVersion\Policies\System).EnableLUA" - ) - return int(value.strip("\n")) == 1 - - @staticmethod - def sys_internal_zip() -> str: - """ - Extracts the SysInternal_Suite zip file if it exists and is not ignored. - - This function checks if the SysInternal_Suite zip file exists and if it is not ignored. - If the zip file exists and is not ignored, - it extracts its contents to the SysInternal_Suite directory. - If the zip file is ignored, it prints a message indicating that it is skipping the extraction. - - Raises: - Exception: If there is an error during the extraction process. The error message is printed to the console and the program exits. - """ - try: - ignore_file = os.path.exists("../SysInternal_Suite/.sys.ignore") - zip_file = os.path.exists("../SysInternal_Suite/SysInternal_Suite.zip") - - if zip_file and not ignore_file: - with zipfile.ZipFile( - "../SysInternal_Suite/SysInternal_Suite.zip" - ) as zip_ref: - zip_ref.extractall("SysInternal_Suite") - return "SysInternal_Suite zip extracted" - - elif ignore_file: - return "Found .sys.ignore file, skipping SysInternal_Suite zip extraction" - - return None - except Exception as err: - exit(f"Failed to unzip SysInternal_Suite: {err}") diff --git a/CODE/logicytics/Config.py b/CODE/logicytics/Config.py deleted file mode 100644 index 6e0bc187..00000000 --- a/CODE/logicytics/Config.py +++ /dev/null @@ -1,48 +0,0 @@ -import configparser -import os - - -def __config_data() -> tuple[str, str, list[str], bool, str]: - """ - Retrieves configuration data from the 'config.ini' file. - - If the configuration file is not found in any of these locations, - the program exits with an error message. - - Returns: - tuple[str, str, list[str], bool]: A tuple containing: - - Log level (str): Either "DEBUG" or "INFO" - - Version (str): System version from configuration - - Files (list[str]): List of files specified in configuration - - Delete old logs (bool): Flag indicating whether to delete old log files - - config itself - - Raises: - SystemExit: If the 'config.ini' file cannot be found in any of the attempted locations - """ - - def _config_path() -> str: - configs_path = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "config.ini") - - if os.path.exists(configs_path): - return configs_path - exit("The config.ini file is not found in the expected location.") - - config_local = configparser.ConfigParser() - path = _config_path() - config_local.read(path) - - log_using_debug = config_local.getboolean("Settings", "log_using_debug") - delete_old_logs = config_local.getboolean("Settings", "delete_old_logs") - version = config_local.get("System Settings", "version") - files = config_local.get("System Settings", "files").split(", ") - - log_using_debug = "DEBUG" if log_using_debug else "INFO" - - return log_using_debug, version, files, delete_old_logs, config_local - - -# Check if the script is being run directly, if not, set up the library -if __name__ == '__main__': - exit("This is a library, Please import rather than directly run.") -DEBUG, VERSION, CURRENT_FILES, DELETE_LOGS, config = __config_data() diff --git a/CODE/logicytics/Execute.py b/CODE/logicytics/Execute.py deleted file mode 100644 index 85de3f8b..00000000 --- a/CODE/logicytics/Execute.py +++ /dev/null @@ -1,96 +0,0 @@ -from __future__ import annotations - -import subprocess -from subprocess import CompletedProcess - - -class Execute: - @classmethod - def script(cls, script_path: str) -> list[tuple[str, str]] | None: - """ - Execute a script file based on its file extension. - - Executes Python and PowerShell scripts with different handling mechanisms. - For Python scripts, runs the script and returns None. - For PowerShell scripts, first unblocks the script and then executes it, - returning a list of message-ID pairs. - - Parameters: - script_path (str): Path to the script file to be executed. - - Returns: - list[list[str]] | None: A list of message-ID pairs for PowerShell scripts, - or None for Python scripts. - - Raises: - Potential subprocess-related exceptions during script execution. - """ - if script_path.endswith(".py"): - cls.__run_python_script(script_path) - return None - else: - if script_path.endswith(".ps1"): - cls.__unblock_ps1_script(script_path) - return cls.__run_other_script(script_path) - - @staticmethod - def command(command: str) -> str: - """ - Runs a command in a subprocess and returns the output as a string. - - Parameters: - command (str): The command to be executed. - - Returns: - CompletedProcess.stdout: The output of the command as a string. - """ - process = subprocess.run(command, capture_output=True, text=True) - return process.stdout - - @staticmethod - def __unblock_ps1_script(script: str): - """ - Unblocks and runs a PowerShell (.ps1) script. - Parameters: - script (str): The path of the PowerShell script. - Returns: - None - """ - try: - unblock_command = f'powershell.exe -Command "Unblock-File -Path {script}"' - subprocess.run(unblock_command, shell=False, check=True) - except Exception as err: - exit(f"Failed to unblock script: {err}") - - @staticmethod - def __run_python_script(script: str): - """ - Runs a Python (.py) script. - Parameters: - script (str): The path of the Python script. - Returns: - None - """ - result = subprocess.Popen( - ["python", script], stdout=subprocess.PIPE - ).communicate()[0] - # LEAVE AS PRINT - print(result.decode()) - - @classmethod - def __run_other_script(cls, script: str) -> list[list[str]]: - """ - Runs a script with other extensions and logs output based on its content. - Parameters: - script (str): The path of the script. - Returns: - None - """ - result = cls.command(f"powershell.exe -File {script}") - lines = result.splitlines() - messages = [] - for line in lines: - if ":" in line: - id_part, message_part = line.split(":", 1) - messages.append([message_part.strip(), id_part.strip()]) - return messages diff --git a/CODE/logicytics/FileManagement.py b/CODE/logicytics/FileManagement.py deleted file mode 100644 index a2633feb..00000000 --- a/CODE/logicytics/FileManagement.py +++ /dev/null @@ -1,230 +0,0 @@ -from __future__ import annotations -from __future__ import annotations - -import hashlib -import os.path -import shutil -import subprocess -import zipfile -from datetime import datetime - - -class FileManagement: - @staticmethod - def open_file(file: str, use_full_path: bool = False) -> str | None: - """ - Opens a specified file using its default application in a cross-platform manner. - Args: - file (str): The path to the file to be opened. - use_full_path (bool): Whether to use the full path of the file or not. - Returns: - None - """ - if not file == "": - if use_full_path: - current_dir = os.path.dirname(os.path.abspath(__file__)) - file_path = os.path.join(current_dir, file) - else: - file_path = os.path.realpath(file) - try: - subprocess.run(["start", file_path], shell=False) - except Exception as e: - return f"Error opening file: {e}" - return None - - @staticmethod - def mkdir(): - """ - Creates the necessary directories for storing logs, and data. - - This method ensures the existence of specific directory structures used by the application, including: - - Log directories for general, debug, and performance logs - - Data directories for storing hashes and zip files - - The method uses `os.makedirs()` with `exist_ok=True` to create directories without raising an error if they already exist. - - Returns: - None: No return value. Directories are created as a side effect. - """ - os.makedirs("../ACCESS/LOGS/", exist_ok=True) - os.makedirs("../ACCESS/LOGS/DEBUG", exist_ok=True) - os.makedirs("../ACCESS/LOGS/PERFORMANCE", exist_ok=True) - os.makedirs("../ACCESS/DATA/Hashes", exist_ok=True) - os.makedirs("../ACCESS/DATA/Zip", exist_ok=True) - - class Zip: - """ - A class to handle zipping files, generating SHA256 hashes, and moving files. - - Methods: - __get_files_to_zip(path: str) -> list: - Returns a list of files to be zipped, excluding certain file types and names. - - __create_zip_file(path: str, files: list, filename: str): - Creates a zip file from the given list of files. - - __remove_files(path: str, files: list): - Removes the specified files from the given path. - - __generate_sha256_hash(filename: str) -> str: - Generates a SHA256 hash for the specified zip file. - - __write_hash_to_file(filename: str, sha256_hash: str): - Writes the SHA256 hash to a file. - - __move_files(filename: str): - Moves the zip file and its hash file to designated directories. - - and_hash(cls, path: str, name: str, flag: str) -> tuple | str: - Zips files, generates a SHA256 hash, and moves the files. - """ - - @staticmethod - def __get_files_to_zip(path: str) -> list: - """ - Returns a list of files and directories to be zipped, excluding certain file types and names. - - Args: - path (str): The directory path to search for files. - - Returns: - list: A list of file and directory names to be zipped. - """ - excluded_extensions = (".py", ".exe", ".bat", ".ps1", ".pkl", ".pth") - excluded_prefixes = ( - "config.ini", - "SysInternal_Suite", - "__pycache__", - "logicytics", - "vulnscan", - ) - - return [ - f - for f in os.listdir(path) - if not f.endswith(excluded_extensions) - and not f.startswith(excluded_prefixes) - ] - - @staticmethod - def __create_zip_file(path: str, files: list, filename: str): - """ - Creates a zip file from the given list of files. - - Args: - path (str): The directory path containing the files. - files (list): A list of file names to be zipped. - filename (str): The name of the output zip file. - - Returns: - None - """ - - def ignore_files(files_func): - for root, _, file_func in os.walk(os.path.join(path, files_func)): - for f in file_func: - zip_file.write( - os.path.join(root, f), - os.path.relpath(os.path.join(root, f), path), - ) - - with zipfile.ZipFile(f"{filename}.zip", "w") as zip_file: - for file in files: - if os.path.isdir(os.path.join(path, file)): - ignore_files(file) - else: - zip_file.write(os.path.join(path, file)) - - @staticmethod - def __remove_files(path: str, files: list) -> str | None: - """ - Removes the specified files from the given path. - - Args: - path (str): The directory path containing the files. - files (list): A list of file names to be removed. - - Returns: - None or str: Returns an error message if an exception occurs. - """ - for file in files: - try: - shutil.rmtree(os.path.join(path, file)) - except OSError: - os.remove(os.path.join(path, file)) - except Exception as e: - return f"Error: {e}" - return None - - @staticmethod - def __generate_sha256_hash(filename: str) -> str: - """ - Generates a SHA256 hash for the specified zip file. - - Args: - filename (str): The name of the zip file. - - Returns: - str: The SHA256 hash of the zip file. - """ - with open(f"{filename}.zip", "rb") as zip_file: - zip_data = zip_file.read() - return hashlib.sha256(zip_data).hexdigest() - - @staticmethod - def __write_hash_to_file(filename: str, sha256_hash: str): - """ - Writes the SHA256 hash to a file. - - Args: - filename (str): The name of the hash file. - sha256_hash (str): The SHA256 hash to be written. - - Returns: - None - """ - with open(f"{filename}.hash", "w") as hash_file: - hash_file.write(sha256_hash) - - @staticmethod - def __move_files(filename: str): - """ - Moves the zip file and its hash file to designated directories. - - Args: - filename (str): The name of the files to be moved. - - Returns: - None - """ - shutil.move(f"{filename}.zip", "../ACCESS/DATA/Zip") - shutil.move(f"{filename}.hash", "../ACCESS/DATA/Hashes") - - @classmethod - def and_hash(cls, path: str, name: str, flag: str) -> tuple | str: - """ - Zips files, generates a SHA256 hash, and moves the files. - - Args: - path (str): The directory path containing the files. - name (str): The base name for the output files. - flag (str): A flag to be included in the output file names. - - Returns: - tuple or str: A tuple containing success messages or an error message. - """ - time = datetime.now().strftime("%Y-%m-%d_%H-%M-%S") - filename = f"Logicytics_{name}_{flag}_{time}" - files_to_zip = cls.__get_files_to_zip(path) - cls.__create_zip_file(path, files_to_zip, filename) - check = cls.__remove_files(path, files_to_zip) - if isinstance(check, str): - return check - else: - sha256_hash = cls.__generate_sha256_hash(filename) - cls.__write_hash_to_file(filename, sha256_hash) - cls.__move_files(filename) - return ( - f"Zip file moved to ../ACCESS/DATA/Zip/{filename}.zip", - f"SHA256 Hash file moved to ../ACCESS/DATA/Hashes/{filename}.hash", - ) diff --git a/CODE/logicytics/Flag.py b/CODE/logicytics/Flag.py deleted file mode 100644 index 0a4367f0..00000000 --- a/CODE/logicytics/Flag.py +++ /dev/null @@ -1,820 +0,0 @@ -from __future__ import annotations - -import argparse -import difflib -import gzip -import json -import os -from collections import Counter -from datetime import datetime - -from logicytics.Config import config -from logicytics.Logger import log - -# Check if the script is being run directly, if not, set up the library -if __name__ == "__main__": - exit("This is a library, Please import rather than directly run.") -else: - # Save user preferences? - SAVE_PREFERENCES = config.getboolean("Settings", "save_preferences") - # Debug mode for Sentence Transformer - DEBUG_MODE = config.getboolean( - "Flag Settings", "model_debug" - ) # Debug mode for Sentence Transformer - # File for storing user history data - HISTORY_FILE = os.path.join( - os.path.dirname(os.path.abspath(__file__)), "User_History.json.gz" - ) # User history file - # Minimum accuracy threshold for flag suggestions - MIN_ACCURACY_THRESHOLD = float( - config.get("Flag Settings", "accuracy_min") - ) # Minimum accuracy threshold for flag suggestions - if not 1 <= MIN_ACCURACY_THRESHOLD <= 99: - raise ValueError("accuracy_min must be between 1 and 99") - - -class Flag: - class Match: - @staticmethod - def __get_sim(user_input: str, all_descriptions: list[str]) -> list[float]: - """ - Compute cosine similarity between user input and flag descriptions using a Sentence Transformer model. - - This method encodes the user input and historical flag descriptions into embeddings and calculates their cosine similarities. It handles model loading, logging configuration, and error handling for the embedding process. - - Parameters: - user_input (str): The current user input to match against historical descriptions - all_descriptions (list[str]): A list of historical flag descriptions to compare - - Returns: - list[float]: A list of similarity scores between the user input and each historical description - - Raises: - SystemExit: If there is an error loading the specified Sentence Transformer model - - Notes: - - Uses the model specified in the configuration file - - Configures logging based on the global DEBUG_MODE setting - - Converts embeddings to tensors for efficient similarity computation - """ - # Encode the current user input and historical inputs - from sentence_transformers import SentenceTransformer, util - import logging # Suppress logging messages from Sentence Transformer due to verbosity - - # Set the logging level based on the debug mode, either DEBUG or ERROR (aka only important messages) - if DEBUG_MODE: - logging.getLogger("sentence_transformers").setLevel(logging.DEBUG) - else: - logging.getLogger("sentence_transformers").setLevel(logging.ERROR) - - try: - MODEL = SentenceTransformer(config.get("Flag Settings", "model_to_use")) - except Exception as e: - log.critical(f"Error: {e}") - log.error("Please check the model name in the config file.") - log.error( - f"Model name {config.get('Flag Settings', 'model_to_use')} may not be valid." - ) - exit(1) - - user_embedding = MODEL.encode( - user_input, convert_to_tensor=True, show_progress_bar=DEBUG_MODE - ) - historical_embeddings = MODEL.encode( - all_descriptions, convert_to_tensor=True, show_progress_bar=DEBUG_MODE - ) - - # Compute cosine similarities - similarities = ( - util.pytorch_cos_sim(user_embedding, historical_embeddings) - .squeeze(0) - .tolist() - ) - return similarities - - @classmethod - def __suggest_flags_based_on_history(cls, user_input: str) -> list[str]: - """ - Suggests flags based on historical data and similarity to the current input. - - This method analyzes historical user interactions to recommend relevant flags when preferences for saving history are enabled. It uses semantic similarity to find the most contextually related flags from past interactions. - - Parameters: - user_input (str): The current input for which suggestions are needed. - - Returns: - list[str]: A list of suggested flags derived from historical interactions, filtered by similarity threshold. - - Notes: - - Returns an empty list if history saving is disabled or no interaction history exists - - Uses cosine similarity with a minimum threshold of 0.3 to filter suggestions - - Limits suggestions to top 3 most similar historical inputs - - Removes duplicate flag suggestions - """ - if not SAVE_PREFERENCES: - return [] - history_data = cls.load_history() - if not history_data or "interactions" not in history_data: - return [] - - interactions = history_data["interactions"] - all_descriptions = [] - all_flags = [] - - # Combine all flags and their respective user inputs - for flag, details in interactions.items(): - all_flags.extend([flag] * len(details)) - all_descriptions.extend([detail["user_input"] for detail in details]) - - # Encode the current user input and historical inputs - # Compute cosine similarities - similarities = cls.__get_sim(user_input, all_descriptions) - - # Find the top 3 most similar historical inputs - top_indices = sorted( - range(len(similarities)), key=lambda i: similarities[i], reverse=True - )[:3] - suggested_flags = [ - all_flags[i] for i in top_indices if similarities[i] > 0.3 - ] - - # Remove duplicates and return suggestions - return list(dict.fromkeys(suggested_flags)) - - @classmethod - def generate_summary_and_graph(cls): - """ - Generates a comprehensive summary and visualization of user interaction history with command-line flags. - - This method processes historical interaction data, computes statistical insights, and creates a bar graph representing flag usage frequency. It performs the following key tasks: - - Loads historical interaction data from a compressed file - - Calculates and prints detailed statistics for each flag - - Generates a horizontal bar graph of flag usage counts - - Saves the graph visualization to a PNG file - - Parameters: - cls (_Match): The class instance containing historical data methods - - Raises: - SystemExit: If no history data file is found - FileNotFoundError: If unable to save the graph in default locations - - Side Effects: - - Prints detailed interaction summary to console - - Saves flag usage graph as a PNG image - - Uses matplotlib to create visualization - - Notes: - - Currently in beta stage of development - - Requires matplotlib for graph generation - - Attempts to save graph in multiple predefined directory paths - """ - # Load the decompressed history data using the load_history function - import matplotlib.pyplot as plt - - if not os.path.exists(HISTORY_FILE): - exit("No history data found.") - - history_data = cls.load_history() - - # Extract interactions and flag usage count - interactions = history_data["interactions"] - flags_usage = history_data["flags_usage"] - - # Summary of flag usage - total_interactions = sum(flags_usage.values()) - - log.info( - "--------------------------------------------------\n Flag guessing statistics:\n --------------------------------------------------" - ) - - for flag, details in interactions.items(): - accuracies = [detail["accuracy"] for detail in details] - device_names = [detail["device_name"] for detail in details] - user_inputs = [detail["user_input"] for detail in details] - - average_accuracy = sum(accuracies) / len(accuracies) - most_common_device = Counter(device_names).most_common(1)[0][0] - average_user_input = Counter(user_inputs).most_common(1)[0][0] - log.info(f"""Flag: {flag} - Average Accuracy: {average_accuracy:.2f}% - Most Common Device Name: {most_common_device} - Most Common User Input: {average_user_input}""") - - # Print the summary to the console - log.info( - "--------------------------------------------------\n User Interaction Summary:\n --------------------------------------------------" - ) - - log.info( - f"Total Interactions with the match flag feature: {total_interactions}" - ) - flag_usage_summary = "\n".join( - [ - f" {flag}: {count} times" - for flag, count in flags_usage.items() - ] - ) - log.info(f"Flag Usage Summary:\n{flag_usage_summary}") - - # Generate the graph for flag usage - flags = list(flags_usage.keys()) - counts = list(flags_usage.values()) - - plt.figure(figsize=(10, 6)) - plt.barh(flags, counts, color="skyblue") - plt.xlabel("Usage Count") - plt.title("Flag Usage Frequency") - plt.gca().invert_yaxis() # Invert y-axis for better readability - plt.subplots_adjust( - left=0.2, right=0.8, top=0.9, bottom=0.1 - ) # Adjust layout - - # Save and display the graph - try: - plt.savefig("../ACCESS/DATA/Flag_usage_summary.png") - log.info( - "Flag Usage Summary Graph saved to 'ACCESS/DATA/Flag_usage_summary.png'" - ) - except FileNotFoundError: - try: - plt.savefig("../../ACCESS/DATA/Flag_usage_summary.png") - log.info( - "Flag Usage Summary Graph saved to 'ACCESS/DATA/Flag_usage_summary.png'" - ) - except FileNotFoundError: - plt.savefig("Flag_usage_summary.png") - log.info( - "Flag Usage Summary Graph saved in current working directory as 'Flag_usage_summary.png'" - ) - - @staticmethod - def load_history() -> dict: - """ - Load user interaction history from a gzipped JSON file. - - This method attempts to read and parse historical interaction data from a compressed JSON file. If the file is not found, it returns an empty history structure with an empty interactions dictionary and a zero-initialized flags usage counter. - - Returns: - dict: A dictionary containing: - - 'interactions': A dictionary of past user interactions - - 'flags_usage': A Counter object tracking flag usage frequencies - - Raises: - json.JSONDecodeError: If the JSON file is malformed - gzip.BadGzipFile: If the gzipped file is corrupted - """ - try: - with gzip.open( - HISTORY_FILE, "rt", encoding="utf-8" - ) as f: # Use 'rt' mode for text read - return json.load(f) - except FileNotFoundError: - return {"interactions": {}, "flags_usage": Counter()} - - @staticmethod - def save_history(history_data: dict): - """ - Save user interaction history to a gzipped JSON file. - - This method writes the user history to a compressed JSON file only if saving preferences are enabled. - The history is saved with an indentation of 4 spaces for readability. - - Parameters: - history_data (dict[str, any]): A dictionary containing user interaction history data to be saved. - - Notes: - - Saves only if SAVE_PREFERENCES is True - - Uses gzip compression to reduce file size - - Writes in UTF-8 encoding - - Indents JSON for human-readable format - """ - if SAVE_PREFERENCES: - with gzip.open( - HISTORY_FILE, "wt", encoding="utf-8" - ) as f: # Use 'wt' mode for text write - json.dump(history_data, f, indent=4) - - @classmethod - def update_history(cls, user_input: str, matched_flag: str, accuracy: float): - """ - Update the user interaction history with details of a matched flag. - - This method records user interactions with flags, including timestamp, input, match accuracy, - and device information. It only updates history if save preferences are enabled. - - Parameters: - user_input (str): The original input text provided by the user. - matched_flag (str): The flag that was successfully matched to the user input. - accuracy (float): The similarity/match accuracy score for the flag. - - Side Effects: - - Modifies the history JSON file by adding a new interaction entry - - Increments the usage count for the matched flag - - Requires write access to the history file - - Notes: - - Skips history update if SAVE_PREFERENCES is False - - Creates new flag entries in history if they do not exist - - Uses current timestamp and logged-in user's device name - """ - if not SAVE_PREFERENCES: - return - history_data = cls.load_history() - matched_flag = matched_flag.lstrip("-") - - # Ensure that interactions is a dictionary (not a list) - if not isinstance(history_data["interactions"], dict): - history_data["interactions"] = {} - - # Create a new interaction dictionary - interaction = { - "timestamp": datetime.now().strftime("%H:%M:%S - %d/%m/%Y"), - "user_input": user_input, - "accuracy": accuracy, - "device_name": os.getlogin(), - } - - # Ensure the flag exists in the interactions dictionary - if matched_flag not in history_data["interactions"]: - history_data["interactions"][matched_flag] = [] - - # Append the new interaction to the flag's list of interactions - history_data["interactions"][matched_flag].append(interaction) - - # Ensure the flag exists in the flags_usage counter and increment it - if matched_flag not in history_data["flags_usage"]: - history_data["flags_usage"][matched_flag] = 0 - history_data["flags_usage"][matched_flag] += 1 - - cls.save_history(history_data) - - @classmethod - def flag( - cls, user_input: str, flags: list[str], flag_description: list[str] - ) -> tuple[str, float]: - """ - Matches user input to flag descriptions using advanced semantic similarity. - - Computes the best matching flag based on cosine similarity between the user input and flag descriptions. - Handles matching with a minimum accuracy threshold and provides flag suggestions from historical data - if no direct match is found. - - Parameters: - user_input (str): The input string to match against available flags. - flags (list[str]): List of available command flags. - flag_description (list[str]): Corresponding descriptions for each flag. - - Returns: - tuple[str, float]: A tuple containing: - - The best matched flag (or 'Nothing matched') - - Accuracy percentage of the match (0.0-100.0) - - Raises: - ValueError: If the number of flags and descriptions do not match. - - Side Effects: - - Updates user interaction history - - Prints flag suggestions if no direct match is found - - Requires a global MIN_ACCURACY_THRESHOLD to be defined - - Example: - matched_flag, accuracy = Flag.flag("show help", - ["-h", "--verbose"], - ["Display help", "Enable verbose output"]) - """ - if len(flags) != len(flag_description): - raise ValueError( - "flags and flag_description lists must be of the same length" - ) - - # Combine flags and descriptions for better matching context - combined_descriptions = [ - f"{flag} {desc}" for flag, desc in zip(flags, flag_description) - ] - - # Encode user input and all descriptions - # Compute cosine similarities - similarities = cls.__get_sim(user_input, combined_descriptions) - - # Find the best match - best_index = max(range(len(similarities)), key=lambda i: similarities[i]) - best_accuracy = similarities[best_index] * 100 - best_match = ( - flags[best_index] - if best_accuracy > MIN_ACCURACY_THRESHOLD - else "Nothing matched" - ) - - # Update history - cls.update_history(user_input, best_match, best_accuracy) - - # Suggest flags if accuracy is low - if best_accuracy < MIN_ACCURACY_THRESHOLD: - suggested_flags = cls.__suggest_flags_based_on_history(user_input) - if suggested_flags: - log.warning( - f"No Flags matched so suggestions based on historical data: " - f"{', '.join(suggested_flags)}" - ) - - return best_match, best_accuracy - - @classmethod - def __colorify(cls, text: str, color: str) -> str: - """ - Colorize text with ANSI color codes. - - Args: - text (str): The text to colorize - color (str): The color code ('y' for yellow, 'r' for red, 'b' for blue) - - Returns: - str: The colorized text with ANSI escape codes - """ - colors = {"y": "\033[93m", "r": "\033[91m", "b": "\033[94m"} - RESET = "\033[0m" - return f"{colors.get(color, '')}{text}{RESET}" if color in colors else text - - @classmethod - def __available_arguments( - cls, - ) -> tuple[argparse.Namespace, argparse.ArgumentParser]: - """ - Defines and parses command-line arguments for the Logicytics application. - - This method creates an ArgumentParser with a comprehensive set of flags for customizing the application's behavior. It supports various execution modes, debugging options, system management flags, and post-execution actions. - - The method handles argument parsing, provides helpful descriptions for each flag, and includes color-coded hints for user guidance. It also supports suggesting valid flags if an unknown flag is provided. - - Returns: - tuple[argparse.Namespace, argparse.ArgumentParser]: A tuple containing: - - Parsed command-line arguments (Namespace) - - The configured argument parser object - """ - # Define the argument parser - parser = argparse.ArgumentParser( - description="Logicytics, The most powerful tool for system data analysis. " - "This tool provides a comprehensive suite of features for analyzing system data, " - "including various modes for different levels of detail and customization.", - allow_abbrev=False, - ) - - # Define Actions Flags - parser.add_argument( - "--default", - action="store_true", - help="Runs Logicytics with its default settings and scripts. " - f"{cls.__colorify('- Recommended for most users -', 'b')}", - ) - - parser.add_argument( - "--threaded", - action="store_true", - help="Runs Logicytics using threads, where it runs in parallel, default settings though" - f"{cls.__colorify('- Recommended for some users -', 'b')}", - ) - - parser.add_argument( - "--modded", - action="store_true", - help="Runs the normal Logicytics, as well as any File in the MODS directory, " - "Used for custom scripts as well as default ones.", - ) - - parser.add_argument( - "--depth", - action="store_true", - help="This flag will run all default script's in threading mode, " - "as well as any clunky and huge code, which produces a lot of data " - f"{cls.__colorify('- Will take a long time -', 'y')}", - ) - - parser.add_argument( - "--nopy", - action="store_true", - help="Run Logicytics using all non-python scripts, " - f"These may be {cls.__colorify('outdated', 'y')} " - "and not the best, use only if the device doesnt have python installed.", - ) - - # TODO v3.6.1 -> Out of beta - parser.add_argument( - "--vulnscan-ai", - action="store_true", - help="Run's Logicytics new Sensitive data Detection AI, its a new feature that will " - "detect any files that are out of the ordinary, and logs their path. Runs threaded." - f"{cls.__colorify('- Beta Mode -', 'y')} " - f"{cls.__colorify('- Will take a long time -', 'y')}", - ) - - parser.add_argument( - "--minimal", - action="store_true", - help="Run Logicytics in minimal mode. Just bare essential scraping using only quick scripts", - ) - - parser.add_argument( - "--performance-check", - action="store_true", - help="Run's Logicytics default while testing its performance and time, " - "this then shows a table with the file names and time to executed. ", - ) - - parser.add_argument( - "--usage", - action="store_true", - help="Run's script that shows and gives your local statistics, on the flags used by you", - ) - - # Define Side Flags - parser.add_argument( - "--debug", - action="store_true", - help="Runs the Debugger, Will check for any issues, " - "warning etc, useful for debugging and issue reporting " - f"{cls.__colorify('- Use to get a special log file to report the bug -', 'b')}.", - ) - - parser.add_argument( - "--update", - action="store_true", - help="Update Logicytics from GitHub, only if you have git properly installed " - "and the project was downloaded via git " - f"{cls.__colorify('- Use on your own device only -', 'y')}.", - ) - - parser.add_argument( - "--dev", - action="store_true", - help="Run Logicytics developer mod, this is only for people who want to " - "register their contributions properly. " - f"{cls.__colorify('- Use on your own device only -', 'y')}.", - ) - - # Define After-Execution Flags - parser.add_argument( - "--reboot", - action="store_true", - help="Execute Flag that will reboot the device afterward", - ) - - parser.add_argument( - "--shutdown", - action="store_true", - help="Execute Flag that will shutdown the device afterward", - ) - - # Parse the arguments - args, unknown = parser.parse_known_args() - valid_flags = [ - action.dest for action in parser._actions if action.dest != "help" - ] - if unknown: - cls.__suggest_flag(unknown[0], valid_flags) - exit(1) - return args, parser - - @staticmethod - def __exclusivity_logic(args: argparse.Namespace) -> bool: - """ - Validates the mutual exclusivity of command-line flags to prevent invalid flag combinations. - - This method checks for conflicting or mutually exclusive flags across three flag categories: - - Special flags (reboot, shutdown, webhook) - - Action flags (default, threaded, modded, minimal, nopy, depth, performance_check) - - Exclusive flags (vulnscan_ai) - - Parameters: - args (argparse.Namespace): Parsed command-line arguments to validate. - - Returns: - bool: True if any special flags are set, False otherwise. - - Raises: - SystemExit: If incompatible flag combinations are detected, with an error message describing the conflict. - """ - special_flags = { - args.reboot, - args.shutdown, - } - action_flags = { - args.default, - args.threaded, - args.modded, - args.minimal, - args.nopy, - args.depth, - args.performance_check, - args.usage, - } - exclusive_flags = { - args.vulnscan_ai, - } - - if any(special_flags) and not any(action_flags): - log.error( - "Invalid combination of flags_list: Special and Action flag exclusivity issue." - ) - exit(1) - - if any(exclusive_flags) and any(action_flags): - log.error( - "Invalid combination of flags_list: Exclusive and Action flag exclusivity issue." - ) - exit(1) - - if any(exclusive_flags) and any(special_flags): - log.error( - "Invalid combination of flags_list: Exclusive and Special flag exclusivity issue." - ) - exit(1) - - return any(special_flags) - - @staticmethod - def __used_flags_logic(args: argparse.Namespace) -> tuple[str, ...]: - """ - Determines the flags that are set to True in the provided command-line arguments. - - This method examines the arguments namespace and returns a tuple of flag names - that have been activated. It limits the returned flags to a maximum of two to - prevent excessive flag usage. - - Parameters: - args (argparse.Namespace): Parsed command-line arguments to be analyzed. - - Returns: - tuple[str, ...]: A tuple containing the names of flags set to True, - with a maximum of two flags. - - Notes: - - If no flags are set, returns an empty tuple. - - Stops collecting flags after finding two True flags to limit complexity. - """ - flags = {key: getattr(args, key) for key in vars(args)} - true_keys = [] - for key, value in flags.items(): - if value: - true_keys.append(key) - if len(true_keys) == 2: - break - return tuple(true_keys) - - @classmethod - def __suggest_flag(cls, user_input: str, valid_flags: list[str]): - """ - Suggests the closest valid flag based on the user's input and provides interactive flag matching. - - This method handles flag suggestion through two mechanisms: - 1. Using difflib to find close flag matches - 2. Prompting user for a description to find the most relevant flag - - Args: - user_input (str): The flag input by the user. - valid_flags (list[str]): The list of valid flags. - - Behavior: - - If a close flag match exists, suggests the closest match - - If no close match, prompts user for a description - - Uses the _Match.flag method to find the most accurate flag based on description - - Prints matching results, with optional detailed output in debug mode - - Side Effects: - - Prints suggestions and matched flags to console - - Prompts user for additional input if no direct match is found - """ - # Get the closest valid flag match based on the user's input - closest_matches = difflib.get_close_matches( - user_input, valid_flags, n=1, cutoff=0.6 - ) - if closest_matches: - log.warning( - f"Invalid flag '{user_input}', Did you mean '--{closest_matches[0].replace('_', '-')}'?" - ) - exit(1) - - # Prompt the user for a description if no close match is found - user_input_desc = input( - "We can't find a match, Please provide a description: " - ).lower() - - # Map the user-provided description to the closest valid flag - flags_list = [f"--{flag}" for flag in valid_flags] - descriptions_list = [f"Run Logicytics with {flag}" for flag in valid_flags] - flag_received, accuracy_received = cls.Match.flag( - user_input_desc, flags_list, descriptions_list - ) - if DEBUG_MODE: - log.info( - f"User input: {user_input_desc}\nMatched flag: {flag_received.replace('_', '-')}\nAccuracy: {accuracy_received:.2f}%\n" - ) - else: - log.info( - f"Matched flag: {flag_received.replace('_', '-')} (Accuracy: {accuracy_received:.2f}%)\n" - ) - - @staticmethod - def show_help_menu(return_output: bool = False) -> str | None: - """ - Display the help menu for the Logicytics application. - - This method retrieves the argument parser from the Flag class and either prints or returns the help text based on the input parameter. - - Args: - return_output (bool, optional): Controls the method's behavior. - - If True, returns the formatted help text as a string. - - If False (default), prints the help text directly to the console. - - Returns: - str or None: Help text as a string if return_output is True, otherwise None. - - Example: - # Print help menu to console - Flag.show_help_menu() - - # Get help menu as a string - help_text = Flag.show_help_menu(return_output=True) - print(help_text) - """ - parser = Flag.__available_arguments()[1] - if return_output: - return parser.format_help() - parser.print_help() - return None - - @classmethod - def data(cls) -> tuple[str, str | None]: - """ - Handles the parsing and validation of command-line flags. - - This method processes command-line arguments, validates their usage, and manages flag interactions. It ensures that: - - Only one primary action flag is used at a time - - Special flags are handled with specific logic - - No invalid flag combinations are permitted - - User history is optionally updated based on preferences - - Returns: - tuple[str, str | None]: A tuple containing: - - The primary matched flag - - An optional secondary flag (None if not applicable) - - Exits the program if no flags are used or invalid combinations are detected - - Raises: - SystemExit: Terminates the program with an error message for: - - Invalid flag combinations - - No flags specified - """ - args, parser = cls.__available_arguments() - special_flag_used = cls.__exclusivity_logic(args) - - used_flags = [flag for flag in vars(args) if getattr(args, flag)] - - if not special_flag_used and len(used_flags) > 1: - log.error("Invalid combination of flags: Maximum 1 action flag allowed.") - exit(1) - - if special_flag_used: - used_flags = cls.__used_flags_logic(args) - if len(used_flags) > 2: - log.error("Invalid combination of flags: Maximum 2 flag mixes allowed.") - exit(1) - - if not used_flags: - cls.show_help_menu() - exit(0) - - # Update history with the matched flag(s) - if not SAVE_PREFERENCES: - return None - - def update_data_history(matched_flag: str): - """ - Update the usage count for a specific flag in the user's interaction history. - - This method increments the usage count for a given flag in the historical data. If the flag - does not exist in the history, it initializes its count to 0 before incrementing. - - Parameters: - matched_flag (str): The flag whose usage count needs to be updated. - - Side Effects: - - Modifies the 'flags_usage' dictionary in the user's history file - - Saves the updated history data to a persistent storage - - Example: - update_data_history('--verbose') # Increments usage count for '--verbose' flag - """ - history_data = cls.Match.load_history() - # Ensure the flag exists in the flags_usage counter and increment it - if matched_flag.replace("--", "") not in history_data["flags_usage"]: - history_data["flags_usage"][matched_flag.replace("--", "")] = 0 - history_data["flags_usage"][matched_flag.replace("--", "")] += 1 - cls.Match.save_history(history_data) - - if len(used_flags) == 2: - for flag in used_flags: - update_data_history(flag) - return tuple(used_flags) - update_data_history(used_flags[0]) - return used_flags[0], None diff --git a/CODE/logicytics/Get.py b/CODE/logicytics/Get.py deleted file mode 100644 index 8954000e..00000000 --- a/CODE/logicytics/Get.py +++ /dev/null @@ -1,55 +0,0 @@ -from __future__ import annotations - -import os - - -class Get: - @staticmethod - def list_of_files( - directory: str, - only_extensions: list[str] = None, - append_file_list: list[str] = None, - exclude_files: list[str] = None, - exclude_extensions: list[str] = None, - exclude_dirs: list[str] = None, - ) -> list[str]: - """ - Retrieves a list of files in the specified directory based on given extensions and exclusion criteria. - - Parameters: - directory (str): Path of the directory to search for files. - only_extensions (list[str], optional): List of file extensions to filter. If None, retrieves all files. Defaults to None. - append_file_list (list[str], optional): Existing list to append found filenames to. Defaults to None. - exclude_files (list[str], optional): List of filenames to exclude from results. Defaults to None. - exclude_extensions (list[str], optional): List of extensions to exclude from results. Defaults to None. - exclude_dirs (list[str], optional): List of directory names to ignore. Defaults to None. - - Returns: - list[str]: A list of filenames matching the specified criteria. - - Exclusion rules: - - Ignores files starting with an underscore (_) - - Skips files specified in `exclude_files` - - Skips directories specified in `ignore_dirs` - """ - append_file_list = append_file_list or [] - exclude_files = set(exclude_files or []) - exclude_extensions = set(exclude_extensions or []) - exclude_dirs = set(exclude_dirs or []) # Set for faster lookup - - for root, dirs, filenames in os.walk(directory): - # Remove ignored directories from the dirs list to prevent os.walk from entering them - dirs[:] = [d for d in dirs if d not in exclude_dirs] - - for filename in filenames: - if filename.startswith("_") or filename in exclude_files: - continue # Skip excluded files - if any(filename.endswith(ext) for ext in exclude_extensions): - continue # Skip excluded files - - file_path = os.path.relpath(os.path.join(root, filename), directory) - - if only_extensions is None or any(filename.endswith(ext) for ext in only_extensions): - append_file_list.append(file_path) - - return append_file_list diff --git a/CODE/logicytics/Logger.py b/CODE/logicytics/Logger.py deleted file mode 100644 index d4078f33..00000000 --- a/CODE/logicytics/Logger.py +++ /dev/null @@ -1,454 +0,0 @@ -from __future__ import annotations - -import inspect -import logging -import os -import re -import time -from datetime import datetime -from typing import Type - -import colorlog - -from logicytics.Config import DEBUG -from logicytics.FileManagement import FileManagement - - -class Log: - """ - A logging class that supports colored output using the colorlog library. - """ - - _instance = None - - def __new__(cls, *args, **kwargs): - """ - Ensures that only one instance of the Log class is created (Singleton pattern). - - :param cls: The class being instantiated. - :param args: Positional arguments. - :param kwargs: Keyword arguments. - :return: The single instance of the Log class. - """ - if cls._instance is None: - cls._instance = super(Log, cls).__new__(cls) - cls._instance._initialized = False - return cls._instance - - def __init__(self, config: dict = None): - """ - Initializes the Log class with the given configuration. - - :param config: A dictionary containing configuration options. - """ - FileManagement.mkdir() # Ensure the necessary directories are created - if self._initialized and config is None: - return - self._initialized = True - if config: - self.reset() - # log_path_relative variable takes Logger.py full path, - # goes up twice then joins with ACCESS\\LOGS\\Logicytics.log - log_path_relative = os.path.join( - os.path.dirname( - os.path.dirname(os.path.dirname(os.path.abspath(__file__))) - ), - "ACCESS\\LOGS\\Logicytics.log", - ) - config = config or { - "filename": log_path_relative, - "use_colorlog": True, - "log_level": "INFO", - "debug_color": "cyan", - "info_color": "green", - "warning_color": "yellow", - "error_color": "red", - "critical_color": "bold_red", - "exception_color": "red", - "colorlog_fmt_parameters": "%(log_color)s%(levelname)-8s%(reset)s %(blue)s%(message)s", - "truncate_message": True, - "delete_log": False, - } - self.EXCEPTION_LOG_LEVEL = 45 - self.INTERNAL_LOG_LEVEL = 15 - logging.addLevelName(self.EXCEPTION_LOG_LEVEL, "EXCEPTION") - logging.addLevelName(self.INTERNAL_LOG_LEVEL, "INTERNAL") - self.color = config.get("use_colorlog", True) - self.truncate = config.get("truncate_message", True) - - self.filename = config.get("filename", log_path_relative) - if self.color: - logger = colorlog.getLogger() - logger.setLevel(getattr(logging, config["log_level"].upper(), logging.INFO)) - handler = colorlog.StreamHandler() - log_colors = { - "INTERNAL": "cyan", - "DEBUG": config.get("debug_color", "cyan"), - "INFO": config.get("info_color", "green"), - "WARNING": config.get("warning_color", "yellow"), - "ERROR": config.get("error_color", "red"), - "CRITICAL": config.get("critical_color", "bold_red"), - "EXCEPTION": config.get("exception_color", "red"), - } - - formatter = colorlog.ColoredFormatter( - config.get( - "colorlog_fmt_parameters", - "%(log_color)s%(levelname)-8s%(reset)s %(blue)s%(message)s", - ), - log_colors=log_colors, - ) - - handler.setFormatter(formatter) - logger.addHandler(handler) - try: - getattr(logging, config["log_level"].upper()) - except AttributeError as AE: - self.__internal( - f"Log Level {config['log_level']} not found, setting default level to INFO -> {AE}" - ) - - if not os.path.exists(self.filename): - self.newline() - self._raw( - "| Timestamp | LOG Level |" - + " " * 71 - + "LOG Messages" - + " " * 71 - + "|" - ) - elif os.path.exists(self.filename) and config.get("delete_log", False): - with open(self.filename, "w") as f: - f.write( - "| Timestamp | LOG Level |" - + " " * 71 - + "LOG Messages" - + " " * 71 - + "|" - + "\n" - ) - self.newline() - - @staticmethod - def reset(): - """ - Resets the logger by removing all existing handlers. - """ - logger = logging.getLogger() - for handler in logger.handlers[:]: - logger.removeHandler(handler) - - @staticmethod - def __timestamp() -> str: - """ - Returns the current timestamp as a string. - - :return: Current timestamp in 'YYYY-MM-DD HH:MM:SS' format. - """ - return datetime.now().strftime("%Y-%m-%d %H:%M:%S") - - def __trunc_message(self, message: str) -> str: - """ - Pads or truncates the message to fit the log format. - - :param message: The log message to be padded or truncated. - :return: The padded or truncated message. - """ - if self.truncate is False: - return message + " " * (153 - len(message)) + "|" - return ( - message + " " * (153 - len(message)) - if len(message) < 153 - else message[:150] + "..." - ) + "|" - - def __internal(self, message): - """ - Log an internal message exclusively to the console. - - Internal messages are used for logging system states or debug information - that should not be written to log files. - These messages are only displayed in the console when color logging is enabled. - - Parameters: - message (str): The internal message to be logged. - If the message is "None" or None, no logging occurs. - - Notes: - - Requires color logging to be enabled - - Uses a custom internal log level - - Converts the message to a string before logging - """ - if self.color and message != "None" and message is not None: - colorlog.log(self.INTERNAL_LOG_LEVEL, str(message)) - - def debug(self, message): - """ - Logs a debug message. - - :param message: The debug message to be logged. - """ - if self.color and message != "None" and message is not None: - colorlog.debug(str(message)) - - def _raw(self, message): - """ - Log a raw message directly to the log file. - - This method writes a message directly to the log file without any additional formatting - or logging levels. - - WARNING: This method is for internal use only! Using it directly can mess up - your log file format and make it hard to read. Use info(), debug(), or - other public methods instead. - - Parameters: - message (str): The raw message to be written to the log file. - - Notes: - - Checks the calling context to warn about non-function calls - - Handles potential Unicode encoding errors - - Skips logging if message is None or "None" - - Writes message with a newline character - - Logs internal errors if file writing fails - - Raises: - Logs internal errors for Unicode or file writing issues without stopping execution - """ - frame = inspect.currentframe().f_back - if frame and frame.f_code.co_name == "": - self.__internal( - f"Raw message called from a non-function - This is not recommended" - ) - # Precompiled regex for ANSI escape codes - # Remove all ANSI escape sequences in one pass - message = re.compile(r"\033\[\d+(;\d+)*m").sub("", message) - - if message and message != "None": - try: - with open(self.filename, "a", encoding="utf-8") as f: - f.write(f"{str(message)}\n") - except (UnicodeDecodeError, UnicodeEncodeError) as UDE: - self.__internal(f"UnicodeDecodeError: {UDE} - Message: {str(message)}") - except Exception as E: - self.__internal(f"Error: {E} - Message: {str(message)}") - - def newline(self): - """ - Write a newline separator to the log file, creating a visual divider between log entries. - - This method writes a formatted horizontal line to the log file using ASCII characters, - which helps visually separate different sections or log entries. - The line consists of vertical bars and dashes creating a structured tabular-like separator. - - Side Effects: - Appends a newline separator to the log file specified by `self.filename`. - """ - with open(self.filename, "a") as f: - f.write("|" + "-" * 19 + "|" + "-" * 13 + "|" + "-" * 154 + "|" + "\n") - - def info(self, message): - """ - Logs an info message. - - :param message: The info message to be logged. - """ - if self.color and message != "None" and message is not None: - colorlog.info(str(message)) - self._raw( - f"[{self.__timestamp()}] > INFO: | {self.__trunc_message(str(message))}" - ) - - def warning(self, message): - """ - Logs a warning message. - - :param message: The warning message to be logged. - """ - if self.color and message != "None" and message is not None: - colorlog.warning(str(message)) - self._raw( - f"[{self.__timestamp()}] > WARNING: | {self.__trunc_message(str(message))}" - ) - - def error(self, message): - """ - Logs an error message. - - :param message: The error message to be logged. - """ - if self.color and message != "None" and message is not None: - colorlog.error(str(message)) - self._raw( - f"[{self.__timestamp()}] > ERROR: | {self.__trunc_message(str(message))}" - ) - - def critical(self, message): - """ - Logs a critical message. - - :param message: The critical message to be logged. - """ - if self.color and message != "None" and message is not None: - colorlog.critical(str(message)) - self._raw( - f"[{self.__timestamp()}] > CRITICAL: | {self.__trunc_message(str(message))}" - ) - - def string(self, message, type: str): - """ - Logs a message with a specified log type, supporting multiple type aliases. - - This method allows logging messages with flexible type specifications, - mapping aliases to standard log types and handling potential errors in type selection. - It supports logging with color if enabled. - - Parameters: - message (str): The message to be logged. Skipped if "None" or None. - type (str): The log type, which can be one of: - - Standard types: 'debug', 'info', 'warning', 'error', 'critical' - - Aliases: 'err' (error), 'warn' (warning), 'crit' (critical), 'except' (exception) - - Behavior: - - Converts type to lowercase and maps aliases to standard log types - - Logs message using the corresponding log method - - Falls back to debug logging if an invalid type is provided - - Only logs if color is enabled and message is not "None" - - Raises: - AttributeError: If no matching log method is found (internally handled) - """ - if self.color and message != "None" and message is not None: - type_map = { - "err": "error", - "warn": "warning", - "crit": "critical", - "except": "exception", - } - type = type_map.get(type.lower(), type) - try: - getattr(self, type.lower())(str(message)) - except AttributeError as AE: - self.__internal( - f"A wrong Log Type was called: {type} not found. -> {AE}" - ) - getattr(self, "Debug".lower())(str(message)) - - def exception(self, message, exception_type: Type = Exception): - """ - Log an exception message and raise the specified exception. - - Warning: Not recommended for production use. Prefer Log().error() for logging exceptions. - - Args: - message (str): The exception message to be logged. - exception_type (Type, optional): The type of exception to raise. Defaults to Exception. - - Raises: - The specified exception type with the provided message. - - Note: - - Only logs the exception if color logging is enabled and message is not None - - Logs the exception with a timestamp and truncated message - - Includes both the original message and the exception type in the log - """ - if self.color and message != "None" and message is not None: - self._raw( - f"[{self.__timestamp()}] > EXCEPTION:| {self.__trunc_message(f'{message} -> Exception provoked: {str(exception_type)}')}" - ) - raise exception_type(message) - - def execution(self, message_log: list[tuple[str, str]]): - """ - Parse and log multiple messages with their corresponding log types. - - This method processes a list of messages, where each message is associated with a specific log type. It is designed for scenarios where multiple log entries need to be processed simultaneously, such as logging script execution results. - - Parameters: - message_log (list[tuple[str, str]]): A list of message entries. - Each entry is a list containing two elements: - - First element: The log message (str) - - Second element: The log type (str) - - Behavior: - - Iterates through the provided message log - - Logs each message using the specified log type via `self.string()` - - Logs an internal warning if a message list does not contain exactly two elements - - Example: - log = Log() - log.parse_execution([ - ['Operation started', 'info'], - ['Processing data', 'debug'], - ['Completed successfully', 'info'] - ]) - """ - if message_log: - for message_list in message_log: - if len(message_list) == 2: - self.string(message_list[0], message_list[1]) - else: - self.__internal( - f"Message List is not in the correct format: {message_list}" - ) - - def function(self, func: callable): - """ - A decorator that logs the execution details of a function, tracking its performance and providing runtime insights. - - Parameters: - func (callable): The function to be decorated and monitored. - - Returns: - callable: A wrapper function that logs execution metrics. - - Raises: - TypeError: If the provided function is not callable. - - Example: - @log.function - def example_function(): - # Function implementation - pass - """ - if not callable(func): - self.exception(f"Function {func.__name__} is not callable.", TypeError) - - def wrapper(*args, **kwargs): - """ - Wrapper function that logs the execution of the decorated function. - - Tracks and logs the start, execution, and completion of a function with performance timing. - - Parameters: - *args (tuple): Positional arguments passed to the decorated function. - **kwargs (dict): Keyword arguments passed to the decorated function. - - Returns: - Any: The original result of the decorated function. - - Raises: - TypeError: If the decorated function is not callable. - - Notes: - - Logs debug messages before and after function execution - - Measures and logs the total execution time with microsecond precision - - Preserves the original function's return value - """ - start_time = time.perf_counter() - func_args = ", ".join( - [str(arg) for arg in args] + [f"{k}={v}" for k, v in kwargs.items()] - ) - self.debug(f"Running the function {func.__name__}({func_args}).") - result = func(*args, **kwargs) - end_time = time.perf_counter() - elapsed_time = end_time - start_time - self.debug( - f"{func.__name__}({func_args}) executed in {elapsed_time} -> returned {type(result).__name__}" - ) - return result - - return wrapper - - -log = Log({"log_level": DEBUG}) diff --git a/CODE/logicytics/__init__.py b/CODE/logicytics/__init__.py deleted file mode 100644 index 2e540e0c..00000000 --- a/CODE/logicytics/__init__.py +++ /dev/null @@ -1,126 +0,0 @@ -import functools -import traceback - -from logicytics.Checks import Check -from logicytics.Config import DEBUG, VERSION, CURRENT_FILES, DELETE_LOGS, config -from logicytics.Execute import Execute -from logicytics.FileManagement import FileManagement -from logicytics.Flag import Flag -from logicytics.Get import Get -from logicytics.Logger import log, Log - -# Check if the script is being run directly, if not, set up the library -if __name__ == "__main__": - exit("This is a library, Please import rather than directly run.") - -execute = Execute() # Initialize the Execute class for executing commands -get = Get() # Initialize the Get class for retrieving data -check = Check() # Initialize the Check class for performing checks -flag = Flag() # Initialize the Flag class for managing cli flags -file_management = ( - FileManagement() -) # Initialize the FileManagement class for file operations -__show_trace = ( - DEBUG == "DEBUG" -) # Determine if stack traces should be shown based on the debug level - - -# Exception for handling object loading errors -class ObjectLoadError(Exception): - """Raised when an Object fails to load.""" - - def __init__(self, message="Failed to load object", object_name=None): - """ - Initialize the exception with a custom message and object details. - - Args: - message (str): The error message - object_name (str, optional): Name of the object that failed to load - """ - self.object_name = object_name - if object_name: - message = f"{message} (Object: {object_name})" - super().__init__(message) - - -# Decorator for marking functions as deprecated [custom] -def deprecated( - removal_version: str, reason: str, show_trace: bool = __show_trace -) -> callable: - """ - Decorator function that marks a function as deprecated - and provides a warning when the function is called. - - Args: - removal_version (str): The version when the function will be removed. - reason (str): The reason for deprecation. - show_trace (bool): Whether to show the stack trace when the function is called. Default is based on DEBUG set by user. - - Returns: - callable: A decorator that marks a function as deprecated. - - Notes: - - Uses a nested decorator function to preserve the original function's metadata - - Prints a colorized deprecation warning - """ - - def decorator(func: callable) -> callable: - """ - Decorator function that marks a function as deprecated and provides a warning when the function is called. - - Args: - func (callable): The function to be decorated with a deprecation warning. - - Returns: - callable: A wrapper function that preserves the original function's metadata and prints a deprecation warning. - - Notes: - - Uses functools.wraps to preserve the original function's metadata - - Prints a colorized deprecation warning to stderr - - Allows the original function to continue executing - """ - - @functools.wraps(func) - def wrapper(*args, **kwargs) -> callable: - """ - Wraps a deprecated function to print a warning message before execution. - - Args: - *args: Positional arguments passed to the original function. - **kwargs: Keyword arguments passed to the original function. - - Returns: - The return value of the original function after printing a deprecation warning. - - Warns: - Prints a colored deprecation warning to stderr with details about: - - Function name being deprecated - - Reason for deprecation - - Version when the function will be removed - """ - message = f"\033[91mDeprecationWarning: A call to the deprecated function {func.__name__}() has been called, {reason}. Function will be removed at version {removal_version}\n" - if show_trace: - stack = "".join(traceback.format_stack()[:-1]) - message += f"Called from:\n{stack}\033[0m" - else: - message += "\033[0m" - print(message) - return func(*args, **kwargs) - - return wrapper - - return decorator - - -__all__ = [ - "execute", - "get", - "check", - "flag", - "file_management", - "deprecated", - "ObjectLoadError", - "log", - "Log", - "config", -] diff --git a/CODE/media_backup.py b/CODE/media_backup.py deleted file mode 100644 index e2480e0f..00000000 --- a/CODE/media_backup.py +++ /dev/null @@ -1,138 +0,0 @@ -import getpass -import os -import shutil -from datetime import datetime - -from logicytics import log - - -class Media: - """ - A class to handle media backup operations. - """ - - @staticmethod - def __get_default_paths() -> list: - """ - Returns the default paths for photos and videos based on the Windows username. - - This method retrieves the current Windows user's default media directories for photos and videos - by using the current username and standard Windows file system paths. - - Returns: - list: A list containing two paths: - - First element: Default photo directory path - - Second element: Default video directory path - - Notes: - - Uses `getpass.getuser()` to dynamically retrieve the current Windows username - - Expands the user path using `os.path.expanduser()` to handle potential path variations - - Assumes standard Windows user directory structure - """ - username = getpass.getuser() - default_photo_path = os.path.expanduser(f"C:\\Users\\{username}\\Pictures") - default_video_path = os.path.expanduser(f"C:\\Users\\{username}\\Videos") - return [default_photo_path, default_video_path] - - @staticmethod - def __ensure_backup_directory_exists(backup_directory: str): - """ - Ensures the backup directory exists, creating it if necessary. - - Args: - backup_directory (str): The full path to the directory where media files will be backed up. - - Raises: - OSError: If the directory cannot be created due to permission issues or other system constraints. - """ - if not os.path.exists(backup_directory): - os.makedirs(backup_directory) - - @staticmethod - def __collect_media_files(source_dirs: list) -> list: - """ - Collects media files from specified source directories. - - Recursively searches through the provided source directories to find image and video files with extensions .jpg, .jpeg, .png, and .mp4. - - Args: - source_dirs (list): List of directory paths to search for media files. - - Returns: - list: Absolute file paths of all discovered media files, including those in subdirectories. - - Raises: - OSError: If any of the source directories are inaccessible or cannot be traversed. - """ - media_files = [] - for source_dir in source_dirs: - for root, _, files in os.walk(source_dir): - for file in files: - if file.endswith((".jpg", ".jpeg", ".png", ".mp4")): - media_files.append(os.path.join(root, file)) - return media_files - - @staticmethod - def __backup_files(media_files: list, backup_directory: str): - """ - Copies media files to a backup directory with timestamped filenames. - - Parameters: - media_files (list): A list of file paths for media files to be backed up. - backup_directory (str): Destination directory path for storing backup files. - - Behavior: - - Iterates through each media file in the input list - - Generates a new filename with current timestamp - - Attempts to copy each file to the backup directory - - Logs successful copy operations - - Logs any errors encountered during file copying - - Exceptions: - Handles and logs any exceptions that occur during file copy process - Does not interrupt the entire backup process if a single file copy fails - """ - for src_file in media_files: - dst_file = os.path.join( - backup_directory, - datetime.now().strftime("%Y-%m-%d_%H-%M-%S") - + "_" - + os.path.basename(src_file), - ) - try: - shutil.copy2(str(src_file), str(dst_file)) - log.info(f"Copied {os.path.basename(src_file)} to {dst_file}") - except Exception as e: - log.error(f"Failed to copy {src_file}: {str(e)}") - - @classmethod - @log.function - def backup(cls): - """ - Orchestrates the complete media backup process by performing sequential backup operations. - - This class method coordinates the backup workflow: - 1. Retrieves default media source directories - 2. Sets a standard backup directory - 3. Ensures the backup directory exists - 4. Collects media files from source directories - 5. Copies media files to the backup directory - 6. Logs the completion of the backup process - - Returns: - None: Performs backup operations without returning a value - - Raises: - OSError: If directory creation or file operations fail - PermissionError: If insufficient permissions for file/directory operations - """ - source_dirs = cls.__get_default_paths() - backup_directory = "MediaBackup" - cls.__ensure_backup_directory_exists(backup_directory) - media_files = cls.__collect_media_files(source_dirs) - cls.__backup_files(media_files, backup_directory) - log.info("Media backup script completed.") - - -if __name__ == "__main__": - Media.backup() diff --git a/CODE/netadapter.ps1 b/CODE/netadapter.ps1 deleted file mode 100644 index 9453acea..00000000 --- a/CODE/netadapter.ps1 +++ /dev/null @@ -1,3 +0,0 @@ -# Get all network details -Write-Output "INFO: Getting NetAdapter Info" -Get-NetAdapter | Select-Object Name, Status, MacAddress, ifIndex, InterfaceAlias, InterfaceDescription | Out-File -FilePath .\Network.txt diff --git a/CODE/network_psutil.py b/CODE/network_psutil.py deleted file mode 100644 index 2fee2afe..00000000 --- a/CODE/network_psutil.py +++ /dev/null @@ -1,195 +0,0 @@ -import asyncio -import os -import socket - -import psutil - -from logicytics import log, execute, config - - -class NetworkInfo: - """ - A class to gather and save various network-related information. - """ - - def __init__(self): - self.SAMPLE_COUNT = config.getint("NetWorkPsutil Settings", "sample_count") - self.INTERVAL = config.getfloat("NetWorkPsutil Settings", "interval") - - @log.function - async def get(self): - """ - Gathers and saves various network-related information by calling multiple internal methods. - """ - try: - self.__fetch_network_io_stats() - self.__fetch_network_connections() - self.__fetch_network_interface_addresses() - self.__fetch_network_interface_stats() - self.__execute_external_network_command() - self.__fetch_network_connections_with_process_info() - await self.__measure_network_bandwidth_usage(sample_count=self.SAMPLE_COUNT, interval=self.INTERVAL) - self.__fetch_hostname_and_ip() - except Exception as e: - log.error(f"Error getting network info: {e}, Type: {type(e).__name__}") - - @staticmethod - def __save_data(filename: str, data: str, father_dir_name: str = "network_data"): - """ - Saves the given data to a file. - - :param filename: The name of the file to save the data in. - :param data: The data to be saved. - :param father_dir_name: The directory to save the file in. Defaults to "network_data". - """ - os.makedirs(father_dir_name, exist_ok=True) - try: - with open(os.path.join(father_dir_name, filename), "w") as f: - f.write(data) - except IOError as e: - log.error(f"Failed to save {filename}: {e}") - - def __fetch_network_io_stats(self): - """ - Fetches and saves network I/O statistics for each network interface. - """ - log.debug("Fetching network interface stats...") - net_io = psutil.net_io_counters(pernic=True) - net_io_data = "" - for iface, stats in net_io.items(): - net_io_data += f"Interface: {iface}\n" - net_io_data += f"Bytes Sent: {stats.bytes_sent}, Bytes Received: {stats.bytes_recv}\n" - net_io_data += f"Packets Sent: {stats.packets_sent}, Packets Received: {stats.packets_recv}\n" - net_io_data += f"Errors In: {stats.errin}, Errors Out: {stats.errout}\n" - net_io_data += f"Dropped In: {stats.dropin}, Dropped Out: {stats.dropout}\n\n" - self.__save_data("network_io.txt", net_io_data) - log.info("Network IO stats saved.") - - def __fetch_network_connections(self): - """ - Fetches and saves information about network connections. - """ - log.debug("Fetching network connections...") - connections = psutil.net_connections(kind='all') - connections_data = "" - for conn in connections: - connections_data += f"Type: {conn.type}, Local: {conn.laddr}, Remote: {conn.raddr}, Status: {conn.status}\n" - self.__save_data("network_connections.txt", connections_data) - log.info("Network connections saved.") - - def __fetch_network_interface_addresses(self): - """ - Fetches and saves network interface addresses. - """ - log.debug("Fetching network interface addresses...") - interfaces = psutil.net_if_addrs() - interfaces_data = "" - for iface, addrs in interfaces.items(): - for addr in addrs: - interfaces_data += f"Interface: {iface}, Address: {addr.address}, Netmask: {addr.netmask}, Broadcast: {addr.broadcast}\n" - self.__save_data("network_interfaces.txt", interfaces_data) - log.info("Network interface addresses saved.") - - def __fetch_network_interface_stats(self): - """ - Fetches and saves network interface statistics. - """ - log.debug("Fetching network interface stats...") - stats = psutil.net_if_stats() - stats_data = "" - for iface, stat in stats.items(): - stats_data += f"Interface: {iface}, Speed: {stat.speed}Mbps, Duplex: {stat.duplex}, Up: {stat.isup}\n" - self.__save_data("network_stats.txt", stats_data) - log.info("Network interface stats saved.") - - def __execute_external_network_command(self): - """ - Executes an external network command and saves the output. - """ - log.debug("Executing external network command...") - result = execute.command("ipconfig") - self.__save_data("network_command_output.txt", result) - log.info("Network command output saved.") - - def __fetch_network_connections_with_process_info(self): - """ - Fetches and saves network connections along with associated process information. - """ - log.debug("Fetching network connections with process info...") - connections_data = "" - for conn in psutil.net_connections(kind='all'): - pid = conn.pid if conn.pid else "N/A" - proc_name = "Unknown" - if pid != "N/A": - try: - proc_name = psutil.Process(pid).name() - except psutil.NoSuchProcess: - proc_name = "Process Exited" - connections_data += f"Type: {conn.type}, Local: {conn.laddr}, Remote: {conn.raddr}, Status: {conn.status}, Process: {proc_name} (PID: {pid})\n" - self.__save_data("network_connections_with_processes.txt", connections_data) - log.info("Network connections with process info saved.") - - async def __measure_network_bandwidth_usage(self, sample_count: int = 5, interval: float = 1.0): - """ - Measures and saves the average network bandwidth usage. - - Args: - sample_count: Number of samples to take (default: 5) - interval: Time between samples in seconds (default: 1.0) - """ - if sample_count < 1 or interval <= 0: - log.critical( - "Invalid values passed down from configuration for `NetworkInfo.__measure_network_bandwidth_usage()`") - log.debug("Measuring network bandwidth usage...") - samples = [] - for _ in range(sample_count): - net1 = psutil.net_io_counters() - await asyncio.sleep(interval) - net2 = psutil.net_io_counters() - samples.append({ - 'up': (net2.bytes_sent - net1.bytes_sent) / 1024, - 'down': (net2.bytes_recv - net1.bytes_recv) / 1024 - }) - if samples: - avg_up = sum(s['up'] for s in samples) / len(samples) - avg_down = sum(s['down'] for s in samples) / len(samples) - max_up = max(s['up'] for s in samples) - max_down = max(s['down'] for s in samples) - else: - avg_up = avg_down = max_up = max_down = 0 - bandwidth_data = f"Average Upload Speed: {avg_up:.2f} KB/s\n" - bandwidth_data += f"Average Download Speed: {avg_down:.2f} KB/s\n" - bandwidth_data += f"Peak Upload Speed: {max_up:.2f} KB/s\n" - bandwidth_data += f"Peak Download Speed: {max_down:.2f} KB/s\n" - self.__save_data("network_bandwidth_usage.txt", bandwidth_data) - log.info("Network bandwidth usage saved.") - - def __fetch_hostname_and_ip(self): - """ - Fetches and saves the hostname and IP addresses of the machine. - """ - try: - hostname = socket.gethostname() - ip_addresses = [] - for res in socket.getaddrinfo(hostname, None): - ip = res[4][0] - if ip not in ip_addresses: - ip_addresses.append(ip) - ip_config_data = f"Hostname: {hostname}\n" - ip_config_data += "IP Addresses:\n" - for ip in ip_addresses: - ip_config_data += f" - {ip}\n" - except socket.gaierror as e: - log.error(f"Failed to resolve hostname: {e}") - ip_config_data = f"Hostname: {hostname}\nFailed to resolve IP addresses\n" - self.__save_data("hostname_ip.txt", ip_config_data) - log.info("Hostname and IP address saved.") - - -if __name__ == "__main__": - try: - asyncio.run(NetworkInfo().get()) # Use asyncio.run to run the async get method - except asyncio.CancelledError: - log.warning("Operation cancelled by user.") - except Exception as err: # Catch all exceptions - log.error(f"An error occurred: {err}") diff --git a/CODE/packet_sniffer.py b/CODE/packet_sniffer.py deleted file mode 100644 index 95d85627..00000000 --- a/CODE/packet_sniffer.py +++ /dev/null @@ -1,149 +0,0 @@ -from __future__ import annotations - -import warnings -from time import time - -import matplotlib.pyplot as plt -import networkx as nx -import pandas as pd -from cryptography.utils import CryptographyDeprecationWarning - -# TripleDES deprecation warning -warnings.filterwarnings("ignore", category=CryptographyDeprecationWarning) -warnings.filterwarnings("ignore", category=DeprecationWarning) - -from scapy.all import sniff, conf -from scapy.layers.inet import IP, TCP, UDP, ICMP - -from logicytics import log, config - - -class PacketSniffer: - def __init__(self): - conf.verb = 0 - self.packet_data = [] - self.G = nx.Graph() - - @staticmethod - def _get_protocol(packet: IP) -> str: - if packet.haslayer(TCP): - return "TCP" - elif packet.haslayer(UDP): - return "UDP" - elif packet.haslayer(ICMP): - return "ICMP" - return "Other" - - @staticmethod - def _get_port(packet: IP, port_type: str) -> int | None: - if port_type == "sport": - return getattr(packet[TCP], "sport", None) if packet.haslayer(TCP) else getattr(packet[UDP], "sport", None) - elif port_type == "dport": - return getattr(packet[TCP], "dport", None) if packet.haslayer(TCP) else getattr(packet[UDP], "dport", None) - return None - - def _log_packet(self, packet: IP): - if not packet.haslayer(IP): - return - - try: - protocol = self._get_protocol(packet) - src_ip = packet[IP].src - dst_ip = packet[IP].dst - - src_port = dst_port = None - if protocol in ("TCP", "UDP"): - src_port = self._get_port(packet, "sport") - dst_port = self._get_port(packet, "dport") - - info = { - "src_ip": src_ip, - "dst_ip": dst_ip, - "protocol": protocol, - "src_port": src_port, - "dst_port": dst_port - } - - self.packet_data.append(info) - self.G.add_edge(src_ip, dst_ip, protocol=protocol) - log.debug(f"{protocol} {src_ip}:{src_port} -> {dst_ip}:{dst_port}") - except Exception as err: - log.error(f"Error logging packet: {err}") - - def _save_to_csv(self, path: str): - if not self.packet_data: - log.warning("No packets to save.") - return - pd.DataFrame(self.packet_data).to_csv(path, index=False) - log.info(f"Saved packet data to {path}") - - def _visualize_graph(self, output: str = "graph.png"): - if self.G.number_of_edges() == 0: - log.warning("No edges to plot in graph.") - return - - pos = nx.spring_layout(self.G) - plt.figure(figsize=(12, 8)) - nx.draw(self.G, pos, with_labels=True, node_color="skyblue", node_size=3000, font_size=10, font_weight="bold") - labels = nx.get_edge_attributes(self.G, 'protocol') - nx.draw_networkx_edge_labels(self.G, pos, edge_labels=labels) - plt.title("Network Graph") - plt.savefig(output) - plt.close() - log.info(f"Graph saved to {output}") - - @staticmethod - def _correct_interface(iface: str) -> str: - corrections = {"WiFi": "Wi-Fi", "Wi-Fi": "WiFi"} - return corrections.get(iface, iface) - - def sniff_packets(self, iface: str, count: int, timeout: int, retry_max: int): - iface = self._correct_interface(iface) - retry_start = time() - - while time() - retry_start < retry_max: - try: - log.info(f"Sniffing on {iface}... (count={count}, timeout={timeout})") - sniff( - iface=iface, - prn=self._log_packet, - count=count, - timeout=timeout - ) - log.info("Sniff complete.") - break - except Exception as err: - log.warning(f"Sniff failed on {iface}: {err}") - iface = self._correct_interface(iface) - else: - log.error("Max retry time exceeded.") - - self._save_to_csv("packets.csv") - self._visualize_graph() - - def run(self): - iface = config.get("PacketSniffer Settings", "interface", fallback="WiFi") - count = config.getint("PacketSniffer Settings", "packet_count", fallback=5000) - timeout = config.getint("PacketSniffer Settings", "timeout", fallback=10) - retry_max = config.getint("PacketSniffer Settings", "max_retry_time", fallback=30) - - if count <= 0 or timeout < 5 or retry_max < timeout: - log.critical("Invalid configuration values.") - return - - self.sniff_packets(iface, count, timeout, retry_max) - - def cleanup(self): - self.G.clear() - plt.close("all") - log.info("Cleanup complete.") - - -if __name__ == "__main__": - sniffer = PacketSniffer() - try: - sniffer.run() - except Exception as e: - log.error(f"Fatal error: {e}") - finally: - sniffer.cleanup() diff --git a/CODE/property_scraper.ps1 b/CODE/property_scraper.ps1 deleted file mode 100644 index 93ef3524..00000000 --- a/CODE/property_scraper.ps1 +++ /dev/null @@ -1,34 +0,0 @@ -# Collect system information -$buildNumber = [System.Environment]::OSVersion.Version.Build -$physicalMemory = [System.Diagnostics.Process]::PhysicalMemorySize64 / 1MB -$virtualMemory = [System.Diagnostics.Process]::WorkingSet64 / 1MB -$userName = [System.Security.Principal.WindowsIdentity]::GetCurrent().Name -$userSid = ([System.Security.Principal.WindowsIdentity]::GetCurrent().UserValue) -$userLanguageId = [System.Globalization.CultureInfo]::CurrentCulture.LCID -$computerName = [System.Net.Dns]::GetHostName() -$systemLanguageId = [System.Globalization.CultureInfo]::CurrentUICulture.LCID -$time = Get-Date -Format HH:mm:ss -$date = Get-Date -Format dd/MM/yyyy -$rootDrive = $env:SystemDrive - -# Prepare the data to be written to the File -$data = @" -Property(C): Windows Build = $buildNumber -Property(C): Physical Memory = $( $physicalMemory -as [int] ) -Property(C): Virtual Memory = $( $virtualMemory -as [int] ) -Property(C): Log on User = $userName -Property(C): User SID = $userSid -Property(C): User Language ID = $userLanguageId -Property(C): Computer Name = $computerName -Property(C): System Language ID = $systemLanguageId -Property(C): Time = $time -Property(C): Date = $date -Property(C): Username = $userName -Property(C): Root Drive = $rootDrive -"@ - -# Write the data to a text File -$data | Out-File -FilePath ".\Extra_Data.txt" - -# Optionally, display a message indicating success -Write-Host "INFO: Data successfully written to Extra_Data.txt" diff --git a/CODE/registry.py b/CODE/registry.py deleted file mode 100644 index e1fa3522..00000000 --- a/CODE/registry.py +++ /dev/null @@ -1,29 +0,0 @@ -import os -import subprocess - -from logicytics import log - - -@log.function -def backup_registry(): - """ - Backs up the Windows registry to a file named 'RegistryBackup.reg' in the current working directory. - - This function uses the reg export command to export the entire - registry (HKEY_LOCAL_MACHINE) and logs the result. - """ - export_path = os.path.join(os.getcwd(), "RegistryBackup.reg") - reg_path = r"C:\Windows\System32\reg.exe" - cmd = [reg_path, "export", "HKLM", export_path] - - try: - result = subprocess.run(cmd, check=True, capture_output=True, text=True) - log.info(f"Registry backed up successfully to {export_path}. Output: {result.stdout}") - except subprocess.CalledProcessError as e: - log.error(f"Failed to back up the registry: {e}.") - except Exception as e: - log.error(f"Failed to back up the registry: {e}") - - -if __name__ == "__main__": - backup_registry() diff --git a/CODE/sensitive_data_miner.py b/CODE/sensitive_data_miner.py deleted file mode 100644 index 89b92324..00000000 --- a/CODE/sensitive_data_miner.py +++ /dev/null @@ -1,179 +0,0 @@ -import os -import shutil -from concurrent.futures import ThreadPoolExecutor - -from pathlib import Path - -from logicytics import log - -# List of allowed extensions -allowed_extensions = [ - ".png", ".txt", ".md", ".json", ".yaml", ".secret", ".jpg", ".jpeg", - ".password", ".text", ".docx", ".doc", ".xls", ".xlsx", ".csv", - ".xml", ".config", ".log", ".pdf", ".zip", ".rar", ".7z", ".tar", - ".gz", ".tgz", ".tar.gz", ".tar.bz2", ".tar.xz", ".tar.zst", - ".sql", ".db", ".dbf", ".sqlite", ".sqlite3", ".bak", ".dbx", - ".mdb", ".accdb", ".pst", ".ost", ".msg", ".eml", ".vsd", - ".vsdx", ".vsdm", ".vss", ".vssx", ".vssm", ".vst", ".vstx", - ".vstm", ".vdx", ".vsx", ".vtx", ".vdw", ".vsw", ".vst", - ".mpp", ".mppx", ".mpt", ".mpd", ".mpx", ".mpd", ".mdf", -] - - -class Mine: - @staticmethod - def __search_files_by_keyword(root: Path, keyword: str) -> list: - """ - Searches for files containing a specified keyword in their names within a given directory. - - Parameters: - root (Path): The root directory to search in for files. - keyword (str): The keyword to search for in file names (case-insensitive). - - Returns: - list: A list of file paths matching the search criteria, which: - - Contain the keyword in their filename (case-insensitive) - - Are files (not directories) - - Have file extensions in the allowed_extensions list - - Raises: - WindowsError: If permission is denied when accessing the directory (logged as a warning in debug mode) - - Notes: - - Skips files with unsupported extensions, logging debug information - - Uses case-insensitive keyword matching - """ - matching_files = [] - path_list = [] - try: - path_list = os.listdir(root) - except (WindowsError, PermissionError) as e: - log.warning(f"Permission Denied: {e}") - except Exception as e: - log.error(f"Failed to access directory: {e}") - - for filename in path_list: - file_path = root / filename - if ( - keyword.lower() in filename.lower() - and file_path.is_file() - and file_path.suffix in allowed_extensions - ): - matching_files.append(file_path) - else: - log.debug(f"Skipped {file_path}, Unsupported due to {file_path.suffix} extension") - return matching_files - - @staticmethod - def __copy_file(src_file_path: Path, dst_file_path: Path): - """ - Copy a file from the source path to the destination path. - - Parameters: - src_file_path (Path): The full path of the source file to be copied. - dst_file_path (Path): The full path where the file will be copied. - - Raises: - FileExistsError: If a file already exists at the destination path. - Exception: For any other unexpected errors during file copying. - - Notes: - - Uses shutil.copy() for file copying - - Logs debug message on successful copy - - Logs warning if file already exists - - Logs error for any other copying failures - """ - try: - # Check file size and permissions - if src_file_path.stat().st_size > 10_000_000: # 10MB limit - log.warning("File exceeds size limit") - return - shutil.copy(src_file_path, dst_file_path) - log.debug(f"Copied {src_file_path} to {dst_file_path}") - except FileExistsError as e: - log.warning(f"File already exists in destination: {e}") - except Exception as e: - log.error(f"Failed to copy file: {e}") - - @classmethod - def __search_and_copy_files(cls, keyword: str): - """ - Searches for files containing the specified keyword in their names and copies them to a destination directory. - - This method performs a comprehensive file search across the C: drive, identifying files that match a given keyword and concurrently copying them to a dedicated "Password_Files" directory. - - Parameters: - keyword (str): The keyword to search for in file names. Used to filter and identify potentially sensitive files. - - Side Effects: - - Creates a "Password_Files" directory if it does not exist - - Logs informational messages about the search and copy process - - Utilizes multithreading to efficiently search and copy files - - Notes: - - Searches recursively through all directories starting from C:\ - - Uses ThreadPoolExecutor for concurrent file searching and copying - - Handles potential permission and file access errors during search - """ - log.info(f"Searching/Copying file's with keyword: {keyword}") - drives_root = Path("C:\\") - destination = Path("Password_Files") - if not destination.exists(): - destination.mkdir() - - with ThreadPoolExecutor() as executor: - for root, dirs, _ in os.walk(drives_root): - future_to_file = { - executor.submit(cls.__search_files_by_keyword, Path(root) / sub_dir, keyword): sub_dir - for sub_dir in dirs - } - for future in future_to_file: - for file_path in future.result(): - dst_file_path = destination / file_path.name - executor.submit(cls.__copy_file, file_path, dst_file_path) - - @classmethod - @log.function - def passwords(cls): - """ - Searches for and copies files containing sensitive data keywords to a dedicated directory. - - This method performs a comprehensive search for files with predefined sensitive keywords in their names, - copying matching files to a "Password_Files" directory. It handles directory cleanup and uses predefined - keywords related to sensitive information. - - Side Effects: - - Creates or recreates the "Password_Files" directory - - Copies files matching sensitive keywords to the destination directory - - Logs the completion of the sensitive data mining process - - Keywords Searched: - - "password" - - "secret" - - "code" - - "login" - - "api" - - "key" - - Logging: - - Logs an informational message upon completion of the search and copy process - """ - keywords = ["password", "secret", "code", "login", "api", "key", - "token", "auth", "credentials", "private", "cert", "ssh", "pgp", "wallet"] - - # Ensure the destination directory is clean - destination = Path("Password_Files") - if destination.exists(): - shutil.rmtree(destination) - destination.mkdir() - - for word in keywords: - cls.__search_and_copy_files(word) - - log.info("Sensitive Data Miner Completed") - - -if __name__ == "__main__": - log.warning( - "Sensitive Data Miner Initialized. Processing may take a while... (Consider a break: coffee or fresh air recommended!)") - Mine.passwords() diff --git a/CODE/ssh_miner.py b/CODE/ssh_miner.py deleted file mode 100644 index 567cc539..00000000 --- a/CODE/ssh_miner.py +++ /dev/null @@ -1,47 +0,0 @@ -import os -import shutil - -from logicytics import log - - -@log.function -def ssh_miner(): - """ - This function backs up SSH keys and configuration - by copying them from the default SSH directory to a subdirectory - named 'ssh_backup' in the current working directory. - - Returns: - None - """ - # Get the current working directory - current_dir = os.getcwd() - - # Define the path to the SSH directory - ssh_folder = os.path.join(os.environ["USERPROFILE"], ".ssh") - - # Define the destination directory as the current working directory - destination_dir = current_dir - - # Ensure the destination directory exists - if not os.path.exists(destination_dir): - os.makedirs(destination_dir) - - # Define source and destination directories - source_dir = ssh_folder - destination_dir = os.path.join( - current_dir, "ssh_backup" - ) # Use a subdirectory named 'ssh_backup' in the current directory - - # Copy SSH keys and config - try: - shutil.copytree(source_dir, destination_dir) - log.info("SSH keys and configuration backed up successfully.") - except Exception as e: - log.error(f"Failed to back up SSH keys and configuration: {e}") - - log.info("SSH Miner completed.") - - -if __name__ == "__main__": - ssh_miner() diff --git a/CODE/sys_internal.py b/CODE/sys_internal.py deleted file mode 100644 index da71bea3..00000000 --- a/CODE/sys_internal.py +++ /dev/null @@ -1,93 +0,0 @@ -import os -import subprocess - -from logicytics import log - -sys_internal_executables = [ - "psfile.exe", - "PsGetsid.exe", - "PsInfo.exe", - "pslist.exe", - "PsLoggedon.exe", - "psloglist.exe", -] - -# Check if the executables exist -sys_internal_executables = [ - exe for exe in sys_internal_executables - if os.path.exists(os.path.join("SysInternal_Suite", exe)) -] - - -@log.function -def sys_internal(): - """ - This function runs a series of system internal sys_internal_executables and logs their output. - - It iterates over a list of executable names, constructs the command to run each one, - captures the output, and writes it to a file named 'SysInternal.txt'. - - The function also logs information and warning messages for each executable, - including any errors that occur during execution. - """ - with open("SysInternal.txt", "a") as outfile: - # Iterate over each executable - for executable in sys_internal_executables: - try: - # Construct the command to run the executable - command = f"{os.path.join('SysInternal_Suite', executable)}" - - # Execute the command and capture the output - result = subprocess.run( - command, stdout=subprocess.PIPE, stderr=subprocess.PIPE - ) - - # Write the output to the File - outfile.write("-" * 190) - outfile.write(f"{executable} Output:\n{result.stdout.decode()}") - log.info(f"{executable}: Successfully executed") - - # Optionally, handle errors if any - if ( - result.stderr.decode() != "" - and result.returncode != 0 - and result.stderr.decode() is not None - ): - log.warning(f"{executable}: {result.stderr.decode()}") - outfile.write(f"{executable}:\n{result.stderr.decode()}") - - except Exception as e: - log.error(f"Error executing {executable}: {str(e)}") - outfile.write(f"Error executing {executable}: {str(e)}\n") - log.info("SysInternal Suite fully executed") - - -def check_sys_internal_dir() -> tuple[bool, bool]: - """ - Checks the existence of the 'SysInternal_Suite' directory and its contents. - - Returns: - tuple[bool, bool]: A tuple where the first element is True if any of the - sys_internal_executables exist in the 'SysInternal_Suite' directory, and - the second element is True if 'SysInternal_Suite.zip' exists in the directory. - """ - if os.path.exists("SysInternal_Suite"): - return any( - os.path.exists(f"SysInternal_Suite/{file}") - for file in sys_internal_executables - ), os.path.exists("SysInternal_Suite/SysInternal_Suite.zip") - else: - log.error( - "SysInternal_Suite cannot be found as a directory, force closing the sys_internal.py program, continuing Logicytics" - ) - return False, False - - -if __name__ == "__main__": - if check_sys_internal_dir()[0]: - sys_internal() - elif check_sys_internal_dir()[1]: - log.warning( - "Files are not found, They are still zipped, most likely due to a .ignore file being present, continuing Logicytics") - else: - log.error("Files are not found, The zip file is also missing!, continuing Logicytics") diff --git a/CODE/tasklist.py b/CODE/tasklist.py deleted file mode 100644 index 5f135453..00000000 --- a/CODE/tasklist.py +++ /dev/null @@ -1,33 +0,0 @@ -import subprocess - -from logicytics import log - - -@log.function -def tasklist(): - """ - Retrieves a list of running tasks on the system and exports the result to a CSV file. - - Parameters: - None - - Returns: - None - """ - try: - result = subprocess.run( - "tasklist /v /fo csv", - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - ) - with open("tasks.csv", "wb") as file: - file.write(result.stdout) - log.info("Tasklist exported to tasks.csv") - except subprocess.CalledProcessError as e: - log.error(f"Subprocess Error: {e}") - except Exception as e: - log.error(f"Error: {e}") - - -if __name__ == "__main__": - tasklist() diff --git a/CODE/tree.ps1 b/CODE/tree.ps1 deleted file mode 100644 index d19fde86..00000000 --- a/CODE/tree.ps1 +++ /dev/null @@ -1,9 +0,0 @@ -Write-Host "INFO: Starting Tree Command" - -# Define the output file name as Tree.txt -$outputFile = "Tree.txt" - -# Run the tree command and redirect the output to the file -tree /f C:\ | Out-File -FilePath $outputFile -Force - -Write-Host "INFO: Saved $outputFile" \ No newline at end of file diff --git a/CODE/usb_history.py b/CODE/usb_history.py deleted file mode 100644 index 912823ce..00000000 --- a/CODE/usb_history.py +++ /dev/null @@ -1,89 +0,0 @@ -import ctypes -import os -import winreg -from datetime import datetime, timedelta - -from logicytics import log - - -class USBHistory: - def __init__(self): - self.history_path = os.path.join(os.path.dirname(os.path.abspath(__file__)), "usb_history.txt") - - def _save_history(self, message: str): - """Append a timestamped message to the history file and log it.""" - timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S") - entry = f"{timestamp} - {message}\n" - try: - with open(self.history_path, "a", encoding="utf-8") as f: - f.write(entry) - log.debug(f"Saved entry: {message}") - except Exception as e: - log.error(f"Failed to write history: {e}") - - # noinspection PyUnresolvedReferences - @staticmethod - def _get_last_write_time(root_key, sub_key_path): - """Return the precise last write time of a registry key, or None on failure.""" - handle = ctypes.wintypes.HANDLE() - try: - advapi32 = ctypes.windll.advapi32 - if advapi32.RegOpenKeyExW(root_key, sub_key_path, 0, winreg.KEY_READ, ctypes.byref(handle)) != 0: - return None - ft = ctypes.wintypes.FILETIME() - if advapi32.RegQueryInfoKeyW(handle, None, None, None, None, None, None, None, None, None, None, - ctypes.byref(ft)) != 0: - return None - t = ((ft.dwHighDateTime << 32) + ft.dwLowDateTime) // 10 - return datetime(1601, 1, 1) + timedelta(microseconds=t) - finally: - if handle: - ctypes.windll.advapi32.RegCloseKey(handle) - - @staticmethod - def _enum_subkeys(root, path, warn_func): - """Yield all subkeys of a registry key, logging warnings on errors.""" - try: - with winreg.OpenKey(root, path) as key: - subkey_count, _, _ = winreg.QueryInfoKey(key) - for i in range(subkey_count): - try: - yield winreg.EnumKey(key, i) - except OSError as e: - if getattr(e, "winerror", None) == 259: # ERROR_NO_MORE_ITEMS - break - warn_func(f"Error enumerating {path} index {i}: {e}") - except OSError as e: - warn_func(f"Failed to open registry key {path}: {e}") - - @staticmethod - def _get_friendly_name(dev_info_path, device_id): - """Return the friendly name of a device if available, else the device ID.""" - try: - with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, dev_info_path) as dev_key: - return winreg.QueryValueEx(dev_key, "FriendlyName")[0] - except FileNotFoundError: - return device_id - except Exception as e: - log.warning(f"Failed to read friendly name for {dev_info_path}: {e}") - return device_id - - def read(self): - """Read all USB devices from USBSTOR and log their info.""" - log.info("Starting USB history extraction...") - reg_path = r"SYSTEM\CurrentControlSet\Enum\USBSTOR" - try: - for device_class in self._enum_subkeys(winreg.HKEY_LOCAL_MACHINE, reg_path, log.warning): - dev_class_path = f"{reg_path}\\{device_class}" - for device_id in self._enum_subkeys(winreg.HKEY_LOCAL_MACHINE, dev_class_path, log.warning): - dev_info_path = f"{dev_class_path}\\{device_id}" - friendly_name = self._get_friendly_name(dev_info_path, device_id) - last_write = self._get_last_write_time(winreg.HKEY_LOCAL_MACHINE, dev_info_path) or "Unknown" - self._save_history(f"USB Device Found: {friendly_name} | LastWriteTime: {last_write}") - log.info(f"USB history extraction complete, saved to {self.history_path}") - except Exception as e: - log.error(f"Error during USB history extraction: {e}") - - -if __name__ == "__main__": - USBHistory().read() diff --git a/CODE/vulnscan.py b/CODE/vulnscan.py deleted file mode 100644 index b31a83ad..00000000 --- a/CODE/vulnscan.py +++ /dev/null @@ -1,196 +0,0 @@ -import csv -import json -import os -import shutil -from concurrent.futures import ThreadPoolExecutor, as_completed - -import torch -from sentence_transformers import SentenceTransformer -from torch import nn - -from logicytics import log, config - -# ================== GLOBAL SETTINGS ================== - -# File scan settings -TEXT_EXTENSIONS = { - ".txt", ".log", ".csv", ".json", ".xml", ".html", ".md", ".cfg", ".ini", ".yml", ".yaml", - ".rtf", ".tex", ".rst", ".adoc", ".properties", ".conf", ".bat", ".ps1", ".sh", ".tsv", - ".dat", ".env", ".toml", ".dockerfile", ".gitignore", ".gitattributes", ".npmrc", ".editorconfig" -} -MAX_TEXT_LENGTH = config.get("VulnScan Settings", "text_char_limit", fallback=None) -MAX_TEXT_LENGTH = int(MAX_TEXT_LENGTH) if MAX_TEXT_LENGTH not in (None, "None", "") else None -# Threading -NUM_WORKERS = config.get("VulnScan Settings", "max_workers", fallback="auto") -NUM_WORKERS = min(32, (os.cpu_count() or 1) * 2) if NUM_WORKERS == "auto" else int(NUM_WORKERS) -# Classification threshold -SENSITIVE_THRESHOLD = float( - config.get("VulnScan Settings", "threshold", fallback=0.6)) # Probability cutoff to consider a file sensitive - -# Paths -SENSITIVE_PATHS = [ - r"C:\Users\%USERNAME%\Documents", - r"C:\Users\%USERNAME%\Desktop", - r"C:\Users\%USERNAME%\Downloads", - r"C:\Users\%USERNAME%\AppData\Roaming", - r"C:\Users\%USERNAME%\AppData\Local", - r"C:\Users\%USERNAME%\OneDrive", - r"C:\Users\%USERNAME%\Dropbox", - r"C:\Users\%USERNAME%\Google Drive", -] -SAVE_DIR = r"VulnScan_Files" # Backup folder -MODEL_PATH = r"vulnscan/Model_SenseMacro.4n1.pth" # Your trained model checkpoint -REPORT_JSON = "report.json" -REPORT_CSV = "report.csv" - -# ================== DEVICE SETUP ================== -DEVICE = "cuda" if torch.cuda.is_available() else "cpu" -log.debug(f"Using device: {DEVICE}") - - -# ================== MODEL DEFINITION ================== -class SimpleNN(nn.Module): - def __init__(self, input_dim): - super().__init__() - self.fc = nn.Sequential( - nn.Linear(in_features=input_dim, out_features=256), - nn.ReLU(), - nn.Linear(in_features=256, out_features=64), - nn.ReLU(), - nn.Linear(in_features=64, out_features=1), - ) - - def forward(self, x): - return self.fc(x) - - -# ================== LOAD MODELS ================== -# Load classifier -checkpoint = torch.load(MODEL_PATH, map_location=DEVICE) -model = SimpleNN(input_dim=384) -model.load_state_dict(checkpoint["model_state_dict"]) -model.to(DEVICE) -model.eval() - -# Load embedding model -embed_model = SentenceTransformer("sentence-transformers/all-MiniLM-L6-v2", device=DEVICE) - -# Make backup folder -os.makedirs(SAVE_DIR, exist_ok=True) - - -# ================== FILE PROCESSING ================== -def process_file(filepath): - try: - _, ext = os.path.splitext(filepath) - if ext.lower() not in TEXT_EXTENSIONS: - return None - - with open(filepath, "r", encoding="utf-8", errors="ignore") as f_: - content = f_.read() - if not content.strip(): - return None - - # Limit file length - if MAX_TEXT_LENGTH is not None: - content = content[:MAX_TEXT_LENGTH] - - # Split content into lines - lines = [line_ for line_ in content.splitlines() if line_.strip()] - if not lines: - return None - - # Embed all lines - embeddings = embed_model.encode(lines, convert_to_tensor=True, device=DEVICE) - - # Predict per line - probs = [] - for emb in embeddings: - with torch.no_grad(): - output = model(emb.unsqueeze(0)) - probs.append(torch.sigmoid(output).item()) - - max_prob = max(probs) - if max_prob < SENSITIVE_THRESHOLD: - return None - - # Get top 5 lines contributing most - top_lines = [lines[i] for i, p in sorted(enumerate(probs), key=lambda x: x[1], reverse=True)[:5]] - - # Backup file - rel_path = os.path.relpath(filepath, ROOT_DIR) - backup_path = os.path.join(SAVE_DIR, rel_path) - os.makedirs(os.path.dirname(backup_path), exist_ok=True) - shutil.copy2(filepath, backup_path) - - return { - "file": filepath, - "probability": max_prob, - "copied_to": backup_path, - "reason": top_lines - } - - except Exception as e: - log.error(f"Could not process {filepath}: {e}") - return None - - -# ================== DIRECTORY SCAN ================== -def scan_directory(root): - sensitive_files = [] - with ThreadPoolExecutor(max_workers=NUM_WORKERS) as executor: - futures = [] - for dirpath, _, filenames in os.walk(root): - for file in filenames: - futures.append(executor.submit(process_file, os.path.join(dirpath, file))) - - for future in as_completed(futures): - result = future.result() - if result: - sensitive_files.append(result) - - return sensitive_files - - -# ================== MAIN ================== -def main(): - log.info(f"Scanning directory: {ROOT_DIR} - This will take some time...") - sensitive = scan_directory(ROOT_DIR) - - # Save JSON report - with open(REPORT_JSON, "w", encoding="utf-8") as f: - json.dump(sensitive, f, indent=2, ensure_ascii=False) - - # Save CSV report - with open(REPORT_CSV, "w", newline="", encoding="utf-8") as f: - writer = csv.DictWriter(f, fieldnames=["file", "probability", "copied_to", "reason"]) - writer.writeheader() - for entry in sensitive: - # Join top lines as single string for CSV - entry_csv = entry.copy() - entry_csv["reason"] = " | ".join(entry["reason"]) - writer.writerow(entry_csv) - - print() - log.debug("Sensitive files detected and backed up:") - for entry in sensitive: - log.debug(f" - {entry['file']} (prob={entry['probability']:.4f})") - for line in entry["reason"]: - log.debug(f" -> {line}") - - print() - log.info("Backup completed.\n") - log.debug(f"Files copied into: {SAVE_DIR}") - log.debug(f"JSON report saved as: {REPORT_JSON}") - log.debug(f"CSV report saved as: {REPORT_CSV}") - - -if __name__ == "__main__": - log.info(f"Starting VulnScan with {NUM_WORKERS} thread workers and {len(SENSITIVE_PATHS)} paths...") - for path in SENSITIVE_PATHS: - expanded_path = os.path.expandvars(path) - if os.path.exists(expanded_path): - ROOT_DIR = expanded_path - main() - else: - log.warning(f"Path does not exist and will be skipped: {expanded_path}") diff --git a/CODE/vulnscan/Model_SenseMacro.4n1.pth b/CODE/vulnscan/Model_SenseMacro.4n1.pth deleted file mode 100644 index 4f8182cc..00000000 Binary files a/CODE/vulnscan/Model_SenseMacro.4n1.pth and /dev/null differ diff --git a/CODE/wifi_stealer.py b/CODE/wifi_stealer.py deleted file mode 100644 index efb95567..00000000 --- a/CODE/wifi_stealer.py +++ /dev/null @@ -1,109 +0,0 @@ -from __future__ import annotations - -from logicytics import log, execute - - -def get_password(ssid: str) -> str | None: - """ - Retrieves the password for a specified Wi-Fi network. - - Args: - ssid (str): The name (SSID) of the Wi-Fi network to retrieve the password for. - - Returns: - str or None: The Wi-Fi network password if found, otherwise None. - - Raises: - Exception: If an error occurs during command execution or password retrieval. - - Notes: - - Uses the Windows `netsh` command to extract network profile details - - Searches command output for "Key Content" to find the password - - Logs any errors encountered during the process - """ - try: - command_output = execute.command( - f'netsh wlan show profile name="{ssid}" key=clear' - ) - if command_output: - key_content = command_output.splitlines() - for line in key_content: - if "Key Content" in line: - return line.split(":")[1].strip() - return None - except Exception as err: - log.error(err) - return None - - -def parse_wifi_names(command_output: str) -> list: - """ - Parses the output of the command to extract Wi-Fi profile names. - - Args: - command_output (str): The output of the command "netsh wlan show profile" containing Wi-Fi profile information. - - Returns: - list: A list of extracted Wi-Fi profile names, stripped of whitespace. - """ - wifi_names = [] - - for line in command_output.split("\n"): - if "All User Profile" in line: - start_index = line.find("All User Profile") + len("All User Profile") - wifi_name = line[start_index:].strip() - wifi_names.append(wifi_name) - - return wifi_names - - -def get_wifi_names() -> list: - """ - Retrieves the names of all Wi-Fi profiles on the system. - - Executes the "netsh wlan show profile" command to list available Wi-Fi network profiles. - Parses the command output to extract individual profile names. - - Returns: - list: A list of Wi-Fi network profile names discovered on the system. - - Raises: - Exception: If an error occurs during the retrieval of Wi-Fi names. - - Example: - wifi_profiles = get_wifi_names() # Returns ['HomeNetwork', 'CoffeeShop', ...] - """ - try: - log.info("Retrieving Wi-Fi names...") - wifi_names = parse_wifi_names(execute.command("netsh wlan show profile")) - log.info(f"Retrieved {len(wifi_names)} Wi-Fi names.") - return wifi_names - except Exception as err: - log.error(err) - return [] - - -@log.function -def get_wifi_passwords(): - """ - Retrieves the passwords for all Wi-Fi profiles on the system. - - This function retrieves the names of all Wi-Fi profiles on the system using the get_wifi_names() function. - It then iterates over each Wi-Fi profile name and retrieves the password associated with the profile using the get_password() function. - The Wi-Fi profile names and passwords are stored in a dictionary where the key is the Wi-Fi profile name and the value is the password. - """ - with open("WiFi.txt", "w") as file: - for name in get_wifi_names(): - try: - log.info(f"Retrieving password for {name.removeprefix(': ')}") - file.write( - f"Name: {name.removeprefix(': ')}, Password: {get_password(name.removeprefix(': '))}\n" - ) - except UnicodeDecodeError as e: - log.error(e) - except Exception as e: - log.error(e) - - -if __name__ == "__main__": - get_wifi_passwords() diff --git a/CODE/window_feature_miner.ps1 b/CODE/window_feature_miner.ps1 deleted file mode 100644 index 83832b77..00000000 --- a/CODE/window_feature_miner.ps1 +++ /dev/null @@ -1,3 +0,0 @@ -# List all optional features and save them to Features.txt -Write-Output "INFO: Starting Feature mining, saving to Features.txt" -Get-WindowsOptionalFeature -Online | Format-Table -Property FeatureName, State > Features.txt diff --git a/CODE/wmic.py b/CODE/wmic.py deleted file mode 100644 index 8978bf41..00000000 --- a/CODE/wmic.py +++ /dev/null @@ -1,40 +0,0 @@ -from logicytics import log, execute - - -@log.function -def wmic(): - """ - Retrieves system information using WMIC commands. - - This function runs a series of WMIC commands to gather information about the system's BIOS, - operating system, computer system, and disk drives. - The output of each command is written to a file named "wmic_output.txt". - - Parameters: - None - - Returns: - None - """ - data = execute.command("wmic BIOS get Manufacturer,Name,Version /format:htable") - with open("WMIC.html", "w") as file: - file.write(data) - wmic_commands = [ - "wmic os get Caption,CSDVersion,ServicePackMajorVersion", - "wmic computersystem get Model,Manufacturer,NumberOfProcessors", - "wmic BIOS get Manufacturer,Name,Version", - "wmic diskdrive get model,size", - ] - with open("wmic_output.txt", "w") as file: - for index, command in enumerate(wmic_commands): - log.info(f"Executing Command Number {index + 1}: {command}") - output = execute.command(command) - file.write("-" * 190) - file.write(f"Command {index + 1}: {command}\n") - file.write(output) - - file.write("-" * 190) - - -if __name__ == "__main__": - wmic() diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f9be04e7..5e05277e 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,125 +1,173 @@ # Contributing to Logicytics -Looking to contribute something to Logicytics? **Here's how you can help.** - -Please take a moment to review this document to make the contribution -process easy and effective for everyone involved. - -Following these guidelines helps to communicate that you respect the time of -the developers managing and developing this open source project. In return, -they should reciprocate that respect in addressing your issue or assessing -patches and features. - -## Using the issue tracker - -The [issue tracker](https://github.com/DefinetlyNotAI/Logicytics/issues) is -the preferred channel for bug reports and features requests -and submitting pull requests, but please respect the following -restrictions: - -- Please **Do not** derail or troll issues. Keep the discussion on topic and - respect the opinions of others. - -- Please **Do not** post comments consisting solely of "+1" or "👍 ". - Use [GitHub's "reactions" feature](https://blog.github.com/2016-03-10-add-reactions-to-pull-requests-issues-and-comments) - instead. We reserve the right to delete comments which violate this rule. - -## Issues assignment - -I will be looking at the open issues, analyse them, and provide guidance on how to proceed. - -Issues can be assigned to anyone other than me and contributors are welcome -to participate in the discussion and provide their input on how to best solve the issue, -and even submit a PR if they want to. - -Please wait that the issue is ready to be worked on before submitting a PR. -We don't want to waste your time. - -Please keep in mind that I am a human and have limited resources and am not always able to respond immediately. -I will try to provide feedback as soon as possible, but please be patient. - -If you don't get a response immediately, -it doesn't mean that we are ignoring you or that we don't care about your issue or PR. -We will get back to you as soon as we can. - -If you decide to pull a PR or fork the project, keep in mind that you should only add/edit the scripts you need to, -leave core files alone. - -## Guidelines for Modifications 📃 - -When making modifications to the Logicytics project, -please adhere to the following guidelines on the Wiki page. - -## Issues and labels 🛠️ - -Our bug tracker utilizes several labels to help organize and identify issues. - -For a complete look at our labels, see the [project labels page](https://github.com/DefinetlyNotAI/Logicytics/labels). - -## Bug reports 🐛 - -A bug is a _demonstrable problem_ that is caused by the code in the repository. -Good bug reports are extremely helpful! - -Guidelines for bug reports: - -1. **Use the GitHub issue search** — check if the issue has already been - reported. - -2. **Check if the issue has been fixed** — try to reproduce it using the - latest `main` (or `version` branch if the issue is about a version) in the repository. - -A good bug report shouldn't leave others needing to chase you up for more -information. Please try to be as detailed as possible in your report. What is -your environment? What steps will reproduce the issue? What browser(s) and OS -experience the problem? Do other browsers show the bug differently? What -would you expect to be the outcome? All these details will help people to fix -any potential bugs. - -## Feature requests 🚀 - -Feature requests are welcome. But take a moment to find out whether your idea -fits with the scope and aims of the project. It's up to _you_ to make a strong -case to convince the project's developers of the merits of this feature. Please -provide as much detail and context as possible. - -## Pull requests 📝 - -Good pull requests—patches, improvements, new features—are a fantastic -help. They should remain focused in scope and avoid containing unrelated -commits. - -**Please ask first** before embarking on any **significant** pull request (e.g. -implementing features, refactoring code, porting to a different language), -otherwise you risk spending a lot of time working on something that the -project's developers might not want to merge into the project. For trivial -things, or things that don't require a lot of your time, you can go ahead and -make a PR. - -Please adhere to the coding guidelines used throughout the -project (indentation, accurate comments, etc.) and any other requirements -(such as test coverage). - -View the Wiki for more information on how to write pull requests. - -**IMPORTANT**: By submitting a patch, you agree to allow the project owners to -license your work under the terms of the [License](https://github.com/DefinetlyNotAI/Logicytics/blob/main/LICENSE) - -## License 📝 - -By contributing your code, you agree to license your contribution under -the [MIT License](https://github.com/DefinetlyNotAI/Logicytics/blob/main/LICENSE). -By contributing to the documentation, you agree to license your contribution under -the [Creative Commons Attribution 3.0 Unported License](https://creativecommons.org/licenses/by/3.0/). - -You also agree to the [Developer Certificate of Origin](DCO.md). - -## Communication 🗣️ - -- **Issues**: Use GitHub issues for bug reports and feature requests. Keep the discussion focused and relevant. -- **Pull Requests**: Use pull requests to propose changes. Be prepared to discuss your changes and address any feedback. - -If you have any questions or need further clarification, please feel free to [contact](mailto:Nirt_12023@outlook.com) -me. - -Thank you for your contributions! +Logicytics v4 is a Windows-focused evidence collector with strict authorization, +isolation, output, and compatibility contracts. Keep changes focused and preserve +those contracts. Use the [issue tracker](https://github.com/DefinetlyNotAI/Logicytics/issues) +for reproducible bugs and scoped feature proposals. + +## Development setup + +Requirements: + +- Windows for live collector and platform integration checks. +- Python 3.11 or later. The v4 engine has no third-party runtime dependency. +- Git for developer integrity and explicit update actions. + +From a fresh checkout: + +```powershell +python -m logicytics preflight +python -m unittest discover -v +python -m compileall -q logicytics core tests +``` + +`preflight` must report no invalid core collector. Some live Windows features such +as WMIC, BitLocker, or Sysinternals are optional; absence must produce an explicit +skip or availability result rather than breaking unrelated collection. + +## Architecture boundaries + +The canonical pipeline is: + +`request -> validated plan -> isolated collectors -> registered artifacts -> manifest -> package` + +- `logicytics/cli/` parses and renders. It does not collect evidence. +- `logicytics/module/planner.py` resolves one deterministic plan from immutable + `RunRequest` policy and validated metadata. +- `logicytics/module/runtime.py` owns per-run and per-worker lifecycle, cancellation, + retries, timeouts, failure aggregation, and post-run actions. +- `logicytics/module/platform_adapters.py` owns host command, process, registry, + filesystem, network, privilege, and Win32 access. Collectors must use these + injectable seams instead of importing host APIs directly. +- `logicytics/module/artifacts.py` is the only publication path from collector + workspaces into the run artifact catalog. +- `logicytics/module/packaging.py` consumes the finalized catalog; it never scans source + directories for arbitrary files. +- Maintenance, debug, update, developer, and usage behavior stays separate from + normal collection. + +Importing `logicytics` must not start collection, load the supervisor, or create +files. Do not add global mutable run state, implicit current-directory behavior, +collector-to-collector calls, or a second execution/output path. + +## Core collector changes + +Each `core//.py` module owns exactly one public +`Collector` and one primary job. Split a feature into a new ID when +it needs a different capability, privilege level, network reach, sensitive-data +category, timeout/cost policy, or output contract. + +A collector must: + +- inherit `CoreCollector` and provide fully typed, documented `metadata`, + `validate`, `prepare`, `collect`, `finalize`, and `cleanup` behavior; +- use an ID, class name, file name, and `Specialty` that agree; +- declare platform, capabilities, privilege, network access, sensitivity, + profiles, dependencies, scheduling policy, timeouts, retries, memory/output + limits, artifact count, and every MIME type it can publish; +- check `context.is_cancelled` before work and during every long loop, query, + capture, copy, or traversal; +- write only beneath `context.workspace`, report structured progress, register + every result through `context.artifacts`, and return a typed result; +- turn expected absence or permission denial into an actionable `skipped` or + `partial` result without hiding an actual failure; +- avoid `print`, `exit`, nested worker pools, mutable globals, repository writes, + package-wide cleanup, secrets in logs, and unbounded reads or subprocess output. + +Run the collector directly and through the orchestrator. Add mocked Windows +responses, cancellation coverage, and output-contract evidence. If structured +bytes change intentionally, update the relevant golden file in `tests/golden/` +and explain why. + +## Plugins + +Plugins implement the typed `PluginCollector` contract and remain opt-in. They +may not bypass preflight, planning, capability approval, isolated workspaces, +artifact registration, or package filtering. + +Read [PLUGIN_AUTHORING.md](docs/PLUGIN_AUTHORING.md) before changing discovery or +extension behavior. Read [MIGRATION.md](docs/MIGRATION.md) before changing a legacy +flag, schema migration, or historical `CODE` evidence import. Compatibility code must remain +a bounded translation into the canonical v4 model. + +## Configuration changes + +Configuration changes must preserve the strict schema in +[CONFIGURATION.md](docs/CONFIGURATION.md). Update the field reference and its +parser-backed tests in the same commit as any setting, default, bound, migration +alias, or source-precedence change. + +Profiles, modes, include/exclude selections, plugin enablement, capability +approval, authorization acknowledgement, scheduling overrides, reruns, package +policy, and post-run power actions are invocation-only `RunRequest` behavior. +They do not belong in persistent configuration. + +## Evidence, security, and compatibility + +- Update [OUTPUTS.md](docs/OUTPUTS.md) when a filename, MIME type, package path, + evidence kind, or retention rule changes. +- Keep source, executables, models, configuration secrets, caches, and library + internals out of evidence packages. +- Preserve reproducible package hashes and manifest schema validation. +- Add explicit capabilities for sensitive or elevated access. Never weaken + authorization to make a test pass. +- Follow [SECURITY.md](SECURITY.md) for vulnerability reports. Do not put secrets + or real private evidence in issues, fixtures, logs, or commits. + +## Testing expectations + +Use the narrowest relevant tests while iterating, then run the complete gates +before opening a pull request: + +```powershell +python -m unittest discover -v +python -m compileall -q logicytics core tests +python -m logicytics preflight +git diff --check +``` + +On Windows, also run: + +```powershell +python -m unittest tests.test_windows_integration -v +``` + +The suite includes static preflight, lifecycle/cancellation checks, mocked +collector responses, golden output bytes, flow/mode matrices, package/hash +reproduction, documentation contracts, and bounded live Windows probes. A build +or compile check alone is not sufficient evidence. + +## Documentation changes + +Keep user and developer documentation synchronized with behavior: + +- [README.md](README.md): installation, quick start, CLI, permissions, and + troubleshooting. +- [CONFIGURATION.md](docs/CONFIGURATION.md): every persistent setting and migration. +- [OUTPUTS.md](docs/OUTPUTS.md): artifact and retention contracts. +- [MIGRATION.md](docs/MIGRATION.md): supported compatibility boundary. +- [FLOW_MATRIX.md](docs/FLOW_MATRIX.md): executable flow evidence. + +The repository wiki is complementary documentation, not a substitute for the +versioned contract files required to review a change. + +## Issues and pull requests + +A useful bug report includes the Logicytics version, Windows edition/build, +Python version, command and sanitized configuration, expected and actual result, +exit code, relevant redacted logs, and exact reproduction steps. State whether +the process was elevated and whether an optional Windows feature was installed. + +Pull requests should: + +- solve one coherent problem and avoid unrelated edits; +- use conventional commit subjects such as `feat:`, `fix:`, `refactor:`, + `test:`, or `docs:` with a detailed body; +- include tests and documentation proportional to the changed contract; +- preserve unrelated work and never include generated evidence or secrets; +- pass the complete verification gates above; and +- comply with the [Developer Certificate of Origin](.github/DCO.md), + [Code of Conduct](CODE_OF_CONDUCT.md), and repository license. + +By contributing code, you agree to license it under the [MIT License](LICENSE). +Documentation contributions use the repository's stated documentation license. diff --git a/MODS/_MOD_SKELETON.py b/MODS/_MOD_SKELETON.py deleted file mode 100644 index c77a6ecd..00000000 --- a/MODS/_MOD_SKELETON.py +++ /dev/null @@ -1,54 +0,0 @@ -# If using the future annotations, it should be ontop of the file -# from __future__ import annotations - -# Other Imports if needed or necessary go here - -# To know more check the WiKi -from logicytics import log # And more if needed - - -# Your actual code, must be able to run without any interference by outside actions -# USE log.debug, log.info, log.error, log.warning and log.critical and log.string as well -# You can choose to use any other of the code without issues -# Example of said code:- - - -# This log decorator logs the function name and the time it took to run, -# It is recommended to use this, -# as it only logs the function and the time it took to run -# in debug mode thus helping when people enable debug mode -# Do note however, if you are using multiple decorators, this should be the last one -# check the WiKi for more information -# Do not use this decorator if you are running a function that is part of another function -@log.function -def MOD_EXAMPLE() -> None: - """ - This function MOD is used to log different types of messages. - - It logs an error message, a warning message, an info message, and a debug message. - - Parameters: - None - - Returns: - None - """ - log.error("This is an error") - log.warning("This is a warning") - log.info("This is a info message") - log.debug("This is a debug message") - log.critical("This is a critical message") - # This is special, allows you to use strings to specify the log level, it is not recommended to use these - # Options are error, warning, info, debug, critical - It is case-insensitive and can be used with any of the log levels - # Defaults with the log level of debug - log.string("This is a random message", "ERROR") - pass # Your code here with proper logging like the above log options - - -# It is recommended to call your function at the end of the file using the following code -# This is to ensure that the function is called only when directly executed and not when imported -if __name__ == "__main__": - MOD_EXAMPLE() - -# Always remember to call your function at the end of the file and then leave a new line -# This is to ensure that the function is called and the file is not empty diff --git a/PLANS.md b/PLANS.md deleted file mode 100644 index 221aa875..00000000 --- a/PLANS.md +++ /dev/null @@ -1,15 +0,0 @@ -# To-Do List - -> [!TIP] -> Here is a key for the table above: -> -> - ❌ ➡️ Might be done, Not sure yet -> - ✅ ➡️ Will be done, 100% sure - -| Task | Version | Might or Will be done? | -|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------|---------|------------------------| -| Update to model 4n2 of vulnscan | v3.6.1 | ✅ | -| Merge `sensitive data miner` with `vulnscan` to be 1 tool | v4.0.0 | ❌ | -| Remake Logicytics End-Execution cycle, where files created must go in `temp/` directory, and zipper takes it from there only, simplifying any code logic with this as well | v4.0.0 | ✅ | -| Replace Logger.py with Util that contains (tprint), also implement the ExceptionHandler and UpdateManager from Util | v4.0.0 | ✅ | -| Make WIKI in the git repo, with a yaml file that updates it to the default github wiki | v4.0.0 | ✅ | diff --git a/README.md b/README.md index 19b25e9f..c2631523 100644 --- a/README.md +++ b/README.md @@ -1,252 +1,146 @@ -# Logicytics: System Data Harvester - -Logicytics is a cutting-edge tool designed to -meticulously harvest and collect a vast array of Windows system data for forensic analysis. -Crafted with Python, it's an actively developed project that is dedicated -to gathering as much sensitive data as possible and packaging it neatly into a ZIP file. -This comprehensive guide is here to equip you with everything you need to use Logicytics effectively. - -
- GitHub Issues - GitHub Tag - GitHub Commit Activity - GitHub Language Count - GitHub Branch Check Runs - GitHub Repo Size -
-
- GitHub Repo CodeFactor Rating - Maintainability - OpenSSF Best Practices Score - OpenSSF Best Practices Badge -
- -> [!CAUTION] -> By using this software, you agree to the license, and agree that you hold responsibility of how you use and modify the -> code. - -## Installation and Setup - -To install and setup Logicytics, follow these steps: - -1. **Install Python**: If you don't have Python installed, you can download it from - the [official website](https://www.python.org/downloads/). - -2. **Install Dependencies**: Logicytics requires Python modules. You can install all the required modules by running the - following command in your terminal: `pip install -r requirements.txt` - -3. **Run Logicytics**: To run Logicytics, simply run the following command in your terminal: `python Logicytics.py -h` - - This opens a help menu. - -> [!IMPORTANT] -> We recommend Python Version `3.11` or higher, as the project is developed and tested on this version. -> -> To use vulnscan, you will need `torch` - Installation instructions can be -> found [here](https://pytorch.org/#fws_68845ae25b0fb). -> If you have a supported GPU, it is recommended to install the Nvidea GPU version of PyTorch for better performance. -> -> Settings should be: `Stable -> Windows -> Pip -> Python` and if you have a supported CUDA version, select that too -> else CPU. - -### Prerequisites - -- **Python**: The project requires Python 3.8 or higher. You can download Python from - the [official website](https://www.python.org/downloads/). - -- **Dependencies**: The project requires certain Python modules to be installed. You can install all the required - modules by running the following command in your terminal: `pip install -r requirements.txt`. - -- **Administrative Privileges**: To be able to run the program using certain features of the project, like registry - modification, you must run the program with administrative privileges. - -- **System Requirements**: The project has been tested on Windows 10 and 11. It will not work on other operating - systems. - -- **Knowledge of Command Line**: The project uses command line options for the user to interact with the program. It is - recommended to have a basic understanding of command line options. - -> [!IMPORTANT] -> You may create a `.sys.ignore` file in the `CODE/SysInternal_Suite` directory to not extract the exe binaries from the -> ZIP file (This is done for the OpenSSF score and to discourage binaries being used without source code), if the -`.sys.ignore` file is not found, it will auto extract the binaries and run them using `Logicytics`. -> -> For more details on these binaries, -> go [here](https://learn.microsoft.com/en-us/sysinternals/downloads/sysinternals-suite) - For you weary cautious -> internet -> crusaders, you can view the [source code here](https://github.com/MicrosoftDocs/sysinternals) and compare hashes and -> perform your audits. - -## Step-by-Step Installation and Usage - -1) Install Python - If you don't have Python installed, you can download it from the official - website. - Make sure to select the option to "Add Python to PATH" during installation. - -2) Install Dependencies - Logicytics requires Python modules. You can install all the required modules by running the following command in your - terminal: - `pip install -r requirements.txt` - -3) Run Logicytics - To run Logicytics, simply run the following command in your terminal: - python Logicytics.py -h - This opens a help menu. - -4) Run the Program - Once you have run the program, you can run the program with the following command: - `python Logicytics.py -h` - Replace the flags with the ones you want to use. - you must have admin privileges while running! - -> [!TIP] -> Although it's really recommended to use admin, by setting debug in the config.json to true, you can bypass this -> requirement - -5) Wait for magic to happen - Logicytics will now run and gather data according to the flags you used. - -6) Enjoy the gathered data - Once the program has finished running, you can find the gathered data in the "ACCESS/DATA" folder. Both Zip and Hash - will be found there. - -> [!NOTE] -> All Zips and Hashes follow a conventional naming mechanism that goes as follows -> `Logicytics_{CODE-or-MODS}_{Flag-Used}_{Date-And-Time}.zip` - -7) Share the love - If you like Logicytics, please consider sharing it with others or spreading the word about it. - -8) Contribute to the project - If you have an idea or want to contribute to the project, you can submit an issue or PR on - the GitHub repository. - -### Basic Usage - -After running and successfully collecting data, you may traverse the ACCESS directory as much as you like, -Remove add and delete files, it's the safe directory where your backups, hashes, data zips and logs are found. - -> [!TIP] -> Watch this [video](https://www.youtube.com/watch?v=XVTBmdTQqOs) to see a real life demo of Logicytics (Although the -> tools and interface may be changed as it's an older version `2.1.1` - `2.3.3`) - -## Configuration - -Logicytics uses a config.ini file to store configurations. The config.ini is located in the CODE directory. - -The config.ini file is a INI file that contains important information, you can find it [here](CODE/config.ini) - -The config.ini file is used to store the DEBUG flag bool, the VERSION, and the CURRENT_FILES. -It is also used to store and save settings for other programs. - -> [!TIP] -> CURRENT_FILES is an array of strings that contains the names of the files you have, -> this is used to later check for corruption or bugs. -> VERSION is the version of the project, used to check and pull for updates. - -## Mods - -Mods are special files that are run with the `--modded` flag. -These files are essentially scripts that are run after the main Logicytics.py script is run -and the verified scripts are run. - -They are used to add extra functionality to the script. -They are located in the `MODS` directory. In order to make a mod, -you need to create a python file with the `.py` extension or any of the supported extensions `.exe .ps1 .bat` -in the `MODS` directory. - -These file will be run after the main script is run. -When making a mod, you should avoid acting based on other files directly, -as this can cause conflicts with the data harvesting. -Instead, you should use the `Logicytics.py` file and other scripts as a reference -for how to add features to the script. - -The `--modded` flag is used to run all files in the `MODS` directory. -This flag is not needed for other files in the `CODE` directory to run, -but it is needed for mods to run. - -The `--modded` flag can also be used to run custom scripts. -If you want to run a custom script with the `--modded` flag, -you can add the script to the `MODS` directory, and it will be run with the `--modded` flag. - -To check all the mods and how to make your own, you can check the `Logicytics.py` file and the Wiki. -Also refer to the contributing.md for more info - -## Troubleshooting - -If you are having issues, here are some troubleshooting tips: - -Some errors may not necessarily mean the script is at fault, -but other OS related faults like files not existing, -or files not being modified, or files not being created. - -Some tips are: - -- Check if the script is running as admin and not in a VM -- Check if the script has the correct permissions and correct dependencies to run -- Check if the script is not being blocked by a firewall or antivirus or by a VPN or proxy -- Check if the script is not being blocked by any other software or service - -If those don't work attempt: - -- Try running the script with powershell instead of cmd, or vice versa -- Try running the script in a different directory, computer or python version above 3.8 - - Note: The version used to develop, test and run the script is 3.11 -- Try running the `--debug` flag and check the logs - -### Support Resources - -Check out the [wiki](https://github.com/DefinetlyNotAI/Logicytics/wiki) for help. - -> [!TIP] -> You can check out future plans [here](PLANS.md), -> you can contribute these plans if you have no idea's on what to contribute! - -### Want to create your own mod? - -Check out the [contributing guidlines](CONTRIBUTING.md) file for more info - -### Want More? - -If there is a specific piece of data that you would like to see extracted by Logicytics, -please let us know. We are constantly working to improve the project and adding new features. - -### Want to create your own mod? - -Check out the [contributing guidlines](CONTRIBUTING.md) file for more info, -as well as the [wiki guidelines](https://github.com/DefinetlyNotAI/Logicytics/wiki/5-How-to-Contribute) for more info -Tips and tricks of the given modules/APIs can be -found [here](https://github.com/DefinetlyNotAI/Logicytics/wiki/6-Code-tips-and-tricks) too! - -> [!IMPORTANT] -> Always adhere to the [coding standards](https://github.com/DefinetlyNotAI/Logicytics/wiki/7-Advanced-Coding-Standards) -> of Logicytics! - -## Conclusion - -Logicytics is a powerful tool that can extract a wide variety of data from a Windows system. -With its ability to extract data from various sources, Logicytics can be used for a variety of purposes, -from forensics to system information gathering. -Its ability to extract data from various sources makes it a valuable tool -for any Windows system administrator or forensic investigator. - -> [!CAUTION] -> Please remember that extracting data from a system without proper authorization is illegal and unethical. -> Always obtain proper authorization before extracting any data from a system. - -## Support Me - -Please consider buying me a coffee or sponsoring me in GitHub sponsor, -I am saving for my college funds, and I need your help! -Supporters will be placed in the Credits ❤️ - -### Links - -- [Project's Wiki](https://github.com/DefinetlyNotAI/Logicytics/wiki) -- [Project's Future](PLANS.md) -- [Project's License](LICENSE) - -### License - -- [Developer Certificate of Origin](DCO.md) -- [Our License](LICENSE) +# Logicytics + +

+ Reliable Windows evidence collection, organized around one verified run at a time. +

+ +
+ GitHub Issues + GitHub Tag + GitHub Commit Activity + GitHub Language Count + GitHub Branch Check Runs + GitHub Repo Size +
+
+ GitHub Repo CodeFactor Rating + Maintainability + OpenSSF Best Practices Score + OpenSSF Best Practices Badge +
+ +Logicytics is a Windows evidence collection framework. It validates every collector before use, runs each one in +isolation, and keeps the result in a manifest-backed run folder. A collector can succeed, skip, or fail without +obscuring the rest of the verified run. + +The complete user and developer manual is in [`docs/README.md`](docs/README.md) and is mirrored to +the [Logicytics Wiki](https://github.com/DefinetlyNotAI/Logicytics/wiki). + +> Use Logicytics only on systems and data you are authorized to inspect. + +## Start here + +The installer is the only command intended to run outside the managed virtual environment. Run it once from the +repository root: + +```powershell +python -m logicytics.cli.installer +``` + +Then activate the environment and check the installation: + +```powershell +.\.venv\Scripts\Activate.ps1 +python -m logicytics preflight +``` + +When preflight reports no invalid collectors, make a plan and run it: + +```powershell +python -m logicytics plan --profile standard +python -m logicytics run --profile standard --acknowledge-authorization +``` + +If a normal command says the environment is missing, run the installer. If it says the environment is not active, run +`.\.venv\Scripts\Activate.ps1` first. + +## Choose a run + +Every run validates collectors, records a manifest, and packages the result unless `--no-package` is supplied. + +| Need | Command | +|-------------------------------------|--------------------------------------------------------------------------------------------| +| Fast local inventory | `python -m logicytics run --mode quick --acknowledge-authorization` | +| Everyday collection | `python -m logicytics run --mode balanced --acknowledge-authorization` | +| Deterministic sequential collection | `python -m logicytics run --mode standard --acknowledge-authorization` | +| Local-only collection | `python -m logicytics run --mode offline --acknowledge-authorization` | +| Extended collection | `python -m logicytics run --mode thorough --acknowledge-authorization` | +| Thorough duration report | `python -m logicytics run --mode thorough --acknowledge-authorization --performance-check` | + +`thorough` can include administrator-only collectors. Start an elevated shell when the plan reports that requirement. +See every available mode with `python -m logicytics --modes`. +Add `--performance-check` to any `run --mode ...` command to time that mode's selected collectors serially. + +For offline collection from removable storage, add `--usb` to `preflight`, +`plan`, `run`, or `collector`. It scans `A:` through `Z:` and uses the first +drive containing `Windows`; use `--usb=E` to select a specific Windows drive. +USB mode rejects output, cache, and temporary storage on that Windows disk. + +## Where results go + +Each run receives its own directory under `output/data/`: + +```text +output/data/run// + manifest.json # status, collector results, and artifact catalog + artifacts/ # collected evidence + logs/ # run and collector JSONL events + reports/ # generated summaries + +output/data/zip/.zip +output/data/hashes/.zip.sha256 +``` + +The console is intentionally brief. Use `manifest.json` to inspect a run, the package hash to verify a package, and +`output/logs/Logicytics.log` for the human-readable application log. Interaction history and its usage graph live in +`.cache/`, which is created automatically. Worker scratch files default to project-local `.temp/`; set +`runtime.temporary_directory: system` in `logicytics.yaml` to use `%TEMP%/logicytics/` instead. The fingerprint is a +SHA-256 identity derived from the immutable run ID; output uses its shortest unique prefix, starting at eight characters +and extending only on a collision. Set `logging.level: DEBUG` in `logicytics.yaml` when you need detailed worker +lifecycle information and file call sites. + +## Useful commands + +```powershell +# Revalidate every collector instead of reusing cached preflight probes +python -m logicytics preflight --invalidate-cache + +# Inspect a plan without collecting evidence +python -m logicytics plan --profile standard + +# Run one collector only +python -m logicytics collector core.system.system_info --acknowledge-authorization + +# Run the complete test suite +python -m logicytics.cli.tests + +# See diagnostics, configuration, and maintenance state +python -m logicytics debug +``` + +Use `python -m logicytics --help` or append `--help` to any command for its full flags. + +## Configuration and extensions + +`logicytics.yaml` is the single user configuration file. It controls output locations, worker limits, logging, optional +Sysinternals setup, and declared collector settings. Keep credentials and secrets out of it. + +Core collectors are shipped and validated as part of the application. Plugins are opt-in and must pass the same +validation boundary before they can run. + +- [Configuration reference](docs/CONFIGURATION.md) +- [Output contract](docs/OUTPUTS.md) +- [Migration guide](docs/MIGRATION.md) +- [Flow matrix](docs/FLOW_MATRIX.md) + +## Help and contributing + +The [Logicytics Wiki](https://github.com/DefinetlyNotAI/Logicytics/wiki) covers setup, troubleshooting, collector +development, architecture, and security in more depth. + +For changes to Logicytics, read [CONTRIBUTING.md](CONTRIBUTING.md). Please also review [SECURITY.md](SECURITY.md) +and [CODE_OF_CONDUCT.md](CODE_OF_CONDUCT.md). + +## License + +Logicytics is released under the [project license](LICENSE). diff --git a/SECURITY.md b/SECURITY.md index 110e6ab6..39ab0d78 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,74 +1,62 @@ -# Security Policy - -## Supported Versions - -This section outlines the versions of our project that are currently supported with security updates. - -| Version | Supported | Major Release Date | -|---------|-----------|--------------------| -| 3.6.x | ✅ | July 26, 2025 | -| 3.5.x | ✅ | July 26, 2025 | -| 3.4.x | ✖️ | January 3, 2025 | -| 3.3.x | ✖️ | January 3, 2025 | -| 3.2.x | ✖️ | Dec 19, 2024 | -| 3.1.x | ✖️ | Dec 11, 2024 | -| 3.0.x | ❌ | Dec 6, 2024 | -| 2.5.x | ❌ | Nov 25, 2024 | -| 2.4.x | ❌ | Nov 12, 2024 | -| 2.3.x | ❌ | Sep 21, 2024 | -| 2.2.x | ❌ | Sep 9, 2024 | -| 2.1.x | ❌ | Aug 29, 2024 | -| 2.0.x | ❌ | Aug 25, 2024 | -| 1.6.x | ❌ | Jun 18, 2024 | -| 1.5.x | ❌ | Jun 10, 2024 | -| 1.4.x | ❌ | May 30, 2024 | -| 1.3.x | ❌ | May 21, 2024 | -| 1.2.x | ❌ | May 16, 2024 | -| 1.1.x | ❌ | May 10, 2024 | -| 1.0.x | ❌ | May 4, 2024 | - -### Key: - -| Key | Desc | -|-----|-----------------------------------------------------| -| ✅ | Supported for all security updates | -| ⚠️ | Supported, but will leave support next major update | -| ✖️ | Only for major security issues (CVSS 8.0+) | -| ❌ | No longer supported for any security updates | - -## Reporting a Vulnerability - -If you believe you have found a security vulnerability in our project, we encourage you to report it to us. Your report -will help us improve the security of our project and ensure the trust of our users. - -### How to Report a Vulnerability - -1. **Identify the Vulnerability**: Clearly describe the vulnerability, including how it can be exploited and any - potential impact. -2. **Provide Detailed Information**: Include as much detail as possible, such as the version of the project affected, - steps to reproduce the vulnerability, and any relevant code snippets or screenshots. -3. **Contact Us**: Send your report to my [email](mailto:Nirt_12023@outlook.com). Please include "Security Vulnerability - Report" in the subject line. - -### What to Expect - -- **Acknowledgment**: Upon receiving your report, we will acknowledge receipt within 2�5 business days. -- **Investigation**: Our security team will investigate the vulnerability and determine its validity. -- **Update**: If the vulnerability is accepted, we will work on a fix and provide an update on the timeline for a - security update. -- **Communication**: We will communicate with you regarding the status of the vulnerability and any necessary actions. - -### Vulnerability Acceptance Criteria - -- The vulnerability must be reproducible. -- The vulnerability must be exploitable. -- The vulnerability must not be a false positive. - -### Vulnerability Decline Criteria - -- The vulnerability is outside the scope of our project. - -Thank you for helping us maintain the security of our project. Your contribution is invaluable in keeping our users -safe. - ---- +# Security policy + +## Supported versions + +| Version | Security support | Release date | +| ------- | ---------------- | ------------------ | +| 4.0.x | Supported | September 25, 2026 | +| 3.6.x | Partial Support | July 26, 2025 | +| 3.5.x | Partial Support | July 26, 2025 | +| 3.4.x | Partial Support | January 3, 2025 | +| 3.3.x | Unsupported | January 3, 2025 | +| 3.2.x | Unsupported | December 19, 2024 | +| 3.1.x | Unsupported | December 11, 2024 | +| 3.0.x | Unsupported | December 6, 2024 | +| 2.5.x | Unsupported | November 25, 2024 | +| 2.4.x | Unsupported | November 12, 2024 | +| 2.3.x | Unsupported | September 21, 2024 | +| 2.2.x | Unsupported | September 9, 2024 | +| 2.1.x | Unsupported | August 29, 2024 | +| 2.0.x | Unsupported | August 25, 2024 | +| 1.6.x | Unsupported | June 18, 2024 | +| 1.5.x | Unsupported | June 10, 2024 | +| 1.4.x | Unsupported | May 30, 2024 | +| 1.3.x | Unsupported | May 21, 2024 | +| 1.2.x | Unsupported | May 16, 2024 | +| 1.1.x | Unsupported | May 10, 2024 | +| 1.0.x | Unsupported | May 4, 2024 | + +Only the current v4 release line receives security fixes. Upgrade through the +documented [migration boundary](docs/MIGRATION.md); do not run an unsupported checkout +against sensitive evidence. + +## Reporting a vulnerability + +Do not open a public issue containing an exploit, credential, private key, +collected artifact, host identifier, or unredacted log. Email +[Nirt_12023@outlook.com](mailto:Nirt_12023@outlook.com) with the subject +`Logicytics security vulnerability`. + +Include: + +- the affected Logicytics version and commit; +- Windows edition/build and Python version; +- the affected collector, command, capability, or package contract; +- reproducible steps and expected security boundary; +- impact and required authorization/elevation state; and +- a minimal redacted proof of concept. + +Reports should target behavior owned by this repository and be reproducible. +Maintainers will acknowledge receipt, investigate the report, coordinate a fix +and disclosure window when accepted, and credit the reporter if requested. + +## Security boundaries + +Logicytics requires authorization acknowledgement and explicit capability +approval. A vulnerability includes bypassing those checks, escaping a collector +workspace, publishing undeclared evidence, leaking secrets to logs/metadata, +executing an undeclared process or network action, corrupting package/hash +verification, or allowing one collector to affect unrelated runs. + +Do not weaken isolation, redaction, path validation, output limits, cancellation, +or package verification while developing a fix. Use synthetic fixtures only. diff --git a/core/bluetooth/bluetooth_addresses.py b/core/bluetooth/bluetooth_addresses.py new file mode 100644 index 00000000..9cf16dcb --- /dev/null +++ b/core/bluetooth/bluetooth_addresses.py @@ -0,0 +1,123 @@ +"""Export paired Bluetooth names and address-like PnP identifiers as bounded JSON.""" + +from __future__ import annotations + +import json +import re + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common access-denied wording from PowerShell PnP queries.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +def _format_address(instance_id: str | None) -> str | None: + """Extract one conventional colon-separated address from a PnP instance identifier.""" + if instance_id is None: + return None + + match = re.search( + r"(? CollectorMetadata: + """Declare the subprocess-gated Bluetooth-address JSON artifact contract.""" + return CollectorMetadata( + id="core.bluetooth.bluetooth_addresses", + name="Bluetooth addresses", + version="4.0.0", + specialty=Specialty.BLUETOOTH, + output_media_types=("application/json",), + description="Exports paired Bluetooth friendly names and address-like identifiers from PnP data.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("wireless_identifiers",), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query PnP data and register Bluetooth names and extracted addresses as JSON.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Bluetooth-address collection") + context.report_progress("bluetooth_addresses_started") + command = "Get-PnpDevice -Class Bluetooth | Select-Object FriendlyName, InstanceId, Present | ConvertTo-Json -Depth 2" + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "Bluetooth PnP access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "Bluetooth PnP query failed", errors=(detail,)) + try: + raw = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "Bluetooth PnP query returned invalid JSON", + errors=(str(error),), + ) + devices = raw if isinstance(raw, list) else [raw] + if not all(isinstance(device, dict) for device in devices): + return CollectorResult(CollectorStatus.FAILED, "Bluetooth PnP query returned an unexpected result") + report = [ + { + "friendly_name": device.get("FriendlyName"), + "instance_id": device.get("InstanceId"), + "address": _format_address(device.get("InstanceId")), + "present": device.get("Present"), + } + for device in devices + ] + output = context.workspace / "bluetooth_addresses.json" + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "bluetooth_addresses_finished", + device_count=len(report), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("Bluetooth names and address identifiers collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/bluetooth/bluetooth_history.py b/core/bluetooth/bluetooth_history.py new file mode 100644 index 00000000..a2599cda --- /dev/null +++ b/core/bluetooth/bluetooth_history.py @@ -0,0 +1,107 @@ +"""Write a timestamped Bluetooth-device snapshot for retained run-history evidence.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common Windows permission-denied wording.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class BluetoothHistoryCollector(CoreCollector): + """Export a timestamped Bluetooth snapshot without mutating shared history state.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the bounded Bluetooth-history JSON artifact contract.""" + return CollectorMetadata( + id="core.bluetooth.bluetooth_history", + name="Bluetooth history snapshot", + version="4.0.0", + specialty=Specialty.BLUETOOTH, + output_media_types=("application/json",), + description="Exports a timestamped Bluetooth PnP snapshot; retained run packages provide historical evidence.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("device_identifiers",), + default_profiles=("deep",), + timeout_seconds=45, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query Bluetooth PnP state and register a timestamped JSON snapshot.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Bluetooth-history collection") + context.report_progress("bluetooth_history_started") + command = ( + "$ErrorActionPreference = 'Stop'; " + "@(Get-PnpDevice -Class Bluetooth | " + "Select-Object FriendlyName, InstanceId, Status, Problem, Present) | " + "ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "Bluetooth-history access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "Bluetooth-history query failed", errors=(detail,)) + try: + devices = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "Bluetooth-history query returned invalid JSON", + errors=(str(error),), + ) + snapshot = { + "collected_at": datetime.now(UTC).isoformat(), + "devices": devices if isinstance(devices, list) else [devices], + } + output = context.workspace / "bluetooth_history.json" + output.write_text(json.dumps(snapshot, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "bluetooth_history_finished", + device_count=len(snapshot["devices"]), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("Bluetooth history snapshot collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Leave temporary snapshot removal to the isolated workspace lifecycle.""" diff --git a/core/bluetooth/paired_devices.py b/core/bluetooth/paired_devices.py new file mode 100644 index 00000000..eb8ce941 --- /dev/null +++ b/core/bluetooth/paired_devices.py @@ -0,0 +1,101 @@ +"""Collect Windows Plug and Play metadata for Bluetooth devices.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize Windows and PowerShell access-denied messages.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class PairedDevicesCollector(CoreCollector): + """Export Bluetooth PnP device metadata without changing Bluetooth state.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the read-only PnP query, subprocess permission, and bounded output.""" + return CollectorMetadata( + id="core.bluetooth.paired_devices", + name="Bluetooth devices", + version="4.0.0", + specialty=Specialty.BLUETOOTH, + output_media_types=("application/json",), + description="Captures available Bluetooth Plug and Play device metadata through PowerShell.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before issuing the PnP query.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run a read-only Bluetooth PnP query and register its normalized JSON output.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Bluetooth collection") + context.report_progress("bluetooth_devices_started") + command = ( + "Get-PnpDevice -Class Bluetooth | " + "Select-Object FriendlyName, InstanceId, Status, Class, Problem, Present | " + "ConvertTo-Json -Depth 2" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=20, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "Bluetooth PnP access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "Bluetooth PnP query failed", errors=(detail,)) + try: + payload = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "Bluetooth PnP query returned invalid JSON", + errors=(str(error),), + ) + devices = payload if isinstance(payload, list) else [payload] + output = context.workspace / "bluetooth_devices.json" + output.write_text(json.dumps(devices, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "bluetooth_devices_finished", + device_count=len(devices), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("Bluetooth device metadata collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the PowerShell process has already exited.""" diff --git a/core/browser/browser_data_backup.py b/core/browser/browser_data_backup.py new file mode 100644 index 00000000..b2b7ae9b --- /dev/null +++ b/core/browser/browser_data_backup.py @@ -0,0 +1,169 @@ +"""Copy bounded browser-profile evidence after explicit browser-data approval.""" + +from __future__ import annotations + +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + EvidenceKind, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +CHROMIUM_FILES = ( + "History", + "History-journal", + "Cookies", + "Cookies-journal", + "Login Data", + "Login Data-journal", + "Bookmarks", + "Preferences", +) +FIREFOX_FILES = ("places.sqlite", "cookies.sqlite", "logins.json", "key4.db", "prefs.js") +MAX_FILE_BYTES = 50 * 1024 * 1024 +MAX_TOTAL_BYTES = 256 * 1024 * 1024 + + +def _copy_file(source: Path, destination: Path, copied_bytes: int) -> tuple[Path | None, int]: + """Copy one bounded regular file, preserving metadata, or return no artifact.""" + try: + if not source.is_file() or source.is_symlink(): + return None, copied_bytes + size = source.stat().st_size + except OSError: + return None, copied_bytes + if size > MAX_FILE_BYTES or copied_bytes + size > MAX_TOTAL_BYTES: + return None, copied_bytes + destination.parent.mkdir(parents=True, exist_ok=True) + filesystem_adapter.copy_file(source, destination) + return destination, copied_bytes + size + + +class BrowserDataBackupCollector(CoreCollector): + """Copy bounded data from supported local browser profiles into a private workspace.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the explicit-consent browser-data artifact contract.""" + return CollectorMetadata( + id="core.browser.browser_data_backup", + name="Browser data backup", + version="4.0.0", + specialty=Specialty.BROWSER, + output_media_types=("application/octet-stream",), + description="Copies bounded local profile evidence from Edge, Chrome, Firefox, Opera, and Opera GX.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=( + Capability.FILESYSTEM_READ, + Capability.BROWSER_DATA, + Capability.SENSITIVE_FILES, + ), + sensitive_data_categories=("browser_history", "cookies", "credentials"), + default_profiles=("deep",), + timeout_seconds=300, + maximum_output_bytes=256 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state before browser-profile copying starts.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Copy supported profile files from configured local browser locations.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before browser data backup") + home = filesystem_adapter.home() + local = filesystem_adapter.environment_path("LOCALAPPDATA", home / "AppData" / "Local") + roaming = filesystem_adapter.environment_path("APPDATA", home / "AppData" / "Roaming") + chromium_roots = ( + ("chrome", local / "Google" / "Chrome" / "User Data"), + ("edge", local / "Microsoft" / "Edge" / "User Data"), + ("opera", roaming / "Opera Software" / "Opera Stable"), + ("opera_gx", roaming / "Opera Software" / "Opera GX Stable"), + ) + copied: list[Path] = [] + copied_bytes = 0 + context.report_progress("browser_data_backup_started") + for browser, root in chromium_roots: + try: + if browser.startswith("opera"): + profile_roots = [root] + else: + profile_roots = [ + path + for path in filesystem_adapter.children(root) + if path.is_dir() and (path.name == "Default" or path.name.startswith("Profile ")) + ] + except OSError: + continue + for profile in profile_roots: + for filename in CHROMIUM_FILES: + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during browser data backup") + destination, copied_bytes = _copy_file( + profile / filename, + context.workspace / "browser_data" / browser / profile.name / filename, + copied_bytes, + ) + if context.is_cancelled: + if destination is not None: + destination.unlink(missing_ok=True) + for copied_path in copied: + copied_path.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during browser data backup") + if destination is not None: + copied.append(destination) + firefox_profiles = roaming / "Mozilla" / "Firefox" / "Profiles" + try: + profiles = [path for path in filesystem_adapter.children(firefox_profiles) if path.is_dir()] + except OSError: + profiles = [] + for profile in profiles: + for filename in FIREFOX_FILES: + if context.is_cancelled: + for copied_path in copied: + copied_path.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during browser data backup") + destination, copied_bytes = _copy_file( + profile / filename, + context.workspace / "browser_data" / "firefox" / profile.name / filename, + copied_bytes, + ) + if context.is_cancelled: + if destination is not None: + destination.unlink(missing_ok=True) + for copied_path in copied: + copied_path.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during browser data backup") + if destination is not None: + copied.append(destination) + if not copied: + context.report_progress("browser_data_backup_finished", copied_files=0, bytes_written=0) + return CollectorResult.succeeded("no supported local browser profile data met the bounded backup policy") + artifacts = [] + for path in copied: + if context.is_cancelled: + for unpublished in copied[len(artifacts):]: + unpublished.unlink(missing_ok=True) + return CollectorResult.cancelled("cancelled during browser-data registration", tuple(artifacts)) + artifacts.append(context.artifacts.register_file(path, evidence_kind=EvidenceKind.RAW)) + artifact_tuple = tuple(artifacts) + context.report_progress( + "browser_data_backup_finished", + copied_files=len(artifact_tuple), + bytes_written=sum(item.size_bytes for item in artifact_tuple), + ) + return CollectorResult.succeeded("browser data backup collected", artifact_tuple) + + def cleanup(self, context: CollectorContext) -> None: + """Leave copied evidence removal to the isolated workspace lifecycle.""" diff --git a/core/diagnostics/sysinternals_report.py b/core/diagnostics/sysinternals_report.py new file mode 100644 index 00000000..8bb5bebd --- /dev/null +++ b/core/diagnostics/sysinternals_report.py @@ -0,0 +1,105 @@ +"""Discover and report bounded output from supported local Sysinternals tools.""" + +from __future__ import annotations + +import os +from pathlib import Path +from subprocess import TimeoutExpired +from time import monotonic + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess + +TOOLS = ("psfile", "psgetsid", "psinfo", "pslist", "psloggedon", "psloglist") +MAX_OUTPUT_CHARS = 512_000 +MAX_COLLECTION_SECONDS = 45 +MAX_TOOL_SECONDS = 8 + + +class SysinternalsReportCollector(CoreCollector): + """Run available local Sysinternals tools and consolidate their diagnostic output.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated Sysinternals report artifact contract.""" + return CollectorMetadata( + id="core.diagnostics.sysinternals_report", + name="Sysinternals report", + version="4.0.0", + specialty=Specialty.DIAGNOSTICS, + output_media_types=("text/plain",), + description="Reports supported Sysinternals binary/archive state and consolidates available tool output.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_diagnostics",), + default_profiles=("deep",), + timeout_seconds=60, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state before local tool discovery.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Discover local binaries, execute available tools, and write one text report.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Sysinternals collection") + project_root = Path(__file__).resolve().parents[2] + search_roots = ( + project_root / "sysinternals", + project_root / "tools" / "sysinternals", + Path(os.environ.get("ProgramFiles", r"C:\Program Files")) / "Sysinternals", + ) + archive_paths = tuple(root.with_suffix(".zip") for root in search_roots) + sections: list[str] = ["Sysinternals report", ""] + context.report_progress("sysinternals_report_started") + deadline = monotonic() + MAX_COLLECTION_SECONDS + for tool in TOOLS: + binary = next( + (root / f"{tool}.exe" for root in search_roots if (root / f"{tool}.exe").is_file()), + None, + ) + sections.append(f"## {tool}") + if binary is None: + archive_state = "archive available" if any( + path.is_file() for path in archive_paths) else "binary and archive missing" + sections.extend((f"status: {archive_state}", "")) + continue + remaining = deadline - monotonic() + if remaining <= 0: + sections.extend(("status: skipped because the collection time budget was exhausted", "")) + continue + timeout = min(MAX_TOOL_SECONDS, max(1, int(remaining))) + context.report_progress("sysinternals_tool_started", tool=tool, timeout_seconds=timeout) + try: + completed = subprocess.run([str(binary)], capture_output=True, check=False, text=True, timeout=timeout) + except OSError as error: + sections.extend((f"status: execution error: {error}", "")) + continue + except TimeoutExpired as error: + sections.extend((f"status: timed out after {error.timeout} seconds", "")) + continue + output = (completed.stdout + ( + "\n" if completed.stdout and completed.stderr else "") + completed.stderr).strip() + sections.extend((f"status: executed (exit {completed.returncode})", output[:MAX_OUTPUT_CHARS], "")) + report = "\n".join(sections) + output_path = context.workspace / "sysinternals_report.txt" + output_path.write_text(report[:MAX_OUTPUT_CHARS], encoding="utf-8") + artifact = context.artifacts.register_file(output_path, media_type="text/plain") + context.report_progress("sysinternals_report_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("Sysinternals report collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because each tool exits before its output is returned.""" diff --git a/core/encryption/bitlocker_status.py b/core/encryption/bitlocker_status.py new file mode 100644 index 00000000..04eea1bb --- /dev/null +++ b/core/encryption/bitlocker_status.py @@ -0,0 +1,77 @@ +"""Export local BitLocker status as bounded, read-only text evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str, return_code: int) -> bool: + """Recognize textual and HRESULT access-denied results from manage-bde.""" + normalized = detail.casefold() + return return_code == 2147749891 or "permission denied" in normalized or ( + "access" in normalized and "denied" in normalized) + + +class BitlockerStatusCollector(CoreCollector): + """Capture local BitLocker status without changing protectors or encryption state.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated BitLocker-status artifact contract.""" + return CollectorMetadata( + id="core.encryption.bitlocker_status", + name="BitLocker status", + version="4.0.0", + specialty=Specialty.ENCRYPTION, + output_media_types=("text/plain",), + description="Exports local drive BitLocker status through the read-only manage-bde command.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("encryption_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and manage-bde availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("manage-bde") is None: + return ValidationResult(False, reasons=("manage-bde is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run read-only BitLocker status and register its text evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before BitLocker-status collection") + context.report_progress("bitlocker_status_started") + completed = subprocess.run(["manage-bde", "-status"], capture_output=True, check=False, text=True, timeout=40) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"manage-bde exit code {completed.returncode}" + if _is_access_denied(detail, completed.returncode): + return CollectorResult( + CollectorStatus.SKIPPED, + "BitLocker status access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "BitLocker status query failed", errors=(detail,)) + output = context.workspace / "bitlocker_status.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + context.report_progress("bitlocker_status_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("BitLocker status collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because manage-bde exits before the result is returned.""" diff --git a/core/encryption/bitlocker_volumes.py b/core/encryption/bitlocker_volumes.py new file mode 100644 index 00000000..7b825d36 --- /dev/null +++ b/core/encryption/bitlocker_volumes.py @@ -0,0 +1,121 @@ +"""Export PowerShell BitLocker volume data as bounded, read-only JSON evidence.""" + +from __future__ import annotations + +import getpass +import json +import platform +from datetime import UTC, datetime + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import network_adapter as socket +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which, windows_api_adapter + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +def _is_administrator() -> bool | None: + """Return the local administrator token state when Windows can report it.""" + try: + return windows_api_adapter.is_administrator() + except OSError: + return None + + +class BitlockerVolumesCollector(CoreCollector): + """Capture local BitLocker-volume metadata without changing encryption configuration.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated BitLocker-volume JSON artifact contract.""" + return CollectorMetadata( + id="core.encryption.bitlocker_volumes", + name="BitLocker volumes", + version="4.0.0", + specialty=Specialty.ENCRYPTION, + output_media_types=("application/json",), + description="Exports local BitLocker volume metadata through read-only Get-BitLockerVolume.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("encryption_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query BitLocker volumes and register their JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before BitLocker-volume collection") + context.report_progress("bitlocker_volumes_started") + command = ( + "$ErrorActionPreference = 'Stop'; " + "ConvertTo-Json -InputObject @(Get-BitLockerVolume | " + "Select-Object MountPoint, VolumeType, VolumeStatus, ProtectionStatus, EncryptionMethod, " + "EncryptionPercentage, LockStatus, AutoUnlockEnabled) -Depth 4" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "BitLocker volume access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "BitLocker volume query failed", errors=(detail,)) + try: + volumes = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "BitLocker volume query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(volumes, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "BitLocker volume query returned an unexpected result") + report = { + "collected_at": datetime.now(UTC).isoformat(), + "user": getpass.getuser(), + "is_administrator": _is_administrator(), + "hostname": socket.gethostname(), + "platform": platform.platform(), + "volumes": volumes, + } + output = context.workspace / "bitlocker_volumes.json" + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(volumes) if isinstance(volumes, list) else 1 + context.report_progress("bitlocker_volumes_finished", volume_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("BitLocker volumes collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/event_log/application_events.py b/core/event_log/application_events.py new file mode 100644 index 00000000..994cdff2 --- /dev/null +++ b/core/event_log/application_events.py @@ -0,0 +1,102 @@ +"""Export a bounded Windows Application event-log sample as CSV evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + ResourceClass, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + +_MAX_EVENTS = 1_000 + + +def _is_access_denied(detail: str) -> bool: + """Recognize common access-denied wording emitted by PowerShell event queries.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class ApplicationEventsCollector(CoreCollector): + """Capture a bounded CSV sample of Application events without changing event logs.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated Application-event sample and output limit.""" + return CollectorMetadata( + id="core.event_log.application_events", + name="Application event log", + version="4.0.0", + specialty=Specialty.EVENT_LOG, + output_media_types=("text/csv",), + description="Exports up to 1,000 local Windows Application events as CSV.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("event_logs",), + parallel_safe=True, + resource_class=ResourceClass.GENERAL, + default_profiles=("deep",), + timeout_seconds=90, + maximum_output_bytes=8 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before querying events.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query a bounded Application-event sample and register its CSV output.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Application event-log collection") + context.report_progress("application_events_started", maximum_events=_MAX_EVENTS) + command = ( + f"Get-WinEvent -LogName Application -MaxEvents {_MAX_EVENTS} | " + "Select-Object TimeCreated, Id, LevelDisplayName, ProviderName, Message | " + "ConvertTo-Csv -NoTypeInformation" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=75, + ) + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during Application event-log query") + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "Application event-log access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "Application event-log query failed", errors=(detail,)) + output = context.workspace / "application_events.csv" + output.write_text(completed.stdout, encoding="utf-8") + if context.is_cancelled: + output.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during Application event-log export") + artifact = context.artifacts.register_file(output, media_type="text/csv") + event_count = max(0, len(completed.stdout.splitlines()) - 1) + context.report_progress( + "application_events_finished", + event_count=event_count, + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("Application event-log sample collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the event query process has already exited.""" diff --git a/core/event_log/security_events.py b/core/event_log/security_events.py new file mode 100644 index 00000000..8cd005c6 --- /dev/null +++ b/core/event_log/security_events.py @@ -0,0 +1,98 @@ +"""Export a bounded Windows Security event-log sample as CSV evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + ResourceClass, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + +_MAX_EVENTS = 1_000 + + +def _is_access_denied(detail: str) -> bool: + """Recognize common access-denied wording emitted by PowerShell event queries.""" + normalized = detail.casefold() + return "permission denied" in normalized or "unauthorized" in normalized or ( + "access" in normalized and "denied" in normalized) + + +class SecurityEventsCollector(CoreCollector): + """Capture a bounded CSV sample of Security events without changing event logs.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated Security-event sample and output limit.""" + return CollectorMetadata( + id="core.event_log.security_events", + name="Security event log", + version="4.0.0", + specialty=Specialty.EVENT_LOG, + output_media_types=("text/csv",), + description="Exports up to 1,000 local Windows Security events as CSV.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("event_logs",), + parallel_safe=True, + resource_class=ResourceClass.GENERAL, + default_profiles=("deep",), + timeout_seconds=90, + maximum_output_bytes=8 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before querying events.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query a bounded Security-event sample and register its CSV output.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Security event-log collection") + context.report_progress("security_events_started", maximum_events=_MAX_EVENTS) + command = ( + f"Get-WinEvent -LogName Security -MaxEvents {_MAX_EVENTS} | " + "Select-Object TimeCreated, Id, LevelDisplayName, ProviderName, Message | ConvertTo-Csv -NoTypeInformation" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=75, + ) + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during Security event-log query") + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "Security event-log access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "Security event-log query failed", errors=(detail,)) + output = context.workspace / "security_events.csv" + output.write_text(completed.stdout, encoding="utf-8") + if context.is_cancelled: + output.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during Security event-log export") + artifact = context.artifacts.register_file(output, media_type="text/csv") + event_count = max(0, len(completed.stdout.splitlines()) - 1) + context.report_progress("security_events_finished", event_count=event_count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("Security event-log sample collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the event query process has already exited.""" diff --git a/core/event_log/system_events.py b/core/event_log/system_events.py new file mode 100644 index 00000000..ea3b24b2 --- /dev/null +++ b/core/event_log/system_events.py @@ -0,0 +1,98 @@ +"""Export a bounded Windows System event-log sample as CSV evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + ResourceClass, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + +_MAX_EVENTS = 1_000 + + +def _is_access_denied(detail: str) -> bool: + """Recognize common access-denied wording emitted by PowerShell event queries.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class SystemEventsCollector(CoreCollector): + """Capture a bounded CSV sample of Windows System events without changing event logs.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated event-log sample and its output limit.""" + return CollectorMetadata( + id="core.event_log.system_events", + name="System event log", + version="4.0.0", + specialty=Specialty.EVENT_LOG, + output_media_types=("text/csv",), + description="Exports up to 1,000 local Windows System events as CSV.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("event_logs",), + parallel_safe=True, + resource_class=ResourceClass.GENERAL, + default_profiles=("deep",), + timeout_seconds=90, + maximum_output_bytes=8 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before querying events.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query a bounded System event sample and register its CSV output.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before event-log collection") + context.report_progress("system_events_started", maximum_events=_MAX_EVENTS) + command = ( + f"Get-WinEvent -LogName System -MaxEvents {_MAX_EVENTS} | " + "Select-Object TimeCreated, Id, LevelDisplayName, ProviderName, Message | " + "ConvertTo-Csv -NoTypeInformation" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=75, + ) + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during System event-log query") + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "System event-log access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "System event-log query failed", errors=(detail,)) + output = context.workspace / "system_events.csv" + output.write_text(completed.stdout, encoding="utf-8") + if context.is_cancelled: + output.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during System event-log export") + artifact = context.artifacts.register_file(output, media_type="text/csv") + event_count = max(0, len(completed.stdout.splitlines()) - 1) + context.report_progress("system_events_finished", event_count=event_count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("System event-log sample collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the event query process has already exited.""" diff --git a/core/filesystem/sensitive_file_inventory.py b/core/filesystem/sensitive_file_inventory.py new file mode 100644 index 00000000..317a89a9 --- /dev/null +++ b/core/filesystem/sensitive_file_inventory.py @@ -0,0 +1,183 @@ +"""Find and copy bounded sensitive-named files after explicit approval.""" + +from __future__ import annotations + +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + EvidenceKind, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +KEYWORDS = ( + "password", + "secret", + "code", + "login", + "api", + "key", + "token", + "auth", + "credential", + "private", + "certificate", + "ssh", + "pgp", + "wallet", +) +EXTENSIONS = { + ".txt", + ".csv", + ".json", + ".xml", + ".yml", + ".yaml", + ".ini", + ".cfg", + ".conf", + ".log", + ".pdf", + ".doc", + ".docx", + ".xls", + ".xlsx", + ".zip", + ".db", + ".sqlite", + ".pem", + ".key", + ".ppk", +} +MAX_FILE_BYTES = 10 * 1024 * 1024 + + +def _copy(source: Path, destination: Path) -> Path | None: + """Copy one regular source file into the private workspace, preserving metadata.""" + try: + destination.parent.mkdir(parents=True, exist_ok=True) + filesystem_adapter.copy_file(source, destination) + except OSError: + return None + return destination + + +class SensitiveFileInventoryCollector(CoreCollector): + """Search a bounded system-drive subset for sensitive-named supported files.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the explicit-consent sensitive-file inventory artifact contract.""" + return CollectorMetadata( + id="core.filesystem.sensitive_file_inventory", + name="Sensitive file inventory", + version="4.0.0", + specialty=Specialty.FILESYSTEM, + output_media_types=("application/octet-stream",), + description="Finds and copies bounded supported files with sensitive-data keywords in their names.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.FILESYSTEM_READ, Capability.SENSITIVE_FILES), + sensitive_data_categories=("credentials", "private_keys", "personal_documents"), + default_profiles=("deep",), + timeout_seconds=300, + maximum_output_bytes=256 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Validate bounded search settings before walking the requested root.""" + if context.is_cancelled: + return ValidationResult( + False, + reasons=("run cancellation was requested",), + ) + + max_directories = context.setting_int("max_directories", 5_000) + max_matches = context.setting_int("max_matches", 500) + + if max_directories < 1 or max_matches < 1: + return ValidationResult( + False, + reasons=("max_directories and max_matches must be positive",), + ) + + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Find matches and copy them through the scheduler-owned collector worker.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before sensitive-file inventory") + root = Path( + context.setting_str( + "root", + str(filesystem_adapter.system_drive_root()), + ) + ) + max_directories = context.setting_int("max_directories", 5_000) + max_matches = context.setting_int("max_matches", 500) + matches: list[Path] = [] + scanned_directories = 0 + context.report_progress("sensitive_file_inventory_started", root=str(root)) + for directory, directories, filenames in filesystem_adapter.walk(root, onerror=lambda _: None): + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during sensitive-file inventory") + scanned_directories += 1 + if scanned_directories > max_directories or len(matches) >= max_matches: + directories.clear() + break + for filename in filenames: + candidate = Path(directory) / filename + name = filename.casefold() + if candidate.suffix.casefold() not in EXTENSIONS or not any(keyword in name for keyword in KEYWORDS): + continue + try: + if candidate.is_symlink() or candidate.stat().st_size > MAX_FILE_BYTES: + continue + except OSError: + continue + matches.append(candidate) + if len(matches) >= max_matches: + break + destination_root = context.workspace / "sensitive_file_inventory" + copied: list[Path] = [] + for source in matches: + if context.is_cancelled: + for copied_path in copied: + copied_path.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during sensitive-file copy") + destination = destination_root / source.drive.replace(":", "") / source.relative_to(root) + if copied_path := _copy(source, destination): + copied.append(copied_path) + if context.is_cancelled: + for copied_path in copied: + copied_path.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during sensitive-file copy") + if not copied: + return CollectorResult( + CollectorStatus.SKIPPED, + "no sensitive-named supported files met the bounded inventory policy", + ) + artifacts = [] + for path in copied: + if context.is_cancelled: + for unpublished in copied[len(artifacts):]: + unpublished.unlink(missing_ok=True) + return CollectorResult.cancelled("cancelled during sensitive-file registration", tuple(artifacts)) + artifacts.append(context.artifacts.register_file(path, evidence_kind=EvidenceKind.RAW)) + artifact_tuple = tuple(artifacts) + context.report_progress( + "sensitive_file_inventory_finished", + scanned_directories=scanned_directories, + copied_files=len(artifact_tuple), + bytes_written=sum(item.size_bytes for item in artifact_tuple), + ) + return CollectorResult.succeeded("sensitive file inventory collected", artifact_tuple) + + def cleanup(self, context: CollectorContext) -> None: + """Leave copied evidence removal to the isolated workspace lifecycle.""" diff --git a/core/filesystem/startup_folder_entries.py b/core/filesystem/startup_folder_entries.py new file mode 100644 index 00000000..24c732d5 --- /dev/null +++ b/core/filesystem/startup_folder_entries.py @@ -0,0 +1,116 @@ +"""Export bounded metadata for files in the standard Windows Startup folders.""" + +from __future__ import annotations + +import json +import os +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +_MAXIMUM_ENTRIES = 500 + + +def _startup_folders() -> tuple[Path, ...]: + """Return the standard per-user and all-users Startup folders without creating them.""" + folders: list[Path] = [] + appdata = os.environ.get("APPDATA") + programdata = os.environ.get("PROGRAMDATA") + if appdata: + folders.append(Path(appdata) / "Microsoft" / "Windows" / "Start Menu" / "Programs" / "Startup") + if programdata: + folders.append(Path(programdata) / "Microsoft" / "Windows" / "Start Menu" / "Programs" / "Startup") + return tuple(folders) + + +class StartupFolderEntriesCollector(CoreCollector): + """Capture read-only metadata for entries configured to start through Startup folders.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the filesystem-read, bounded Startup-folder JSON artifact contract.""" + return CollectorMetadata( + id="core.filesystem.startup_folder_entries", + name="Startup folder entries", + version="4.0.0", + specialty=Specialty.FILESYSTEM, + output_media_types=("application/json",), + description="Exports names, locations, sizes, and timestamps for up to 500 Startup-folder entries.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.FILESYSTEM_READ,), + sensitive_data_categories=("system_configuration",), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and whether at least one standard folder is available.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if not _startup_folders(): + return ValidationResult(False, reasons=("Windows Startup-folder environment variables are unavailable",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Record bounded metadata for direct files in existing Startup folders.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Startup-folder collection") + context.report_progress("startup_folder_entries_started") + entries: list[dict[str, int | str]] = [] + missing_folders = 0 + for folder in _startup_folders(): + if not folder.is_dir(): + missing_folders += 1 + continue + try: + children = sorted(filesystem_adapter.children(folder), key=lambda path: path.name.casefold()) + except OSError as error: + return CollectorResult( + CollectorStatus.SKIPPED, + "Startup-folder access was denied", + errors=(str(error),), + ) + for child in children: + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during Startup-folder collection") + if len(entries) >= _MAXIMUM_ENTRIES: + break + try: + stat = child.stat() + except OSError: + continue + entries.append( + { + "name": child.name, + "path": str(child), + "is_directory": str(child.is_dir()).lower(), + "size_bytes": stat.st_size, + "modified_epoch": int(stat.st_mtime), + } + ) + output = context.workspace / "startup_folder_entries.json" + output.write_text( + json.dumps({"entries": entries, "missing_folders": missing_folders}, indent=2) + "\n", + encoding="utf-8", + ) + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "startup_folder_entries_finished", + entry_count=len(entries), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("Startup-folder entries collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because collection only reads directory metadata.""" diff --git a/core/filesystem/system_drive_listing.py b/core/filesystem/system_drive_listing.py new file mode 100644 index 00000000..ee63f09d --- /dev/null +++ b/core/filesystem/system_drive_listing.py @@ -0,0 +1,135 @@ +"""Create a bounded recursive listing of the Windows system drive.""" + +from __future__ import annotations + +from collections import deque +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +_DEFAULT_MAX_ENTRIES = 10_000 +_DEFAULT_MAX_DEPTH = 16 +_HARD_MAX_ENTRIES = 50_000 +_HARD_MAX_DEPTH = 32 + + +def _setting(settings: object, key: str, default: int, maximum: int) -> int: + """Return one positive integer setting constrained to its documented maximum.""" + value = settings.get(key, default) if isinstance(settings, dict) else default + return value if isinstance(value, int) and 1 <= value <= maximum else default + + +def _scan_directory(directory: Path) -> tuple[list[str], list[Path]]: + """Return metadata-only entries and non-link child directories for one path.""" + entries: list[str] = [] + children: list[Path] = [] + try: + with filesystem_adapter.scan(directory) as scan: + for entry in scan: + try: + is_directory = entry.is_dir(follow_symlinks=False) + except OSError: + continue + entries.append(f"{entry.path}{'/' if is_directory else ''}") + if is_directory: + children.append(Path(entry.path)) + except OSError: + return [], [] + return sorted(entries), sorted(children) + + +class SystemDriveListingCollector(CoreCollector): + """Capture a bounded recursive directory listing without reading file contents.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the filesystem-read system-drive listing contract.""" + return CollectorMetadata( + id="core.filesystem.system_drive_listing", + name="System drive listing", + version="4.0.0", + specialty=Specialty.FILESYSTEM, + output_media_types=("text/plain",), + description="Exports a configurable, bounded recursive listing of the Windows system drive.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.FILESYSTEM_READ,), + sensitive_data_categories=("filesystem_metadata",), + default_profiles=("deep",), + timeout_seconds=180, + maximum_output_bytes=8 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and system-drive availability before traversal.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + root = filesystem_adapter.system_drive_root() + if not root.is_dir(): + return ValidationResult(False, reasons=(f"system drive is unavailable: {root}",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Traverse deterministically inside the scheduler-owned worker and register text evidence.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before system-drive listing collection") + maximum_entries = _setting(context.settings, "max_entries", _DEFAULT_MAX_ENTRIES, _HARD_MAX_ENTRIES) + maximum_depth = _setting(context.settings, "max_depth", _DEFAULT_MAX_DEPTH, _HARD_MAX_DEPTH) + root = filesystem_adapter.system_drive_root() + lines = [ + f"# system_drive={root}", + f"# max_entries={maximum_entries}", + f"# max_depth={maximum_depth}", + "# scheduling=run_supervisor", + ] + pending = deque([(root, 0)]) + entries = 0 + truncated = False + context.report_progress( + "system_drive_listing_started", + max_entries=maximum_entries, + max_depth=maximum_depth, + scheduling="run_supervisor", + ) + while pending and entries < maximum_entries: + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during system-drive listing collection") + directory, depth = pending.popleft() + names, children = _scan_directory(directory) + for name in names: + if entries >= maximum_entries: + truncated = True + break + lines.append(str(Path(name).relative_to(root))) + entries += 1 + if truncated: + break + if depth < maximum_depth: + pending.extend((child, depth + 1) for child in children) + if pending and entries >= maximum_entries: + truncated = True + if truncated: + lines.append("# truncated=true") + output = context.workspace / "system_drive_listing.txt" + output.write_text("\n".join(lines) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + context.report_progress( + "system_drive_listing_finished", + entry_count=entries, + truncated=str(truncated).lower(), + bytes_written=artifact.size_bytes, + ) + summary = "system-drive listing collected" if not truncated else "system-drive listing collected with configured entry limit" + return CollectorResult.succeeded(summary, (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because traversal retains only collection-local state.""" diff --git a/core/filesystem/system_drive_tree.py b/core/filesystem/system_drive_tree.py new file mode 100644 index 00000000..2ce2fca7 --- /dev/null +++ b/core/filesystem/system_drive_tree.py @@ -0,0 +1,110 @@ +"""Create a bounded recursive tree of the Windows system drive without reading file contents.""" + +from __future__ import annotations + +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +_DEFAULT_MAX_ENTRIES = 5_000 +_DEFAULT_MAX_DEPTH = 12 +_HARD_MAX_ENTRIES = 50_000 +_HARD_MAX_DEPTH = 32 + + +def _bounded_setting(settings: object, key: str, default: int, maximum: int) -> int: + """Return one integer traversal setting constrained to a safe positive range.""" + value = settings.get(key, default) if isinstance(settings, dict) else default + return value if isinstance(value, int) and 1 <= value <= maximum else default + + +class SystemDriveTreeCollector(CoreCollector): + """Capture a bounded recursive system-drive tree without reading file contents.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the filesystem-read, bounded system-drive tree artifact contract.""" + return CollectorMetadata( + id="core.filesystem.system_drive_tree", + name="System drive tree", + version="4.0.0", + specialty=Specialty.FILESYSTEM, + output_media_types=("text/plain",), + description="Exports a configurable bounded recursive directory tree for the Windows system drive.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.FILESYSTEM_READ,), + sensitive_data_categories=("filesystem_metadata",), + default_profiles=("deep",), + timeout_seconds=120, + maximum_output_bytes=8 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and system-drive availability before traversal.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + system_drive = filesystem_adapter.system_drive_root() + if not system_drive.is_dir(): + return ValidationResult(False, reasons=(f"system drive is unavailable: {system_drive}",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Walk a bounded tree and register the metadata-only text evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before system-drive tree collection") + maximum_entries = _bounded_setting(context.settings, "max_entries", _DEFAULT_MAX_ENTRIES, _HARD_MAX_ENTRIES) + maximum_depth = _bounded_setting(context.settings, "max_depth", _DEFAULT_MAX_DEPTH, _HARD_MAX_DEPTH) + root = filesystem_adapter.system_drive_root() + lines = [ + f"# system_drive={root}", + f"# max_entries={maximum_entries}", + f"# max_depth={maximum_depth}", + ] + entries = 0 + skipped = 0 + truncated = False + context.report_progress("system_drive_tree_started", max_entries=maximum_entries, max_depth=maximum_depth) + for current, directories, filenames in filesystem_adapter.walk(root, topdown=True, followlinks=False, + onerror=lambda _error: None): + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during system-drive tree collection") + relative = Path(current).relative_to(root) + depth = len(relative.parts) + if depth >= maximum_depth: + skipped += len(directories) + directories[:] = [] + for name, marker in [(name, "/") for name in directories] + [(name, "") for name in filenames]: + if entries >= maximum_entries: + truncated = True + break + lines.append(f"{relative / name}{marker}") + entries += 1 + if truncated: + break + if truncated: + lines.append("# truncated=true") + output = context.workspace / "system_drive_tree.txt" + output.write_text("\n".join(lines) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + context.report_progress( + "system_drive_tree_finished", + entry_count=entries, + skipped_directories=skipped, + truncated=str(truncated).lower(), + bytes_written=artifact.size_bytes, + ) + summary = "system-drive tree collected" if not truncated else "system-drive tree collected with configured entry limit" + return CollectorResult.succeeded(summary, (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because traversal holds no persistent resources.""" diff --git a/core/hardware/battery_status.py b/core/hardware/battery_status.py new file mode 100644 index 00000000..170d5f8f --- /dev/null +++ b/core/hardware/battery_status.py @@ -0,0 +1,100 @@ +"""Export local Windows battery status as bounded read-only JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class BatteryStatusCollector(CoreCollector): + """Capture read-only battery capacity and charge state through Windows CIM.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated battery-status JSON artifact contract.""" + return CollectorMetadata( + id="core.hardware.battery_status", + name="Battery status", + version="4.0.0", + specialty=Specialty.HARDWARE, + output_media_types=("application/json",), + description="Exports local battery name, status, charge, capacity, and estimated runtime metadata.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=128 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query battery CIM metadata and register a bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before battery-status collection") + context.report_progress("battery_status_started") + command = ( + "Get-CimInstance -ClassName Win32_Battery | " + "Select-Object Name, BatteryStatus, EstimatedChargeRemaining, EstimatedRunTime, DesignCapacity, FullChargeCapacity, Status | " + "ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "battery-status access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "battery-status query failed", errors=(detail,)) + try: + batteries = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "battery-status query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(batteries, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "battery-status query returned an unexpected result") + output = context.workspace / "battery_status.json" + output.write_text(json.dumps(batteries, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(batteries) if isinstance(batteries, list) else 1 + context.report_progress("battery_status_finished", battery_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("battery status collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/hardware/display_adapters.py b/core/hardware/display_adapters.py new file mode 100644 index 00000000..e77df794 --- /dev/null +++ b/core/hardware/display_adapters.py @@ -0,0 +1,89 @@ +"""Export local display-adapter metadata as bounded, read-only JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +class DisplayAdaptersCollector(CoreCollector): + """Capture read-only local display-adapter identities, drivers, and memory through CIM.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated display-adapter JSON artifact contract.""" + return CollectorMetadata( + id="core.hardware.display_adapters", + name="Display adapters", + version="4.0.0", + specialty=Specialty.HARDWARE, + output_media_types=("application/json",), + description="Exports local display-adapter names, driver versions, resolution, and memory metadata.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("hardware_inventory",), + default_profiles=("standard", "deep"), + timeout_seconds=30, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query display-adapter CIM metadata and register a bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before display-adapter collection") + context.report_progress("display_adapters_started") + command = ( + "Get-CimInstance Win32_VideoController | Select-Object Name, AdapterCompatibility, DriverVersion, " + "VideoProcessor, AdapterRAM, CurrentHorizontalResolution, CurrentVerticalResolution, CurrentRefreshRate " + "| ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + status = CollectorStatus.SKIPPED if "denied" in detail.casefold() else CollectorStatus.FAILED + return CollectorResult(status, "display-adapter query failed", errors=(detail,)) + try: + adapters = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "display-adapter query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(adapters, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "display-adapter query returned an unexpected result") + output = context.workspace / "display_adapters.json" + output.write_text(json.dumps(adapters, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(adapters) if isinstance(adapters, list) else 1 + context.report_progress("display_adapters_finished", adapter_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("display adapters collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/hardware/windows_features.py b/core/hardware/windows_features.py new file mode 100644 index 00000000..f0c9ccb5 --- /dev/null +++ b/core/hardware/windows_features.py @@ -0,0 +1,106 @@ +"""Collect Windows optional-feature names and states through a bounded PowerShell query.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common Windows and PowerShell permission-denied error wording.""" + normalized = detail.casefold() + return ( + "permission denied" in normalized + or ("access" in normalized and "denied" in normalized) + or "requires elevation" in normalized + or "elevation is required" in normalized + ) + + +class WindowsFeaturesCollector(CoreCollector): + """Export Windows optional-feature state as a JSON artifact in the collector workspace.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated deep inventory and its bounded output limit.""" + return CollectorMetadata( + id="core.hardware.windows_features", + name="Windows optional features", + version="4.0.0", + specialty=Specialty.HARDWARE, + output_media_types=("application/json",), + description="Exports Windows optional-feature names and enabled states through PowerShell.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + default_profiles=("standard", "deep"), + timeout_seconds=60, + maximum_output_bytes=5 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Verify cancellation state and the PowerShell executable before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run a read-only feature query and register its JSON output when available.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before optional-feature collection") + context.report_progress("windows_features_started") + command = "Get-WindowsOptionalFeature -Online | Select-Object FeatureName, State | ConvertTo-Json -Depth 2" + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=45, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "optional-feature access was denied for the current account", + errors=(detail,), + ) + return CollectorResult( + CollectorStatus.FAILED, + "PowerShell optional-feature query failed", + errors=(detail,), + ) + try: + payload = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "PowerShell optional-feature query returned invalid JSON", + errors=(str(error),), + ) + features = payload if isinstance(payload, list) else [payload] + output = context.workspace / "windows_features.json" + output.write_text(json.dumps(features, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "windows_features_finished", + feature_count=len(features), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("Windows optional features collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the PowerShell command is synchronous.""" diff --git a/core/integration/legacy_code_outputs.py b/core/integration/legacy_code_outputs.py new file mode 100644 index 00000000..34e1d591 --- /dev/null +++ b/core/integration/legacy_code_outputs.py @@ -0,0 +1,202 @@ +"""Import generated evidence from historical CODE checkouts into the v4 artifact catalog.""" + +from __future__ import annotations + +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + EvidenceKind, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +_EVIDENCE_EXTENSIONS = { + ".csv", + ".dot", + ".evtx", + ".html", + ".htm", + ".json", + ".log", + ".reg", + ".svg", + ".txt", + ".xml", + ".zip", +} +_EXCLUDED_DIRECTORIES = { + ".git", + ".idea", + ".mypy_cache", + ".pytest_cache", + ".venv", + "__pycache__", + "lib", + "libs", + "site-packages", + "venv", +} +_MAXIMUM_FILES = 500 +_MAXIMUM_FILE_BYTES = 64 * 1024 * 1024 + + +class LegacyCodeOutputsCollector(CoreCollector): + """Import bounded generated files while excluding executable project material.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the compatibility evidence import and its filesystem-read boundary.""" + return CollectorMetadata( + id="core.integration.legacy_code_outputs", + name="Legacy CODE generated outputs", + version="4.0.0", + specialty=Specialty.INTEGRATION, + output_media_types=("application/octet-stream",), + description="Imports bounded generated evidence from CODE without packaging source or configuration.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.FILESYSTEM_READ,), + sensitive_data_categories=("legacy_generated_evidence",), + default_profiles=("manual",), + timeout_seconds=120, + maximum_output_bytes=512 * 1024 * 1024, + maximum_artifact_bytes=_MAXIMUM_FILE_BYTES, + maximum_artifact_files=_MAXIMUM_FILES, + ) + + @staticmethod + def _code_root() -> Path: + """Resolve the historical directory relative to this checked-in collector source.""" + return Path(__file__).resolve().parents[2] / "CODE" + + def validate(self, context: CollectorContext) -> ValidationResult: + """Skip cleanly when the historical CODE directory is not present.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + code_root = self._code_root() + if not code_root.is_dir() or code_root.is_symlink(): + return ValidationResult(False, reasons=("historical CODE directory is unavailable",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Stream allowlisted generated files into the private workspace and artifact store.""" + if context.is_cancelled: + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled before legacy CODE import", + ) + + code_root = self._code_root().resolve() + candidates: list[Path] = [] + + for path in sorted( + filesystem_adapter.recursive(code_root), + key=lambda candidate: candidate.as_posix(), + ): + if context.is_cancelled: + return CollectorResult.cancelled("cancelled during legacy CODE discovery") + + relative = path.relative_to(code_root) + + if any(part.casefold() in _EXCLUDED_DIRECTORIES for part in relative.parts): + continue + + try: + if ( + not path.is_file() + or path.is_symlink() + or path.resolve().parent != (code_root / relative.parent).resolve() + or path.suffix.casefold() not in _EVIDENCE_EXTENSIONS + or path.stat().st_size > _MAXIMUM_FILE_BYTES + ): + continue + except OSError: + continue + + candidates.append(path) + + if len(candidates) >= _MAXIMUM_FILES: + break + + artifacts = [] + errors: list[str] = [] + + for source in candidates: + if context.is_cancelled: + return CollectorResult.cancelled( + "cancelled during legacy CODE import", + tuple(artifacts), + ) + + relative = source.relative_to(code_root) + destination = context.workspace / "legacy_code" / relative + destination.parent.mkdir(parents=True, exist_ok=True) + + temporary = destination.with_suffix(destination.suffix + ".tmp") + + try: + with ( + open(source, "rb") as source_stream, + open(temporary, "xb") as destination_stream, + ): + while block := source_stream.read(1024 * 1024): + if context.is_cancelled: + raise InterruptedError("run cancellation was requested") + + destination_stream.write(block) + + temporary.replace(destination) + + artifacts.append( + context.artifacts.register_file( + destination, + media_type="application/octet-stream", + evidence_kind=EvidenceKind.RAW, + transformations=("imported from historical CODE output",), + ) + ) + + except InterruptedError: + temporary.unlink(missing_ok=True) + destination.unlink(missing_ok=True) + + return CollectorResult.cancelled( + "cancelled during legacy CODE import", + tuple(artifacts), + ) + + except OSError as error: + temporary.unlink(missing_ok=True) + errors.append(f"{relative.as_posix()}: {error}") + + if not artifacts and not errors: + return CollectorResult.skipped("no generated legacy CODE evidence matched the import policy") + + if errors: + return CollectorResult.partial( + "legacy CODE evidence imported with file-level failures", + tuple(artifacts), + errors=tuple(errors), + ) + + context.report_progress( + "legacy_code_outputs_finished", + imported_files=len(artifacts), + bytes_written=sum(artifact.size_bytes for artifact in artifacts), + ) + + return CollectorResult.succeeded( + "legacy CODE evidence imported", + tuple(artifacts), + ) + + def cleanup(self, context: CollectorContext) -> None: + """Remove only an unpublished temporary copy left by an interrupted import.""" + for temporary in filesystem_adapter.recursive(context.workspace, "*.tmp"): + temporary.unlink(missing_ok=True) diff --git a/core/media/media_backup.py b/core/media/media_backup.py new file mode 100644 index 00000000..612ae692 --- /dev/null +++ b/core/media/media_backup.py @@ -0,0 +1,130 @@ +"""Back up bounded current-user Pictures and Videos evidence after explicit approval.""" + +from __future__ import annotations + +from datetime import UTC, datetime +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + EvidenceKind, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +SUPPORTED_EXTENSIONS = {".jpg", ".jpeg", ".png", ".mp4"} +MAX_FILE_BYTES = 50 * 1024 * 1024 +MAX_TOTAL_BYTES = 512 * 1024 * 1024 +MAX_FILES = 1_000 + + +class MediaBackupCollector(CoreCollector): + """Copy supported current-user media into the collector's private workspace.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the explicit-consent media backup artifact contract.""" + return CollectorMetadata( + id="core.media.media_backup", + name="Pictures and Videos backup", + version="4.0.0", + specialty=Specialty.MEDIA, + output_media_types=("application/octet-stream",), + description="Copies bounded JPG, JPEG, PNG, and MP4 files from current-user Pictures and Videos folders.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.FILESYSTEM_READ, Capability.SENSITIVE_FILES), + sensitive_data_categories=("personal_media",), + default_profiles=("deep",), + timeout_seconds=300, + maximum_output_bytes=512 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state before the bounded media scan starts.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Copy supported regular media files while preserving source metadata.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before media backup") + home = filesystem_adapter.home() + source_roots = (("pictures", home / "Pictures"), ("videos", home / "Videos")) + destination_root = context.workspace / "media_backup" + copied: list[Path] = [] + copied_bytes = 0 + skipped_files = 0 + inaccessible_roots = 0 + context.report_progress("media_backup_started") + for category, source_root in source_roots: + try: + exists = source_root.is_dir() + except OSError: + inaccessible_roots += 1 + continue + if not exists: + continue + try: + candidates = filesystem_adapter.recursive(source_root) + for candidate in candidates: + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during media backup") + if len(copied) >= MAX_FILES: + break + try: + if not candidate.is_file() or candidate.is_symlink() or candidate.suffix.casefold() not in SUPPORTED_EXTENSIONS: + continue + size = candidate.stat().st_size + except OSError: + skipped_files += 1 + continue + if size > MAX_FILE_BYTES or copied_bytes + size > MAX_TOTAL_BYTES: + skipped_files += 1 + continue + timestamp = datetime.fromtimestamp(candidate.stat().st_mtime, UTC).strftime("%Y%m%dT%H%M%SZ") + destination = destination_root / category / f"{candidate.stem}_{timestamp}{candidate.suffix.casefold()}" + suffix = 1 + while destination.exists(): + destination = destination_root / category / f"{candidate.stem}_{timestamp}_{suffix}{candidate.suffix.casefold()}" + suffix += 1 + destination.parent.mkdir(parents=True, exist_ok=True) + filesystem_adapter.copy_file(candidate, destination) + if context.is_cancelled: + destination.unlink(missing_ok=True) + for copied_path in copied: + copied_path.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during media backup") + copied.append(destination) + copied_bytes += size + except OSError: + inaccessible_roots += 1 + if not copied: + reason = "no supported media files met the bounded backup policy" + if inaccessible_roots == len(source_roots): + reason = "the current user's Pictures and Videos folders are inaccessible" + return CollectorResult(CollectorStatus.SKIPPED, reason) + artifacts = [] + for path in copied: + if context.is_cancelled: + for unpublished in copied[len(artifacts):]: + unpublished.unlink(missing_ok=True) + return CollectorResult.cancelled("cancelled during media registration", tuple(artifacts)) + artifacts.append(context.artifacts.register_file(path, evidence_kind=EvidenceKind.RAW)) + artifact_tuple = tuple(artifacts) + context.report_progress( + "media_backup_finished", + copied_files=len(artifacts), + skipped_files=skipped_files, + bytes_written=sum(item.size_bytes for item in artifact_tuple), + ) + return CollectorResult.succeeded("current-user media backup collected", artifact_tuple) + + def cleanup(self, context: CollectorContext) -> None: + """Leave copied evidence removal to the isolated workspace lifecycle.""" diff --git a/core/memory/memory_snapshot.py b/core/memory/memory_snapshot.py new file mode 100644 index 00000000..ff507126 --- /dev/null +++ b/core/memory/memory_snapshot.py @@ -0,0 +1,85 @@ +"""Collect bounded aggregate physical and virtual memory statistics from Windows.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime + +from logicytics import ( + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, + global_memory_status, +) +from logicytics.contracts import CollectorContext, CollectorStatus + + +class MemorySnapshotCollector(CoreCollector): + """Write a JSON report containing aggregate memory and page-file availability.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare this bounded, capability-free Windows memory collector.""" + return CollectorMetadata( + id="core.memory.memory_snapshot", + name="Memory snapshot", + version="4.0.0", + specialty=Specialty.MEMORY, + output_media_types=("application/json",), + description="Captures aggregate physical, virtual, and page-file memory statistics.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(), + default_profiles=("minimal", "standard", "deep", "offline"), + timeout_seconds=10, + maximum_output_bytes=64 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation and Windows memory API availability without writing artifacts.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + try: + global_memory_status() + except OSError as error: + return ValidationResult(False, reasons=(f"memory API is unavailable: {error}",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Capture aggregate counters and register the resulting JSON report.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before memory collection") + context.report_progress("memory_snapshot_started") + try: + status = global_memory_status() + except OSError as error: + return CollectorResult(CollectorStatus.FAILED, "could not read Windows memory status", errors=(str(error),)) + report = { + "collected_at": datetime.now(UTC).isoformat(), + "memory_load_percent": status.dwMemoryLoad, + "physical_memory": { + "total_bytes": status.ullTotalPhys, + "available_bytes": status.ullAvailPhys, + "used_bytes": status.ullTotalPhys - status.ullAvailPhys, + }, + "page_file": { + "total_bytes": status.ullTotalPageFile, + "available_bytes": status.ullAvailPageFile, + "used_bytes": status.ullTotalPageFile - status.ullAvailPageFile, + }, + "virtual_memory": { + "total_bytes": status.ullTotalVirtual, + "available_bytes": status.ullAvailVirtual, + "used_bytes": status.ullTotalVirtual - status.ullAvailVirtual, + }, + } + output = context.workspace / "memory_snapshot.json" + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("memory_snapshot_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("memory snapshot collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the Windows memory API is synchronous.""" diff --git a/core/network/active_connections.py b/core/network/active_connections.py new file mode 100644 index 00000000..c31d09cc --- /dev/null +++ b/core/network/active_connections.py @@ -0,0 +1,87 @@ +"""Export local active Windows connections and owning PIDs as bounded text evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from netstat output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class ActiveConnectionsCollector(CoreCollector): + """Capture active TCP and UDP endpoint metadata without network probing or changes.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated active-connection artifact contract.""" + return CollectorMetadata( + id="core.network.active_connections", + name="Active network connections", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("text/plain",), + description="Exports active TCP/UDP endpoints, states, and owning PIDs from netstat.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_identifiers", "process_metadata"), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=4 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and netstat availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("netstat") is None: + return ValidationResult(False, reasons=("netstat is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run the read-only connection query and register its text evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before connection collection") + context.report_progress("active_connections_started") + completed = subprocess.run( + ["netstat", "-ano"], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"netstat exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "netstat access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "active-connection query failed", errors=(detail,)) + output = context.workspace / "active_connections.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + connection_count = sum(1 for line in completed.stdout.splitlines() if line.lstrip().startswith(("TCP", "UDP"))) + context.report_progress( + "active_connections_finished", + connection_count=connection_count, + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("active network connections collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because netstat exits before the result is returned.""" diff --git a/core/network/adapter_statistics.py b/core/network/adapter_statistics.py new file mode 100644 index 00000000..bda39a05 --- /dev/null +++ b/core/network/adapter_statistics.py @@ -0,0 +1,101 @@ +"""Export local Windows network adapter I/O counters as bounded JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class AdapterStatisticsCollector(CoreCollector): + """Capture read-only per-interface byte, packet, error, and discard counters.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated adapter-statistics artifact contract.""" + return CollectorMetadata( + id="core.network.adapter_statistics", + name="Network adapter statistics", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("application/json",), + description="Exports per-interface network byte, packet, error, and discard counters.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_identifiers",), + default_profiles=("deep",), + timeout_seconds=45, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query adapter counters and register their JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before adapter-statistics collection") + context.report_progress("adapter_statistics_started") + command = ( + "Get-NetAdapterStatistics | " + "Select-Object Name, ReceivedBytes, SentBytes, ReceivedUnicastPackets, SentUnicastPackets, " + "ReceivedDiscardedPackets, OutboundDiscardedPackets, ReceivedPacketErrors, OutboundPacketErrors | " + "ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "adapter-statistics access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "adapter-statistics query failed", errors=(detail,)) + try: + statistics = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "adapter-statistics query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(statistics, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "adapter-statistics query returned an unexpected result") + output = context.workspace / "adapter_statistics.json" + output.write_text(json.dumps(statistics, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(statistics) if isinstance(statistics, list) else 1 + context.report_progress("adapter_statistics_finished", interface_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("network adapter statistics collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/network/arp_cache.py b/core/network/arp_cache.py new file mode 100644 index 00000000..d69d92d9 --- /dev/null +++ b/core/network/arp_cache.py @@ -0,0 +1,83 @@ +"""Export the local Windows ARP cache as bounded, read-only network evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from Windows command output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class ArpCacheCollector(CoreCollector): + """Capture current ARP mappings without probing or modifying the network.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated, bounded ARP-cache report.""" + return CollectorMetadata( + id="core.network.arp_cache", + name="ARP cache", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("text/plain",), + description="Exports the local ARP cache without sending network traffic.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_identifiers",), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=1 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and arp availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("arp") is None: + return ValidationResult(False, reasons=("arp is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run the read-only ARP query and register its text artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before ARP-cache collection") + context.report_progress("arp_cache_started") + completed = subprocess.run( + ["arp", "-a"], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"arp exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "ARP-cache access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "ARP-cache query failed", errors=(detail,)) + output = context.workspace / "arp_cache.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + line_count = sum(1 for line in completed.stdout.splitlines() if line.strip()) + context.report_progress("arp_cache_finished", line_count=line_count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("ARP-cache report collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because arp exits before the result is returned.""" diff --git a/core/network/bandwidth_sample.py b/core/network/bandwidth_sample.py new file mode 100644 index 00000000..5cc24d17 --- /dev/null +++ b/core/network/bandwidth_sample.py @@ -0,0 +1,169 @@ +"""Measure local adapter bandwidth from bounded read-only counter samples.""" + +from __future__ import annotations + +import json +import time + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + +_DEFAULT_SAMPLES = 3 +_DEFAULT_INTERVAL_SECONDS = 1 + + +def _setting(settings: object, key: str, default: int, maximum: int) -> int: + """Return one bounded positive integer collector setting.""" + value = settings.get(key, default) if isinstance(settings, dict) else default + return value if isinstance(value, int) and 1 <= value <= maximum else default + + +def _interval_setting(settings: object) -> float: + """Return the configured finite positive sampling interval.""" + value = settings.get("interval_seconds", _DEFAULT_INTERVAL_SECONDS) if isinstance(settings, + dict) else _DEFAULT_INTERVAL_SECONDS + if isinstance(value, (int, float)) and not isinstance(value, bool) and 0.1 <= value <= 60: + return float(value) + return float(_DEFAULT_INTERVAL_SECONDS) + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class BandwidthSampleCollector(CoreCollector): + """Measure receive/send byte rates locally without generating network traffic.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated bounded bandwidth-sampling artifact contract.""" + return CollectorMetadata( + id="core.network.bandwidth_sample", + name="Network bandwidth sample", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("application/json",), + description="Calculates local per-interface average and peak bandwidth from adapter counter samples.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_identifiers",), + default_profiles=("deep",), + timeout_seconds=90, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before sampling.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Take bounded counter samples and register calculated rate evidence as JSON.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before bandwidth sampling") + samples = _setting(context.settings, "sample_count", _DEFAULT_SAMPLES, 10) + interval = _interval_setting(context.settings) + command = "Get-NetAdapterStatistics | Select-Object Name, ReceivedBytes, SentBytes | ConvertTo-Json -Depth 3" + observations: list[dict[str, dict[str, int]]] = [] + context.report_progress("bandwidth_sample_started", sample_count=samples, interval_seconds=interval) + for index in range(samples): + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during bandwidth sampling") + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=20, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "bandwidth-sample access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "bandwidth-sample query failed", errors=(detail,)) + try: + raw = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "bandwidth-sample query returned invalid JSON", + errors=(str(error),), + ) + records = raw if isinstance(raw, list) else [raw] + if not all(isinstance(record, dict) for record in records): + return CollectorResult(CollectorStatus.FAILED, "bandwidth-sample query returned an unexpected result") + observations.append( + { + str(record.get("Name", "unavailable")): { + "received": int(record.get("ReceivedBytes", 0)), + "sent": int(record.get("SentBytes", 0)), + } + for record in records + } + ) + if index + 1 < samples: + time.sleep(interval) + rates: dict[str, dict[str, float]] = {} + for previous, current in zip(observations, observations[1:]): + for name, counters in current.items(): + if name not in previous: + continue + receive = max(0, counters["received"] - previous[name]["received"]) / interval + sent = max(0, counters["sent"] - previous[name]["sent"]) / interval + values = rates.setdefault( + name, + { + "receive_average_bytes_per_second": 0.0, + "receive_peak_bytes_per_second": 0.0, + "send_average_bytes_per_second": 0.0, + "send_peak_bytes_per_second": 0.0, + "sample_intervals": 0.0, + }, + ) + values["sample_intervals"] += 1 + values["receive_average_bytes_per_second"] += receive + values["send_average_bytes_per_second"] += sent + values["receive_peak_bytes_per_second"] = max(values["receive_peak_bytes_per_second"], receive) + values["send_peak_bytes_per_second"] = max(values["send_peak_bytes_per_second"], sent) + for values in rates.values(): + values["receive_average_bytes_per_second"] /= values["sample_intervals"] + values["send_average_bytes_per_second"] /= values["sample_intervals"] + output = context.workspace / "bandwidth_sample.json" + output.write_text( + json.dumps( + {"sample_count": samples, "interval_seconds": interval, "interfaces": rates}, + indent=2, + sort_keys=True, + ) + + "\n", + encoding="utf-8", + ) + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "bandwidth_sample_finished", + interface_count=len(rates), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("network bandwidth sample collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because sampling subprocesses have already exited.""" diff --git a/core/network/connection_processes.py b/core/network/connection_processes.py new file mode 100644 index 00000000..d502f11b --- /dev/null +++ b/core/network/connection_processes.py @@ -0,0 +1,143 @@ +"""Export local active connections correlated with process names as bounded CSV evidence.""" + +from __future__ import annotations + +import csv +import io + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from Windows command output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +def _parse_connections(netstat_output: str) -> list[dict[str, str]]: + """Parse the stable token layout of Windows netstat -ano connection rows.""" + connections: list[dict[str, str]] = [] + for line in netstat_output.splitlines(): + fields = line.split() + if not fields or fields[0] not in {"TCP", "UDP"}: + continue + if fields[0] == "TCP" and len(fields) >= 5: + protocol, local, remote, state, pid = fields[:5] + elif fields[0] == "UDP" and len(fields) >= 4: + protocol, local, remote, pid = fields[:4] + state = "" + else: + continue + connections.append( + { + "protocol": protocol, + "local_endpoint": local, + "remote_endpoint": remote, + "state": state, + "pid": pid, + } + ) + return connections + + +def _parse_processes(tasklist_output: str) -> dict[str, str]: + """Map tasklist CSV process identifiers to their image names.""" + processes: dict[str, str] = {} + for row in csv.reader(io.StringIO(tasklist_output)): + if len(row) >= 2 and row[1].isdigit(): + processes[row[1]] = row[0] + return processes + + +class ConnectionProcessesCollector(CoreCollector): + """Capture active connection endpoints with the local process name for each PID.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated connection-process CSV artifact contract.""" + return CollectorMetadata( + id="core.network.connection_processes", + name="Connection process associations", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("text/csv",), + description="Correlates active Netstat TCP/UDP endpoints with local process names and PIDs.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_identifiers", "process_metadata"), + default_profiles=("deep",), + timeout_seconds=45, + maximum_output_bytes=4 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and required Windows command availability.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + missing = tuple(command for command in ("netstat", "tasklist") if which(command) is None) + if missing: + return ValidationResult(False, reasons=(f"required command is unavailable: {', '.join(missing)}",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Read Netstat and Tasklist data, then register their correlation as CSV.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before connection-process collection") + context.report_progress("connection_processes_started") + netstat = subprocess.run(["netstat", "-ano"], capture_output=True, check=False, text=True, timeout=25) + tasklist = subprocess.run( + ["tasklist", "/fo", "csv", "/nh"], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + failures = [result for result in (netstat, tasklist) if result.returncode != 0] + if failures: + detail = "\n".join(result.stderr.strip() or f"command exit code {result.returncode}" for result in failures) + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "connection-process access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "connection-process query failed", errors=(detail,)) + processes = _parse_processes(tasklist.stdout) + rows = _parse_connections(netstat.stdout) + output = context.workspace / "connection_processes.csv" + with output.open("w", encoding="utf-8", newline="") as stream: + writer = csv.DictWriter( + stream, + fieldnames=( + "protocol", + "local_endpoint", + "remote_endpoint", + "state", + "pid", + "process_name", + ), + ) + writer.writeheader() + for row in rows: + writer.writerow({**row, "process_name": processes.get(row["pid"], "unavailable")}) + artifact = context.artifacts.register_file(output, media_type="text/csv") + context.report_progress( + "connection_processes_finished", + connection_count=len(rows), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("connection-process associations collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because both commands exit before the result is returned.""" diff --git a/core/network/dns_cache.py b/core/network/dns_cache.py new file mode 100644 index 00000000..8af06f18 --- /dev/null +++ b/core/network/dns_cache.py @@ -0,0 +1,76 @@ +"""Export the local Windows DNS resolver cache as bounded sensitive text evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from ipconfig output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class DnsCacheCollector(CoreCollector): + """Capture the local DNS resolver cache without modifying resolver state or sending traffic.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated sensitive DNS-cache artifact contract.""" + return CollectorMetadata( + id="core.network.dns_cache", + name="DNS resolver cache", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("text/plain",), + description="Exports local DNS resolver cache records through read-only ipconfig output.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("dns_history",), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=4 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and ipconfig availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("ipconfig") is None: + return ValidationResult(False, reasons=("ipconfig is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run the read-only DNS-cache query and register its bounded text artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before DNS-cache collection") + context.report_progress("dns_cache_started") + completed = subprocess.run(["ipconfig", "/displaydns"], capture_output=True, check=False, text=True, timeout=25) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"ipconfig exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "DNS-cache access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "DNS-cache query failed", errors=(detail,)) + output = context.workspace / "dns_cache.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + context.report_progress("dns_cache_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("DNS resolver cache collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because ipconfig exits before the result is returned.""" diff --git a/core/network/firewall_profiles.py b/core/network/firewall_profiles.py new file mode 100644 index 00000000..62730961 --- /dev/null +++ b/core/network/firewall_profiles.py @@ -0,0 +1,102 @@ +"""Export local Windows firewall profile settings as bounded read-only JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class FirewallProfilesCollector(CoreCollector): + """Capture read-only local firewall profile configuration without changing firewall state.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated firewall-profile JSON artifact contract.""" + return CollectorMetadata( + id="core.network.firewall_profiles", + name="Firewall profiles", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("application/json",), + description="Exports local Domain, Private, and Public Windows firewall profile settings.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query firewall profiles and register their bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before firewall-profile collection") + context.report_progress("firewall_profiles_started") + command = ( + "Get-NetFirewallProfile | " + "Select-Object Name, Enabled, DefaultInboundAction, DefaultOutboundAction, " + "NotifyOnListen, AllowInboundRules, AllowLocalFirewallRules, " + "AllowLocalIPsecRules, LogFileName | " + "ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "firewall-profile access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "firewall-profile query failed", errors=(detail,)) + try: + profiles = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "firewall-profile query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(profiles, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "firewall-profile query returned an unexpected result") + output = context.workspace / "firewall_profiles.json" + output.write_text(json.dumps(profiles, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(profiles) if isinstance(profiles, list) else 1 + context.report_progress("firewall_profiles_finished", profile_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("firewall profiles collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/network/network_adapters.py b/core/network/network_adapters.py new file mode 100644 index 00000000..155b0904 --- /dev/null +++ b/core/network/network_adapters.py @@ -0,0 +1,81 @@ +"""Collect the local Windows IP configuration as a bounded text artifact.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +class NetworkAdaptersCollector(CoreCollector): + """Run the local IP configuration command and register its output as evidence.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated adapter inventory and its fixed output bound.""" + return CollectorMetadata( + id="core.network.network_adapters", + name="Network adapters", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("text/plain",), + description="Captures local Windows IP configuration and adapter details through ipconfig.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + default_profiles=("standard", "deep"), + timeout_seconds=20, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Confirm cancellation state and that ipconfig is available on the host.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("ipconfig") is None: + return ValidationResult(False, reasons=("ipconfig is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Capture local adapter configuration and register the bounded plain-text report.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before adapter collection") + context.report_progress("network_adapters_started") + completed = subprocess.run( + ["ipconfig", "/all"], + capture_output=True, + check=False, + text=True, + timeout=15, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"ipconfig exit code {completed.returncode}" + if "permission denied" in detail.casefold() or ( + "access" in detail.casefold() and "denied" in detail.casefold()): + return CollectorResult( + CollectorStatus.SKIPPED, + "ipconfig access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "ipconfig failed", errors=(detail,)) + output = context.workspace / "network_adapters.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + adapter_sections = completed.stdout.casefold().count("adapter ") + context.report_progress( + "network_adapters_finished", + adapter_sections=adapter_sections, + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("network adapter configuration collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the ipconfig child process has completed.""" diff --git a/core/network/network_identity.py b/core/network/network_identity.py new file mode 100644 index 00000000..4d67ba12 --- /dev/null +++ b/core/network/network_identity.py @@ -0,0 +1,89 @@ +"""Collect the hostname and resolver-provided addresses for the local host.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + NetworkAccess, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import network_adapter as socket + + +class NetworkIdentityCollector(CoreCollector): + """Capture a bounded local network identity report without probing remote hosts.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare resolver access and the small JSON output produced by this collector.""" + return CollectorMetadata( + id="core.network.network_identity", + name="Network identity", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("application/json",), + description="Records the local hostname and resolver-provided IP addresses without probing hosts.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.NETWORK,), + network_access=NetworkAccess.LOCAL, + default_profiles=("standard", "deep"), + timeout_seconds=10, + maximum_output_bytes=64 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Confirm that the run has not been cancelled before resolving local identity.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Resolve the local hostname and write a deterministic JSON report.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before network identity collection") + context.report_progress("network_identity_started") + hostname = socket.gethostname() + addresses: list[str] = [] + resolver_error: str | None = None + try: + for result in socket.getaddrinfo(hostname, None): + address = result[4][0] + if address not in addresses: + addresses.append(address) + except socket.gaierror as error: + resolver_error = str(error) + report = { + "collected_at": datetime.now(UTC).isoformat(), + "hostname": hostname, + "addresses": sorted(addresses), + } + if resolver_error is not None: + report["resolver_error"] = resolver_error + output = context.workspace / "network_identity.json" + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "network_identity_finished", + address_count=len(addresses), + bytes_written=artifact.size_bytes, + ) + if resolver_error is not None: + return CollectorResult( + CollectorStatus.PARTIAL, + "network identity collected without resolver addresses", + (artifact,), + errors=(resolver_error,), + ) + return CollectorResult.succeeded("network identity collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because hostname resolution has already completed.""" diff --git a/core/network/network_interfaces.py b/core/network/network_interfaces.py new file mode 100644 index 00000000..916f1497 --- /dev/null +++ b/core/network/network_interfaces.py @@ -0,0 +1,143 @@ +"""Export local Windows interface addresses and link state as bounded JSON evidence.""" + +from __future__ import annotations + +import ipaddress +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +def _enrich_ipv4_networks(records: list[dict[str, object]]) -> list[dict[str, object]]: + """Add deterministic IPv4 netmask and broadcast fields from each prefix length.""" + enriched: list[dict[str, object]] = [] + + for record in records: + item = dict(record) + address = item.get("IPAddress") + prefix = item.get("PrefixLength") + + if isinstance(address, str) and isinstance(prefix, int): + try: + network = ipaddress.IPv4Network( + f"{address}/{prefix}", + strict=False, + ) + except ValueError: + pass + else: + item["Netmask"] = network.netmask.compressed + item["BroadcastAddress"] = network.broadcast_address.compressed + + enriched.append(item) + + return enriched + + +class NetworkInterfacesCollector(CoreCollector): + """Capture read-only IPv4 addresses, masks, broadcasts, and adapter link metadata.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated network-interface JSON artifact contract.""" + return CollectorMetadata( + id="core.network.network_interfaces", + name="Network interfaces", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("application/json",), + description="Exports IPv4 addresses, masks, broadcasts, link states, speeds, and duplex data.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_identifiers",), + default_profiles=("deep",), + timeout_seconds=45, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query interface address and link data, then register JSON evidence.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before network-interface collection") + context.report_progress("network_interfaces_started") + command = ( + "$adapters = @(Get-NetAdapter); " + "Get-NetIPAddress -AddressFamily IPv4 | ForEach-Object { " + "$adapter = $adapters | Where-Object ifIndex -eq $_.InterfaceIndex | Select-Object -First 1; " + "$status = if ($null -eq $adapter) { 'Unknown' } else { $adapter.Status.ToString() }; " + "$linkSpeed = if ($null -eq $adapter) { $null } else { $adapter.LinkSpeed }; " + "$mediaConnectionState = if ($null -eq $adapter) { 'Unknown' } else { $adapter.MediaConnectionState.ToString() }; " + "$fullDuplex = if ($null -eq $adapter) { $null } else { $adapter.FullDuplex }; " + "[pscustomobject]@{ InterfaceAlias = $_.InterfaceAlias; IPAddress = $_.IPAddress; " + "PrefixLength = $_.PrefixLength; AddressState = $_.AddressState.ToString(); " + "Status = $status; LinkSpeed = $linkSpeed; " + "MediaConnectionState = $mediaConnectionState; FullDuplex = $fullDuplex } " + "} | ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "network-interface access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "network-interface query failed", errors=(detail,)) + try: + interfaces = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "network-interface query returned invalid JSON", + errors=(str(error),), + ) + records = interfaces if isinstance(interfaces, list) else [interfaces] + if not all(isinstance(record, dict) for record in records): + return CollectorResult(CollectorStatus.FAILED, "network-interface query returned an unexpected result") + output = context.workspace / "network_interfaces.json" + output.write_text( + json.dumps(_enrich_ipv4_networks(records), indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "network_interfaces_finished", + interface_count=len(records), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("network-interface inventory collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/network/routing_table.py b/core/network/routing_table.py new file mode 100644 index 00000000..df8faf67 --- /dev/null +++ b/core/network/routing_table.py @@ -0,0 +1,83 @@ +"""Export the local Windows routing table as bounded, read-only network evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from Windows command output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class RoutingTableCollector(CoreCollector): + """Capture the local routing table without changing routes or sending traffic.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated, bounded routing-table report.""" + return CollectorMetadata( + id="core.network.routing_table", + name="Routing table", + version="4.0.0", + specialty=Specialty.NETWORK, + output_media_types=("text/plain",), + description="Exports the local IPv4 and IPv6 routing tables without changing routes.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_identifiers",), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=1 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and route availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("route") is None: + return ValidationResult(False, reasons=("route is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run the read-only route query and register its text artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before routing-table collection") + context.report_progress("routing_table_started") + completed = subprocess.run( + ["route", "print"], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"route exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "routing-table access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "routing-table query failed", errors=(detail,)) + output = context.workspace / "routing_table.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + line_count = sum(1 for line in completed.stdout.splitlines() if line.strip()) + context.report_progress("routing_table_finished", line_count=line_count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("routing-table report collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because route exits before the result is returned.""" diff --git a/core/packet/connection_graph.py b/core/packet/connection_graph.py new file mode 100644 index 00000000..e8ef3975 --- /dev/null +++ b/core/packet/connection_graph.py @@ -0,0 +1,102 @@ +"""Export a labeled local connection graph in DOT format.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common Windows permission-denied wording.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +def render_connection_graph(command_output: str) -> str: + """Normalize netstat rows into deterministic, sorted Graphviz DOT source.""" + edges: set[tuple[str, str, str]] = set() + for line in command_output.splitlines(): + fields = line.split() + if len(fields) < 4 or fields[0].upper() not in {"TCP", "UDP"}: + continue + protocol, source, destination = fields[:3] + if destination in {"*:*", "*"}: + continue + edges.add((source, destination, protocol.upper())) + + def quote(value: str) -> str: + """Escape one label for safe inclusion in a quoted DOT string.""" + return '"' + value.replace("\\", "\\\\").replace('"', '\\"') + '"' + + lines = ["digraph connection_graph {", " rankdir=LR;"] + lines.extend( + f" {quote(source)} -> {quote(destination)} [label={quote(protocol)}];" for source, destination, protocol in + sorted(edges)) + lines.append("}") + return "\n".join(lines) + "\n" + + +class ConnectionGraphCollector(CoreCollector): + """Build a DOT connection graph without retaining packet payloads.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated network graph artifact contract.""" + return CollectorMetadata( + id="core.packet.connection_graph", + name="Connection graph", + version="4.0.0", + specialty=Specialty.PACKET, + output_media_types=("text/vnd.graphviz",), + description="Exports a DOT source/destination graph with TCP or UDP protocol edge labels.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_metadata",), + default_profiles=("deep",), + timeout_seconds=45, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and netstat availability before graph collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("netstat") is None: + return ValidationResult(False, reasons=("netstat is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Read active connections and write a bounded DOT graph artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before connection graph collection") + context.report_progress("connection_graph_started") + completed = subprocess.run(["netstat", "-ano"], capture_output=True, check=False, text=True, timeout=40) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"netstat exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "connection graph access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "connection graph query failed", errors=(detail,)) + rendered = render_connection_graph(completed.stdout) + output = context.workspace / "connection_graph.dot" + output.write_text(rendered, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/vnd.graphviz") + edge_count = sum(1 for line in rendered.splitlines() if " -> " in line) + context.report_progress("connection_graph_finished", edge_count=edge_count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("connection graph collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Retain no graph or plot state; DOT generation uses only collection-local values.""" diff --git a/core/packet/packet_capture.py b/core/packet/packet_capture.py new file mode 100644 index 00000000..933958fe --- /dev/null +++ b/core/packet/packet_capture.py @@ -0,0 +1,259 @@ +"""Capture bounded local IPv4 packet observations after explicit approval.""" + +from __future__ import annotations + +import csv +import struct +import time + +import select + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + NetworkAccess, + PrivilegeLevel, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import network_adapter as socket + + +def _packet_row(payload: bytes) -> dict[str, str] | None: + """Decode a minimal IPv4/TCP/UDP/ICMP observation without retaining payload data.""" + if len(payload) < 20 or payload[0] >> 4 != 4: + return None + header_length = (payload[0] & 15) * 4 + if header_length < 20 or len(payload) < header_length: + return None + protocol = payload[9] + names = {1: "ICMP", 6: "TCP", 17: "UDP"} + source = socket.inet_ntoa(payload[12:16]) + destination = socket.inet_ntoa(payload[16:20]) + source_port = destination_port = "" + if protocol in {6, 17} and len(payload) >= header_length + 4: + source_port, destination_port = (str(value) for value in + struct.unpack("!HH", payload[header_length: header_length + 4])) + return { + "source_ip": source, + "destination_ip": destination, + "protocol": names.get(protocol, str(protocol)), + "source_port": source_port, + "destination_port": destination_port, + "packet_bytes": str(len(payload)), + } + + +class PacketCaptureCollector(CoreCollector): + """Capture metadata-only packet observations inside the isolated worker process.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the explicitly approved packet-capture CSV artifact contract.""" + return CollectorMetadata( + id="core.packet.packet_capture", + name="IPv4 packet capture", + version="4.0.0", + specialty=Specialty.PACKET, + output_media_types=("text/csv",), + description="Captures bounded IPv4 packet metadata without saving packet payloads.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=( + Capability.NETWORK, + Capability.PACKET_CAPTURE, + Capability.ELEVATED_PRIVILEGES, + ), + privilege_level=PrivilegeLevel.ELEVATED, + network_access=NetworkAccess.LOCAL, + sensitive_data_categories=("network_metadata",), + default_profiles=("deep",), + timeout_seconds=90, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Validate bounded capture settings before raw-socket creation.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + try: + count = context.setting_int("packet_count", 100) + timeout = context.setting_float("timeout_seconds", 10.0) + retry_window = context.setting_float("retry_window_seconds", 0.0) + except (TypeError, ValueError): + return ValidationResult( + False, + reasons=("packet_count, timeout_seconds, and retry_window_seconds must be numeric",), + ) + if not 1 <= count <= 10_000 or not 1 <= timeout <= 60 or not 0 <= retry_window <= 60: + return ValidationResult( + False, + reasons=("packet_count must be 1-10000, timeout_seconds 1-60, and retry_window_seconds 0-60",), + ) + return ValidationResult(True) + + @staticmethod + def _close_capture(capture_socket: socket.socket) -> None: + """Disable Windows promiscuous capture mode and close the socket.""" + try: + capture_socket.ioctl( + socket.SIO_RCVALL, + socket.RCVALL_OFF, + ) + except OSError: + pass + capture_socket.close() + + def collect(self, context: CollectorContext) -> CollectorResult: + """Capture bounded metadata and save it as CSV without retaining payload bytes.""" + if context.is_cancelled: + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled before packet capture", + ) + + count = context.setting_int("packet_count", 100) + timeout = context.setting_float("timeout_seconds", 10.0) + retry_window = context.setting_float("retry_window_seconds", 0.0) + interface = context.setting_str( + "interface", + socket.gethostbyname(socket.gethostname()), + ) + + context.report_progress( + "packet_capture_started", + interface=interface, + packet_count=count, + retry_window_seconds=retry_window, + ) + + output = context.workspace / "packet_capture.csv" + capture: socket.socket | None = None + observation_count = 0 + early_result: CollectorResult | None = None + + with output.open("w", newline="", encoding="utf-8") as stream: + writer = csv.DictWriter( + stream, + fieldnames=( + "source_ip", + "destination_ip", + "protocol", + "source_port", + "destination_port", + "packet_bytes", + ), + ) + writer.writeheader() + + try: + active_capture = socket.socket( + socket.AF_INET, + socket.SOCK_RAW, + socket.IPPROTO_IP, + ) + capture = active_capture + + active_capture.bind((interface, 0)) + active_capture.setsockopt( + socket.IPPROTO_IP, + socket.IP_HDRINCL, + 1, + ) + active_capture.ioctl( + socket.SIO_RCVALL, + socket.RCVALL_ON, + ) + + deadline = time.monotonic() + timeout + retry_deadline = time.monotonic() + retry_window + + while observation_count < count and time.monotonic() < deadline: + if context.is_cancelled: + early_result = CollectorResult( + CollectorStatus.CANCELLED, + "cancelled during packet capture", + ) + break + + remaining = max(0.0, deadline - time.monotonic()) + ready, _, _ = select.select( + [active_capture], + [], + [], + min(1.0, remaining), + ) + + if not ready: + continue + + try: + payload = active_capture.recv(65_535) + except OSError: + if time.monotonic() < retry_deadline: + time.sleep(0.1) + continue + raise + + row = _packet_row(payload) + if row is not None: + writer.writerow(row) + observation_count += 1 + + except PermissionError as error: + early_result = CollectorResult( + CollectorStatus.SKIPPED, + "raw packet capture requires an elevated account", + errors=(str(error),), + ) + + except OSError as error: + if error.winerror in {5, 10013}: + early_result = CollectorResult( + CollectorStatus.SKIPPED, + "raw packet capture was denied for the current account", + errors=(str(error),), + ) + else: + early_result = CollectorResult( + CollectorStatus.FAILED, + "raw packet capture failed", + errors=(str(error),), + ) + + finally: + if capture is not None: + self._close_capture(capture) + + if early_result is not None: + output.unlink(missing_ok=True) + return early_result + + if context.is_cancelled: + output.unlink(missing_ok=True) + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled after packet capture", + ) + + artifact = context.artifacts.register_file( + output, + media_type="text/csv", + ) + + context.report_progress( + "packet_capture_finished", + observation_count=observation_count, + bytes_written=artifact.size_bytes, + ) + + return CollectorResult.succeeded( + "packet metadata captured", + (artifact,), + ) + + def cleanup(self, context: CollectorContext) -> None: + """Raw socket cleanup happens in collect's finally block.""" diff --git a/core/process/detailed_processes.py b/core/process/detailed_processes.py new file mode 100644 index 00000000..65b5e4f2 --- /dev/null +++ b/core/process/detailed_processes.py @@ -0,0 +1,96 @@ +"""Export the verbose Windows task list as bounded CSV process evidence.""" + +from __future__ import annotations + +from subprocess import TimeoutExpired + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from tasklist output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class DetailedProcessesCollector(CoreCollector): + """Capture a read-only verbose tasklist CSV report without changing processes.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated detailed process-report contract.""" + return CollectorMetadata( + id="core.process.detailed_processes", + name="Detailed running processes", + version="4.0.0", + specialty=Specialty.PROCESS, + output_media_types=("text/csv",), + description="Exports the verbose local Windows task list as CSV.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("process_metadata",), + default_profiles=("deep",), + timeout_seconds=45, + maximum_output_bytes=8 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and tasklist availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("tasklist") is None: + return ValidationResult(False, reasons=("tasklist is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run the verbose task list and register its bounded CSV evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before detailed-process collection") + context.report_progress("detailed_processes_started") + try: + completed = subprocess.run( + ["tasklist", "/v", "/fo", "csv", "/nh"], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + except TimeoutExpired as error: + return CollectorResult( + CollectorStatus.SKIPPED, + "detailed tasklist query timed out", + errors=(f"tasklist timed out after {error.timeout} seconds",), + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"tasklist exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "tasklist access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "detailed tasklist query failed", errors=(detail,)) + output = context.workspace / "detailed_processes.csv" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/csv") + process_count = sum(1 for line in completed.stdout.splitlines() if line.strip()) + context.report_progress( + "detailed_processes_finished", + process_count=process_count, + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("detailed process list collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because tasklist exits before the result is returned.""" diff --git a/core/process/memory_map.py b/core/process/memory_map.py new file mode 100644 index 00000000..c9bf8c1d --- /dev/null +++ b/core/process/memory_map.py @@ -0,0 +1,277 @@ +"""Export readable memory-region metadata for the isolated collector process.""" + +from __future__ import annotations + +import json +from pathlib import Path + +from logicytics import ( + CollectorMetadata, + CollectorResult, + CoreCollector, + MemoryBasicInformation, + ProcessMemoryCounters, + Specialty, + ValidationResult, + get_current_process, + get_mapped_file_name, + get_process_memory_info, + pointer_value, + virtual_query, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + + +def _permissions(protection: int) -> str: + """Return a readable Windows page-protection summary.""" + values = { + 0x02: "read", + 0x04: "read_write", + 0x08: "write_copy", + 0x20: "execute_read", + 0x40: "execute_read_write", + 0x80: "execute_write_copy", + } + return values.get(protection & 0xFF, "unreadable") + + +class MemoryMapCollector(CoreCollector): + """Capture bounded virtual-memory metadata from the isolated worker process.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the bounded memory-region JSON artifact contract.""" + return CollectorMetadata( + id="core.process.memory_map", + name="Process memory map", + version="4.0.0", + specialty=Specialty.PROCESS, + output_media_types=("application/json",), + description="Exports readable virtual-memory region addresses, sizes, permissions, paths, and process RSS.", + author="Logicytics", + supported_platforms=("win32",), + sensitive_data_categories=("process_metadata",), + default_profiles=("deep",), + capabilities=(), + timeout_seconds=90, + maximum_output_bytes=64 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Validate cancellation state and bounded region settings.""" + if context.is_cancelled: + return ValidationResult( + False, + reasons=("run cancellation was requested",), + ) + + def setting_int(name: str, default: int) -> int: + """Read one integer setting while rejecting booleans and unsupported objects.""" + value = context.settings.get(name, default) + + if isinstance(value, bool): + raise ValueError(f"{name} must be an integer") + + if isinstance(value, (int, str, bytes, bytearray)): + return int(value) + + raise ValueError(f"{name} must be an integer") + + try: + maximum = setting_int("max_regions", 5_000) + output_limit = setting_int( + "output_limit_bytes", + 64 * 1024 * 1024, + ) + safety_margin = setting_int( + "disk_safety_margin_bytes", + 100 * 1024 * 1024, + ) + except (TypeError, ValueError): + return ValidationResult( + False, + reasons=("memory-map limits must be integers",), + ) + + if not 1 <= maximum <= 100_000 or not 1_024 <= output_limit <= 64 * 1024 * 1024 or safety_margin < 0: + return ValidationResult( + False, + reasons=("invalid memory-map region, output, or safety limits",), + ) + + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query readable local memory regions and write metadata-only JSON evidence.""" + if context.is_cancelled: + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled before memory-map collection", + ) + + maximum = context.setting_int("max_regions", 5_000) + output_limit = context.setting_int( + "output_limit_bytes", + 64 * 1024 * 1024, + ) + safety_margin = context.setting_int( + "disk_safety_margin_bytes", + 100 * 1024 * 1024, + ) + + configured_directory = Path(context.setting_str("dump_directory", "memory_maps")) + output_directory = (context.workspace / configured_directory).resolve() + + try: + output_directory.relative_to(context.workspace.resolve()) + except ValueError: + return CollectorResult( + CollectorStatus.FAILED, + "memory-map dump_directory must stay inside the collector workspace", + ) + + process = get_current_process() + + counters = ProcessMemoryCounters() + if not get_process_memory_info(process, counters): + return CollectorResult( + CollectorStatus.FAILED, + "could not query process memory counters", + ) + + memory = MemoryBasicInformation() + address = 0 + regions: list[dict[str, object]] = [] + + context.report_progress( + "memory_map_started", + max_regions=maximum, + ) + + while len(regions) < maximum: + if context.is_cancelled: + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled during memory-map collection", + ) + + queried = virtual_query(address, memory) + region_size = int(memory.RegionSize) + + if not queried or region_size == 0: + break + + base_address = pointer_value(memory.BaseAddress) + if base_address is None: + base_address = address + + readable = memory.State == 0x1000 and not memory.Protect & 0x101 + + if readable: + mapped_path = get_mapped_file_name( + process, + base_address, + ) + + regions.append( + { + "index": len(regions), + "address": f"0x{base_address:016X}", + "size_bytes": region_size, + "rss_bytes": int(counters.WorkingSetSize), + "permissions": _permissions(int(memory.Protect)), + "mapped_path": mapped_path, + "state": int(memory.State), + "type": int(memory.Type), + } + ) + + next_address = base_address + region_size + + if next_address <= address: + break + + address = next_address + + truncated = False + + while True: + if context.is_cancelled: + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled during memory-map serialization", + ) + + serialized = ( + json.dumps( + { + "region_count": len(regions), + "truncated": truncated, + "regions": regions, + }, + indent=2, + ) + + "\n" + ) + + serialized_size = len(serialized.encode("utf-8")) + + if serialized_size <= output_limit or not regions: + break + + regions.pop() + truncated = True + + if filesystem_adapter.disk_usage(context.workspace).free < serialized_size + safety_margin: + return CollectorResult( + CollectorStatus.SKIPPED, + "insufficient free disk space after configured memory-map safety margin", + ) + + if context.is_cancelled: + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled before memory-map publication", + ) + + output_directory.mkdir( + parents=True, + exist_ok=True, + ) + + output = output_directory / "memory_map.json" + output.write_text( + serialized, + encoding="utf-8", + ) + + if context.is_cancelled: + output.unlink(missing_ok=True) + + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled during memory-map publication", + ) + + artifact = context.artifacts.register_file( + output, + media_type="application/json", + ) + + context.report_progress( + "memory_map_finished", + region_count=len(regions), + truncated=str(truncated).lower(), + bytes_written=artifact.size_bytes, + ) + + summary = "process memory map collected" if not truncated else "process memory map collected with configured output truncation" + + return CollectorResult.succeeded( + summary, + (artifact,), + ) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the collector only queries its own process.""" diff --git a/core/process/process_memory.py b/core/process/process_memory.py new file mode 100644 index 00000000..51586c00 --- /dev/null +++ b/core/process/process_memory.py @@ -0,0 +1,100 @@ +"""Export bounded per-process memory counters as read-only JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class ProcessMemoryCollector(CoreCollector): + """Capture per-process aggregate memory counters without reading memory contents.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated process-memory JSON artifact contract.""" + return CollectorMetadata( + id="core.process.process_memory", + name="Process memory", + version="4.0.0", + specialty=Specialty.PROCESS, + output_media_types=("application/json",), + description="Exports aggregate working-set, private, and virtual memory counters for local processes.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("process_metadata",), + default_profiles=("deep",), + timeout_seconds=60, + maximum_output_bytes=4 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query aggregate process memory counters and register a JSON artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before process-memory collection") + context.report_progress("process_memory_started") + command = ( + "$ErrorActionPreference = 'Continue'; Get-Process | " + "Select-Object Id, ProcessName, WorkingSet64, PrivateMemorySize64, VirtualMemorySize64, HandleCount, CPU, StartTime | " + "ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=55, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "process-memory access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "process-memory query failed", errors=(detail,)) + try: + processes = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "process-memory query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(processes, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "process-memory query returned an unexpected result") + output = context.workspace / "process_memory.json" + output.write_text(json.dumps(processes, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(processes) if isinstance(processes, list) else 1 + context.report_progress("process_memory_finished", process_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("process memory collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/process/running_processes.py b/core/process/running_processes.py new file mode 100644 index 00000000..5b62e08d --- /dev/null +++ b/core/process/running_processes.py @@ -0,0 +1,83 @@ +"""Collect a bounded CSV inventory of running Windows processes.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +class RunningProcessesCollector(CoreCollector): + """Capture the non-verbose Windows Tasklist report through the artifact boundary.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the local subprocess capability and bounded CSV output contract.""" + return CollectorMetadata( + id="core.process.running_processes", + name="Running processes", + version="4.0.0", + specialty=Specialty.PROCESS, + output_media_types=("text/csv",), + description="Exports the non-verbose Windows Tasklist process inventory as CSV.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + default_profiles=("standard", "deep"), + timeout_seconds=20, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Verify that the Windows Tasklist command is available before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("tasklist") is None: + return ValidationResult(False, reasons=("tasklist is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run Tasklist, write its CSV output in the private workspace, and register it.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Tasklist execution") + context.report_progress("tasklist_started") + completed = subprocess.run( + ["tasklist", "/FO", "CSV", "/NH"], + capture_output=True, + check=False, + text=True, + timeout=15, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"tasklist exit code {completed.returncode}" + if "access denied" in detail.casefold() or "access is denied" in detail.casefold(): + return CollectorResult( + CollectorStatus.SKIPPED, + "Tasklist access was denied for the current account", + errors=(detail,), + ) + return CollectorResult( + CollectorStatus.FAILED, + "Tasklist did not complete successfully", + errors=(detail,), + ) + output = context.workspace / "running_processes.csv" + output.write_text(completed.stdout, encoding="utf-8", newline="") + artifact = context.artifacts.register_file(output, media_type="text/csv") + process_count = sum(1 for line in completed.stdout.splitlines() if line.strip()) + context.report_progress("tasklist_finished", processes=process_count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded( + "running-process inventory collected", + (artifact,), + ) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because Tasklist completes before the result is returned.""" diff --git a/core/registry/hklm_backup.py b/core/registry/hklm_backup.py new file mode 100644 index 00000000..be01eb11 --- /dev/null +++ b/core/registry/hklm_backup.py @@ -0,0 +1,95 @@ +"""Export the local HKLM registry hive to a sensitive, read-only .reg artifact.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + EvidenceKind, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize Windows permission-denied or elevation-required wording.""" + normalized = detail.casefold() + return "permission denied" in normalized or ( + "access" in normalized and "denied" in normalized) or "requires elevation" in normalized + + +class HklmBackupCollector(CoreCollector): + """Create a sensitive registry export solely inside the isolated workspace.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the explicit-consent HKLM registry-export artifact contract.""" + return CollectorMetadata( + id="core.registry.hklm_backup", + name="HKLM registry backup", + version="4.0.0", + specialty=Specialty.REGISTRY, + output_media_types=("text/plain",), + description="Exports the local HKLM hive as a .reg backup after explicit sensitive-data approval.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=( + Capability.REGISTRY_READ, + Capability.SUBPROCESS, + Capability.SENSITIVE_FILES, + ), + sensitive_data_categories=("registry", "system_configuration", "credentials"), + default_profiles=("deep",), + timeout_seconds=300, + maximum_output_bytes=2 * 1024 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and the Windows registry tool before export.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("reg") is None: + return ValidationResult(False, reasons=("reg.exe is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Export HKLM to the private workspace and register the resulting artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before HKLM registry backup") + context.report_progress("hklm_backup_started") + output = context.workspace / "hklm_backup.reg" + completed = subprocess.run( + ["reg", "export", "HKLM", str(output), "/y"], + capture_output=True, + check=False, + text=True, + timeout=290, + ) + detail = completed.stderr.strip() or completed.stdout.strip() + if completed.returncode != 0: + message = detail or f"reg.exe exit code {completed.returncode}" + if _is_access_denied(message): + return CollectorResult( + CollectorStatus.SKIPPED, + "HKLM registry export was denied for the current account", + errors=(message,), + ) + return CollectorResult(CollectorStatus.FAILED, "HKLM registry export failed", errors=(message,)) + if not output.is_file() or output.stat().st_size == 0: + return CollectorResult(CollectorStatus.FAILED, "HKLM registry export produced no backup artifact") + artifact = context.artifacts.register_file( + output, + media_type="text/plain", + evidence_kind=EvidenceKind.RAW, + transformations=("exported from the HKLM registry hive",), + ) + context.report_progress("hklm_backup_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("HKLM registry backup collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Leave removal to the isolated collector workspace lifecycle.""" diff --git a/core/registry/installed_applications.py b/core/registry/installed_applications.py new file mode 100644 index 00000000..a0fd392b --- /dev/null +++ b/core/registry/installed_applications.py @@ -0,0 +1,136 @@ +"""Export installed Windows application metadata from read-only uninstall registry keys.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import registry_adapter as winreg + +_UNINSTALL_PATHS = ( + r"SOFTWARE\Microsoft\Windows\CurrentVersion\Uninstall", + r"SOFTWARE\WOW6432Node\Microsoft\Windows\CurrentVersion\Uninstall", +) +_MAX_APPLICATIONS = 5_000 + + +def _registry_value(key: winreg.HKEYType, name: str) -> str | None: + """Return a string registry value when present without exposing registry errors.""" + try: + value, _ = winreg.QueryValueEx(key, name) + except OSError: + return None + return str(value) if value is not None else None + + +def _installed_applications() -> list[dict[str, str | None]]: + """Enumerate uninstall metadata from 64-bit and WOW6432Node registry views.""" + applications: list[dict[str, str | None]] = [] + seen: set[tuple[str, str]] = set() + for path in _UNINSTALL_PATHS: + try: + root = winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, path) + except OSError: + continue + with root: + index = 0 + while len(applications) < _MAX_APPLICATIONS: + try: + key_name = winreg.EnumKey(root, index) + except OSError: + break + index += 1 + try: + with winreg.OpenKey(root, key_name) as key: + display_name = _registry_value(key, "DisplayName") + if not display_name: + continue + version = _registry_value(key, "DisplayVersion") or "" + identity = (display_name, version) + if identity in seen: + continue + seen.add(identity) + applications.append( + { + "display_name": display_name, + "display_version": version or None, + "publisher": _registry_value(key, "Publisher"), + "install_date": _registry_value(key, "InstallDate"), + "install_location": _registry_value(key, "InstallLocation"), + "uninstall_key": key_name, + } + ) + except OSError: + continue + return sorted(applications, key=lambda item: (item["display_name"] or "").casefold()) + + +class InstalledApplicationsCollector(CoreCollector): + """Capture read-only installed-application metadata from Windows registry views.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the registry-read, bounded installed-application artifact contract.""" + return CollectorMetadata( + id="core.registry.installed_applications", + name="Installed applications", + version="4.0.0", + specialty=Specialty.REGISTRY, + output_media_types=("application/json",), + description="Exports installed application names, versions, publishers, and install metadata from uninstall keys.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.REGISTRY_READ,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=60, + maximum_output_bytes=4 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation and whether at least one uninstall registry view is readable.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + for path in _UNINSTALL_PATHS: + try: + with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, path): + return ValidationResult(True) + except OSError: + continue + return ValidationResult(False, reasons=("Windows uninstall registry keys are unavailable",)) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Read installed application metadata and register the bounded JSON artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before installed-application collection") + context.report_progress("installed_applications_started") + applications = _installed_applications() + output = context.workspace / "installed_applications.json" + output.write_text( + json.dumps( + {"collected_at": datetime.now(UTC).isoformat(), "applications": applications}, + indent=2, + sort_keys=True, + ) + + "\n", + encoding="utf-8", + ) + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "installed_applications_finished", + application_count=len(applications), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("installed applications collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because all registry handles are scoped and closed.""" diff --git a/core/registry/startup_applications.py b/core/registry/startup_applications.py new file mode 100644 index 00000000..31d994c6 --- /dev/null +++ b/core/registry/startup_applications.py @@ -0,0 +1,116 @@ +"""Export standard Windows startup Run-key entries as bounded read-only JSON evidence.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import registry_adapter as winreg + +_RUN_PATHS = ( + ( + "HKEY_CURRENT_USER", + winreg.HKEY_CURRENT_USER, + r"Software\Microsoft\Windows\CurrentVersion\Run", + ), + ( + "HKEY_CURRENT_USER", + winreg.HKEY_CURRENT_USER, + r"Software\Microsoft\Windows\CurrentVersion\RunOnce", + ), + ( + "HKEY_LOCAL_MACHINE", + winreg.HKEY_LOCAL_MACHINE, + r"Software\Microsoft\Windows\CurrentVersion\Run", + ), + ( + "HKEY_LOCAL_MACHINE", + winreg.HKEY_LOCAL_MACHINE, + r"Software\Microsoft\Windows\CurrentVersion\RunOnce", + ), +) + + +def _startup_entries() -> list[dict[str, str]]: + """Enumerate values from conventional per-user and machine startup Run keys.""" + entries: list[dict[str, str]] = [] + for hive_name, hive, path in _RUN_PATHS: + try: + key = winreg.OpenKey(hive, path) + except OSError: + continue + with key: + index = 0 + while True: + try: + name, value, _ = winreg.EnumValue(key, index) + except OSError: + break + index += 1 + entries.append({"hive": hive_name, "path": path, "name": name, "command": str(value)}) + return sorted(entries, key=lambda item: (item["hive"], item["path"], item["name"].casefold())) + + +class StartupApplicationsCollector(CoreCollector): + """Capture startup Run-key commands without launching or modifying any application.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the registry-read startup-application artifact contract.""" + return CollectorMetadata( + id="core.registry.startup_applications", + name="Startup applications", + version="4.0.0", + specialty=Specialty.REGISTRY, + output_media_types=("application/json",), + description="Exports standard user and machine Run/RunOnce startup registry entries.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.REGISTRY_READ,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=30, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation before reading potentially available startup key paths.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Read startup metadata and register a bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before startup-application collection") + context.report_progress("startup_applications_started") + entries = _startup_entries() + output = context.workspace / "startup_applications.json" + output.write_text( + json.dumps( + {"collected_at": datetime.now(UTC).isoformat(), "entries": entries}, + indent=2, + sort_keys=True, + ) + + "\n", + encoding="utf-8", + ) + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "startup_applications_finished", + entry_count=len(entries), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("startup applications collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because all registry handles are scoped and closed.""" diff --git a/core/ssh/ssh_backup.py b/core/ssh/ssh_backup.py new file mode 100644 index 00000000..67b75316 --- /dev/null +++ b/core/ssh/ssh_backup.py @@ -0,0 +1,179 @@ +"""Archive the current user's SSH directory after explicit sensitive approval.""" + +from __future__ import annotations + +import zipfile +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + EvidenceKind, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +MAX_FILE_BYTES = 10 * 1024 * 1024 +MAX_ARCHIVE_SOURCE_BYTES = 128 * 1024 * 1024 + + +class SshBackupCollector(CoreCollector): + """Back up current-user SSH keys and configuration into the private workspace.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the explicit-consent SSH archive artifact contract.""" + return CollectorMetadata( + id="core.ssh.ssh_backup", + name="SSH directory backup", + version="4.0.0", + specialty=Specialty.SSH, + output_media_types=("application/zip",), + description="Archives the current user's .ssh keys and configuration after explicit sensitive-data approval.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=( + Capability.FILESYSTEM_READ, + Capability.SENSITIVE_FILES, + Capability.PRIVATE_KEYS, + ), + sensitive_data_categories=("private_keys", "ssh_configuration", "credentials"), + default_profiles=("deep",), + timeout_seconds=120, + maximum_output_bytes=128 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state before scanning the current user's SSH directory.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Archive bounded regular files from .ssh while preserving relative paths.""" + if context.is_cancelled: + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled before SSH backup", + ) + + ssh_directory = filesystem_adapter.home() / ".ssh" + + try: + directory_exists = ssh_directory.is_dir() + except OSError as error: + return CollectorResult( + CollectorStatus.SKIPPED, + "the current user's .ssh directory is inaccessible", + errors=(str(error),), + ) + + if not directory_exists: + return CollectorResult( + CollectorStatus.SKIPPED, + "the current user has no .ssh directory", + ) + + context.report_progress("ssh_backup_started") + + archive = context.workspace / "ssh_backup.zip" + source_bytes = 0 + archived_files = 0 + skipped_files = 0 + cancelled = False + + with zipfile.ZipFile( + archive, + "w", + compression=zipfile.ZIP_DEFLATED, + ) as output: + try: + candidates: list[Path] = sorted( + (Path(path) for path in filesystem_adapter.recursive(ssh_directory)), + key=lambda path: path.as_posix(), + ) + except OSError as error: + return CollectorResult( + CollectorStatus.SKIPPED, + "the current user's .ssh directory is inaccessible", + errors=(str(error),), + ) + + for candidate in candidates: + if context.is_cancelled: + cancelled = True + break + + try: + is_file = candidate.is_file() + is_symlink = candidate.is_symlink() + size = candidate.stat().st_size if is_file else 0 + except OSError: + skipped_files += 1 + continue + + if not is_file or is_symlink: + continue + + if size > MAX_FILE_BYTES or source_bytes + size > MAX_ARCHIVE_SOURCE_BYTES: + skipped_files += 1 + continue + + output.write( + candidate, + arcname=candidate.relative_to(ssh_directory).as_posix(), + ) + + if context.is_cancelled: + cancelled = True + break + + source_bytes += size + archived_files += 1 + + if cancelled: + archive.unlink(missing_ok=True) + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled during SSH backup", + ) + + if archived_files == 0: + return CollectorResult( + CollectorStatus.SKIPPED, + "no SSH files met the bounded backup policy", + ) + + if context.is_cancelled: + archive.unlink(missing_ok=True) + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled before SSH-backup registration", + ) + + artifact = context.artifacts.register_file( + archive, + media_type="application/zip", + evidence_kind=EvidenceKind.RAW, + transformations=("archived from the current user's SSH directory",), + ) + + context.report_progress( + "ssh_backup_finished", + archived_files=archived_files, + skipped_files=skipped_files, + source_bytes=source_bytes, + bytes_written=artifact.size_bytes, + ) + + return CollectorResult.succeeded( + "SSH directory backup collected", + (artifact,), + ) + + def cleanup(self, context: CollectorContext) -> None: + """Leave archive removal to the isolated collector workspace lifecycle.""" diff --git a/core/storage/logical_drives.py b/core/storage/logical_drives.py new file mode 100644 index 00000000..a0314121 --- /dev/null +++ b/core/storage/logical_drives.py @@ -0,0 +1,120 @@ +"""Collect bounded Windows logical-drive capacity and type metadata.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime + +from logicytics import ( + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, + get_disk_free_space, + get_drive_type, + get_logical_drives, + ularge_integer, +) +from logicytics.contracts import CollectorContext, CollectorStatus + +_DRIVE_TYPES = { + 0: "unknown", + 1: "no_root_directory", + 2: "removable", + 3: "fixed", + 4: "remote", + 5: "optical", + 6: "ram_disk", +} + + +def _logical_drives() -> list[dict[str, int | str]]: + """Return Windows logical drive metadata without walking any filesystem contents.""" + mask = get_logical_drives() + drives: list[dict[str, int | str]] = [] + + for offset in range(26): + if not mask & (1 << offset): + continue + + root = f"{chr(ord('A') + offset)}:\\" + available = ularge_integer() + total = ularge_integer() + free = ularge_integer() + + if not get_disk_free_space( + root, + available, + total, + free, + ): + continue + + drive_type_code = get_drive_type(root) + + drives.append( + { + "root": root, + "type": _DRIVE_TYPES.get(drive_type_code, "unknown"), + "total_bytes": int(total.value), + "free_bytes": int(free.value), + "available_bytes": int(available.value), + } + ) + + return drives + + +class LogicalDrivesCollector(CoreCollector): + """Write a JSON inventory of mounted logical-drive capacity and type metadata.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare a bounded, capability-free storage metadata collector.""" + return CollectorMetadata( + id="core.storage.logical_drives", + name="Logical drives", + version="4.0.0", + specialty=Specialty.STORAGE, + output_media_types=("application/json",), + description="Records mounted logical-drive types and aggregate capacity metadata.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(), + default_profiles=("minimal", "standard", "deep", "offline"), + timeout_seconds=10, + maximum_output_bytes=64 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation and ensure the Windows logical-drive API is available.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + try: + _logical_drives() + except OSError as error: + return ValidationResult(False, reasons=(f"logical-drive API is unavailable: {error}",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Capture logical-drive metadata and register a bounded JSON report.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before storage collection") + context.report_progress("logical_drives_started") + try: + drives = _logical_drives() + except OSError as error: + return CollectorResult(CollectorStatus.FAILED, "could not read logical drives", errors=(str(error),)) + report = { + "collected_at": datetime.now(UTC).isoformat(), + "drives": drives, + } + output = context.workspace / "logical_drives.json" + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("logical_drives_finished", drive_count=len(drives), bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("logical-drive metadata collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because the Windows storage API is synchronous.""" diff --git a/core/storage/mounted_volumes.py b/core/storage/mounted_volumes.py new file mode 100644 index 00000000..e6f6dfc4 --- /dev/null +++ b/core/storage/mounted_volumes.py @@ -0,0 +1,78 @@ +"""Export mounted Windows volume GUID mappings as bounded, read-only text evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from mountvol output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class MountedVolumesCollector(CoreCollector): + """Capture mounted volume GUID mappings without changing any mount points.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated mounted-volume text artifact contract.""" + return CollectorMetadata( + id="core.storage.mounted_volumes", + name="Mounted volumes", + version="4.0.0", + specialty=Specialty.STORAGE, + output_media_types=("text/plain",), + description="Exports Windows mounted volume GUID and mount-point mappings from mountvol.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=30, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and mountvol availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("mountvol") is None: + return ValidationResult(False, reasons=("mountvol is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run read-only mountvol and register its bounded text evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before mounted-volume collection") + context.report_progress("mounted_volumes_started") + completed = subprocess.run(["mountvol"], capture_output=True, check=False, text=True, timeout=25) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"mountvol exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "mounted-volume access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "mounted-volume query failed", errors=(detail,)) + output = context.workspace / "mounted_volumes.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + volume_count = sum(1 for line in completed.stdout.splitlines() if line.strip().startswith("\\\\?\\Volume{")) + context.report_progress("mounted_volumes_finished", volume_count=volume_count, + bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("mounted-volume mappings collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because mountvol exits before the result is returned.""" diff --git a/core/storage/physical_disks.py b/core/storage/physical_disks.py new file mode 100644 index 00000000..30a8153d --- /dev/null +++ b/core/storage/physical_disks.py @@ -0,0 +1,100 @@ +"""Export local Windows physical disk model and capacity CIM data as bounded JSON.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class PhysicalDisksCollector(CoreCollector): + """Capture a read-only inventory of locally attached physical disk hardware.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated physical-disk JSON artifact contract.""" + return CollectorMetadata( + id="core.storage.physical_disks", + name="Physical disks", + version="4.0.0", + specialty=Specialty.STORAGE, + output_media_types=("application/json",), + description="Exports local physical disk models, media types, interface types, and sizes.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query physical-disk CIM data and register a JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before physical-disk collection") + context.report_progress("physical_disks_started") + command = ( + "Get-CimInstance -ClassName Win32_DiskDrive | " + "Select-Object Model, Size, MediaType, InterfaceType, DeviceID, Partitions | " + "ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "physical-disk CIM access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "physical-disk CIM query failed", errors=(detail,)) + try: + physical_disks = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "physical-disk CIM query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(physical_disks, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "physical-disk CIM query returned an unexpected result") + output = context.workspace / "physical_disks.json" + output.write_text(json.dumps(physical_disks, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(physical_disks) if isinstance(physical_disks, list) else 1 + context.report_progress("physical_disks_finished", disk_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("physical-disk inventory collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/storage/volume_details.py b/core/storage/volume_details.py new file mode 100644 index 00000000..a9c96439 --- /dev/null +++ b/core/storage/volume_details.py @@ -0,0 +1,145 @@ +"""Export detailed mounted Windows volume metadata through bounded native API calls.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime + +from logicytics import ( + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, + create_unicode_buffer, + dword, + get_disk_free_space, + get_volume_information, + ularge_integer, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import windows_api_adapter + +_DRIVE_TYPES = { + 0: "unknown", + 1: "no_root_directory", + 2: "removable", + 3: "fixed", + 4: "remote", + 5: "optical", + 6: "ram_disk", +} + + +def _volume_details() -> list[dict[str, int | str]]: + """Read mounted logical-volume capacity, label, and filesystem metadata only.""" + kernel32 = windows_api_adapter.load_library("kernel32") + mask = kernel32.GetLogicalDrives() + if mask == 0: + raise OSError("GetLogicalDrives failed") + volumes: list[dict[str, int | str]] = [] + for offset in range(26): + if not mask & (1 << offset): + continue + root = f"{chr(ord('A') + offset)}:\\" + + available = ularge_integer() + total = ularge_integer() + free = ularge_integer() + + if not get_disk_free_space( + root, + available, + total, + free, + ): + continue + + label = create_unicode_buffer(261) + filesystem = create_unicode_buffer(261) + serial = dword() + maximum_component_length = dword() + flags = dword() + + information_available = get_volume_information( + root, + label, + serial, + maximum_component_length, + flags, + filesystem, + ) + volumes.append( + { + "root": root, + "drive_type": _DRIVE_TYPES.get(kernel32.GetDriveTypeW(root), "unknown"), + "filesystem": filesystem.value if information_available else "unavailable", + "volume_name": label.value if information_available else "unavailable", + "total_bytes": total.value, + "free_bytes": free.value, + "available_bytes": available.value, + } + ) + return volumes + + +class VolumeDetailsCollector(CoreCollector): + """Capture detailed mounted logical-volume metadata without walking file contents.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the capability-free detailed volume artifact contract.""" + return CollectorMetadata( + id="core.storage.volume_details", + name="Volume details", + version="4.0.0", + specialty=Specialty.STORAGE, + output_media_types=("application/json",), + description="Exports mounted drive type, filesystem, label, and capacity metadata.", + author="Logicytics", + supported_platforms=("win32",), + default_profiles=("standard", "deep"), + timeout_seconds=15, + capabilities=(), + maximum_output_bytes=128 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation and native volume API availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + try: + _volume_details() + except OSError as error: + return ValidationResult(False, reasons=(f"volume API is unavailable: {error}",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Read detailed volume metadata and register a JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before volume-details collection") + context.report_progress("volume_details_started") + try: + volumes = _volume_details() + except OSError as error: + return CollectorResult( + CollectorStatus.FAILED, + "could not read detailed volume metadata", + errors=(str(error),), + ) + output = context.workspace / "volume_details.json" + output.write_text( + json.dumps( + {"collected_at": datetime.now(UTC).isoformat(), "volumes": volumes}, + indent=2, + sort_keys=True, + ) + + "\n", + encoding="utf-8", + ) + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("volume_details_finished", volume_count=len(volumes), bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("detailed volume metadata collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because Windows volume API calls are synchronous.""" diff --git a/core/system/bios_info.py b/core/system/bios_info.py new file mode 100644 index 00000000..85dc7fee --- /dev/null +++ b/core/system/bios_info.py @@ -0,0 +1,126 @@ +"""Export Windows BIOS manufacturer, name, and version as a bounded HTML table.""" + +from __future__ import annotations + +import html +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +def render_bios_table(bios: dict[str, str | None]) -> str: + """Render trusted structured BIOS fields into a portable evidence table.""" + rows = "\n".join( + f" {html.escape(label)}{html.escape(str(bios.get(key) or 'unavailable'))}" + for key, label in ( + ("Manufacturer", "Manufacturer"), + ("Name", "Name"), + ("SMBIOSBIOSVersion", "SMBIOS BIOS version"), + ("Version", "Version"), + ("ReleaseDate", "Release date"), + ) + ) + return "\n".join( + ( + "", + '', + ' BIOS information', + " ", + "

BIOS information

", + " ", + " ", + f" \n{rows}\n ", + "
FieldValue
", + " ", + "", + "", + ) + ) + + +class BiosInfoCollector(CoreCollector): + """Capture a read-only BIOS inventory through modern Windows CIM data.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated BIOS HTML artifact contract.""" + return CollectorMetadata( + id="core.system.bios_info", + name="BIOS information", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("text/html",), + description="Exports BIOS manufacturer, name, version, and release date as HTML.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=128 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query BIOS CIM data and register an HTML evidence table.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before BIOS collection") + context.report_progress("bios_info_started") + command = ( + "Get-CimInstance -ClassName Win32_BIOS | " + "Select-Object Manufacturer, Name, SMBIOSBIOSVersion, Version, ReleaseDate | " + "ConvertTo-Json -Compress" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "BIOS CIM access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "BIOS CIM query failed", errors=(detail,)) + try: + bios = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult(CollectorStatus.FAILED, "BIOS CIM query returned invalid JSON", errors=(str(error),)) + if not isinstance(bios, dict): + return CollectorResult(CollectorStatus.FAILED, "BIOS CIM query returned an unexpected result") + output = context.workspace / "bios_info.html" + output.write_text(render_bios_table(bios), encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/html") + context.report_progress("bios_info_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("BIOS information collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/computer_system.py b/core/system/computer_system.py new file mode 100644 index 00000000..dfb08a21 --- /dev/null +++ b/core/system/computer_system.py @@ -0,0 +1,99 @@ +"""Export local Windows computer model, manufacturer, and processor-count CIM data.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class ComputerSystemCollector(CoreCollector): + """Capture read-only local computer-system identity and hardware metadata.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated computer-system JSON artifact contract.""" + return CollectorMetadata( + id="core.system.computer_system", + name="Computer system", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports local computer model, manufacturer, and processor-count CIM data.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=128 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query computer-system CIM data and register a JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before computer-system collection") + context.report_progress("computer_system_started") + command = ( + "Get-CimInstance -ClassName Win32_ComputerSystem | " + "Select-Object Name, Manufacturer, Model, SystemType, NumberOfProcessors, " + "NumberOfLogicalProcessors, TotalPhysicalMemory | ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "computer-system CIM access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "computer-system CIM query failed", errors=(detail,)) + try: + computer_system = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "computer-system CIM query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(computer_system, dict): + return CollectorResult(CollectorStatus.FAILED, "computer-system CIM query returned an unexpected result") + output = context.workspace / "computer_system.json" + output.write_text(json.dumps(computer_system, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("computer_system_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("computer-system details collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/defender_status.py b/core/system/defender_status.py new file mode 100644 index 00000000..f8a4b659 --- /dev/null +++ b/core/system/defender_status.py @@ -0,0 +1,105 @@ +"""Export local Microsoft Defender protection status as read-only JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_unavailable(detail: str) -> bool: + """Recognize missing Defender-provider and access-denied results without failing a run.""" + normalized = detail.casefold() + return ( + "permission denied" in normalized + or ("access" in normalized and "denied" in normalized) + or "not recognized" in normalized + or "cannot find" in normalized + ) + + +class DefenderStatusCollector(CoreCollector): + """Capture read-only Microsoft Defender configuration and protection-state metadata.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated Defender-status JSON artifact contract.""" + return CollectorMetadata( + id="core.system.defender_status", + name="Microsoft Defender status", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports installed Microsoft Defender engine, signature, and protection-state metadata.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("security_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=30, + maximum_output_bytes=128 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query Defender status and register its bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Defender-status collection") + context.report_progress("defender_status_started") + command = ( + "Get-MpComputerStatus | Select-Object AMEngineVersion, AMProductVersion, AMServiceEnabled, " + "AntispywareEnabled, AntivirusEnabled, BehaviorMonitorEnabled, IoavProtectionEnabled, " + "NISEnabled, RealTimeProtectionEnabled, AntivirusSignatureVersion, AntivirusSignatureLastUpdated, " + "QuickScanAge, FullScanAge | ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_unavailable(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "Microsoft Defender status is unavailable", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "Microsoft Defender status query failed", errors=(detail,)) + try: + status = json.loads(completed.stdout) if completed.stdout.strip() else {} + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "Defender status query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(status, dict): + return CollectorResult(CollectorStatus.FAILED, "Defender status query returned an unexpected result") + output = context.workspace / "defender_status.json" + output.write_text(json.dumps(status, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("defender_status_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("Microsoft Defender status collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/environment_posture.py b/core/system/environment_posture.py new file mode 100644 index 00000000..07989c2a --- /dev/null +++ b/core/system/environment_posture.py @@ -0,0 +1,107 @@ +"""Export local privilege, UAC, and PowerShell-policy posture as read-only JSON.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common Windows permission-denied wording.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class EnvironmentPostureCollector(CoreCollector): + """Capture local execution and elevation posture without changing system state.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the bounded, subprocess-gated system posture artifact contract.""" + return CollectorMetadata( + id="core.system.environment_posture", + name="Environment posture", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports administrator state, UAC settings, and PowerShell execution policies.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query local posture and register a bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before environment-posture collection") + context.report_progress("environment_posture_started") + command = ( + "$ErrorActionPreference = 'Stop'; " + "$identity = [Security.Principal.WindowsIdentity]::GetCurrent(); " + "$principal = [Security.Principal.WindowsPrincipal]::new($identity); " + "$uac = Get-ItemProperty -Path 'HKLM:\\SOFTWARE\\Microsoft\\Windows\\CurrentVersion\\Policies\\System' " + "-Name EnableLUA, ConsentPromptBehaviorAdmin, PromptOnSecureDesktop; " + "[pscustomobject]@{ IsAdministrator = $principal.IsInRole([Security.Principal.WindowsBuiltInRole]::Administrator); " + "UacEnabled = [bool]$uac.EnableLUA; ConsentPromptBehaviorAdmin = $uac.ConsentPromptBehaviorAdmin; " + "PromptOnSecureDesktop = $uac.PromptOnSecureDesktop; " + "ExecutionPolicies = @(try { Get-ExecutionPolicy -List | Select-Object Scope, ExecutionPolicy } " + "catch { [pscustomobject]@{ Error = $_.Exception.Message } }); " + "CollectedAt = [DateTime]::UtcNow.ToString('o') } | ConvertTo-Json -Depth 4" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "environment-posture access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "environment-posture query failed", errors=(detail,)) + try: + posture = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "environment-posture query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(posture, dict): + return CollectorResult(CollectorStatus.FAILED, "environment-posture query returned an unexpected result") + output = context.workspace / "environment_posture.json" + output.write_text(json.dumps(posture, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("environment_posture_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("environment posture collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/group_policy.py b/core/system/group_policy.py new file mode 100644 index 00000000..45ebe441 --- /dev/null +++ b/core/system/group_policy.py @@ -0,0 +1,82 @@ +"""Collect the local Windows group-policy result report as bounded text evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from Windows command output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class GroupPolicyCollector(CoreCollector): + """Capture the local user/computer group-policy summary without changing policy state.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the read-only subprocess capability and bounded text output contract.""" + return CollectorMetadata( + id="core.system.group_policy", + name="Group policy result", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("text/plain",), + description="Exports the local Windows Group Policy Result summary through gpresult.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and the presence of gpresult before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("gpresult") is None: + return ValidationResult(False, reasons=("gpresult is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run the read-only group-policy summary and register its text output.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before group-policy collection") + context.report_progress("group_policy_started") + completed = subprocess.run( + ["gpresult", "/r"], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"gpresult exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "group-policy access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "gpresult failed", errors=(detail,)) + output = context.workspace / "group_policy.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + line_count = sum(1 for line in completed.stdout.splitlines() if line.strip()) + context.report_progress("group_policy_finished", line_count=line_count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("group-policy summary collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because gpresult exits before the result is returned.""" diff --git a/core/system/installed_drivers.py b/core/system/installed_drivers.py new file mode 100644 index 00000000..abe6d2ea --- /dev/null +++ b/core/system/installed_drivers.py @@ -0,0 +1,86 @@ +"""Collect a bounded CSV inventory of installed Windows device drivers.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common access-denied wording from Windows command output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class InstalledDriversCollector(CoreCollector): + """Capture a read-only driverquery CSV report through the artifact boundary.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the read-only subprocess capability and bounded driver-report output.""" + return CollectorMetadata( + id="core.system.installed_drivers", + name="Installed drivers", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("text/csv",), + description="Exports detailed Windows driver inventory information as CSV.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + default_profiles=("standard", "deep"), + timeout_seconds=30, + maximum_output_bytes=4 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Verify cancellation state and driverquery availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("driverquery") is None: + return ValidationResult(False, reasons=("driverquery is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run driverquery and register its detailed CSV report when permitted.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before driver inventory collection") + context.report_progress("installed_drivers_started") + completed = subprocess.run( + ["driverquery", "/v", "/fo", "csv", "/nh"], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"driverquery exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "driverquery access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "driverquery failed", errors=(detail,)) + output = context.workspace / "installed_drivers.csv" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/csv") + driver_count = sum(1 for line in completed.stdout.splitlines() if line.strip()) + context.report_progress( + "installed_drivers_finished", + driver_count=driver_count, + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("installed-driver inventory collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because driverquery has completed before return.""" diff --git a/core/system/installed_updates.py b/core/system/installed_updates.py new file mode 100644 index 00000000..ae2fecd9 --- /dev/null +++ b/core/system/installed_updates.py @@ -0,0 +1,96 @@ +"""Export installed Windows hotfix metadata as bounded, read-only JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class InstalledUpdatesCollector(CoreCollector): + """Capture read-only local Windows hotfix metadata through the update provider.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated installed-updates JSON artifact contract.""" + return CollectorMetadata( + id="core.system.installed_updates", + name="Installed Windows updates", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports local installed hotfix identifiers, descriptions, and install dates.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query installed hotfixes and register their bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before installed-update collection") + context.report_progress("installed_updates_started") + command = "Get-HotFix | Select-Object HotFixID, Description, InstalledBy, InstalledOn | ConvertTo-Json -Depth 3" + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "installed-update access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "installed-update query failed", errors=(detail,)) + try: + updates = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "installed-update query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(updates, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "installed-update query returned an unexpected result") + output = context.workspace / "installed_updates.json" + output.write_text(json.dumps(updates, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(updates) if isinstance(updates, list) else 1 + context.report_progress("installed_updates_finished", update_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("installed Windows updates collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/local_accounts.py b/core/system/local_accounts.py new file mode 100644 index 00000000..c1ac811e --- /dev/null +++ b/core/system/local_accounts.py @@ -0,0 +1,95 @@ +"""Export local Windows account metadata without accessing credentials or secrets.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common local-account permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class LocalAccountsCollector(CoreCollector): + """Capture read-only local-account names, enabled state, and timestamps through PowerShell.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated local-account JSON artifact contract.""" + return CollectorMetadata( + id="core.system.local_accounts", + name="Local accounts", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports local account names, SIDs, enabled state, descriptions, and login timestamps only.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("account_metadata",), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query local account metadata and register a bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before local-account collection") + context.report_progress("local_accounts_started") + command = ( + "Get-LocalUser | Select-Object Name, Enabled, Description, SID, LastLogon, PasswordLastSet, " + "UserMayChangePassword | ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult(CollectorStatus.SKIPPED, "local-account access was denied", errors=(detail,)) + return CollectorResult(CollectorStatus.FAILED, "local-account query failed", errors=(detail,)) + try: + accounts = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "local-account query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(accounts, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "local-account query returned an unexpected result") + output = context.workspace / "local_accounts.json" + output.write_text(json.dumps(accounts, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(accounts) if isinstance(accounts, list) else 1 + context.report_progress("local_accounts_finished", account_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("local accounts collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/operating_system.py b/core/system/operating_system.py new file mode 100644 index 00000000..31a6dc41 --- /dev/null +++ b/core/system/operating_system.py @@ -0,0 +1,100 @@ +"""Export detailed Windows operating-system CIM data as a bounded JSON artifact.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class OperatingSystemCollector(CoreCollector): + """Capture read-only Windows operating-system details through CIM.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated operating-system JSON artifact contract.""" + return CollectorMetadata( + id="core.system.operating_system", + name="Operating system details", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports detailed local Windows operating-system CIM information.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=128 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query operating-system CIM data and register a JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before operating-system collection") + context.report_progress("operating_system_started") + command = ( + "Get-CimInstance -ClassName Win32_OperatingSystem | " + "Select-Object Caption, CSDVersion, Version, BuildNumber, OSArchitecture, " + "InstallDate, LastBootUpTime, Locale, MUILanguages, SystemDirectory, WindowsDirectory | " + "ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "operating-system CIM access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "operating-system CIM query failed", errors=(detail,)) + try: + operating_system = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "operating-system CIM query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(operating_system, dict): + return CollectorResult(CollectorStatus.FAILED, "operating-system CIM query returned an unexpected result") + output = context.workspace / "operating_system.json" + output.write_text(json.dumps(operating_system, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("operating_system_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("operating-system details collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/scheduled_tasks.py b/core/system/scheduled_tasks.py new file mode 100644 index 00000000..95b5e35e --- /dev/null +++ b/core/system/scheduled_tasks.py @@ -0,0 +1,98 @@ +"""Export bounded local Scheduled Task metadata as read-only JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class ScheduledTasksCollector(CoreCollector): + """Capture read-only local scheduled-task identities and current states.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated scheduled-task JSON artifact contract.""" + return CollectorMetadata( + id="core.system.scheduled_tasks", + name="Scheduled tasks", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports up to 1,000 local scheduled-task names, paths, authors, descriptions, and states.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("deep",), + timeout_seconds=60, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query bounded scheduled-task metadata and register a JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before scheduled-task collection") + context.report_progress("scheduled_tasks_started") + command = ( + "Get-ScheduledTask | Select-Object -First 1000 TaskName, TaskPath, State, Author, Description, URI | ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=55, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "scheduled-task access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "scheduled-task query failed", errors=(detail,)) + try: + tasks = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "scheduled-task query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(tasks, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "scheduled-task query returned an unexpected result") + output = context.workspace / "scheduled_tasks.json" + output.write_text(json.dumps(tasks, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(tasks) if isinstance(tasks, list) else 1 + context.report_progress("scheduled_tasks_finished", task_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("scheduled tasks collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/session_snapshot.py b/core/system/session_snapshot.py new file mode 100644 index 00000000..c3b3803c --- /dev/null +++ b/core/system/session_snapshot.py @@ -0,0 +1,108 @@ +"""Export a bounded Windows session and system-state snapshot as JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class SessionSnapshotCollector(CoreCollector): + """Capture read-only host, user, memory, locale, and system-drive metadata.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated session snapshot artifact contract.""" + return CollectorMetadata( + id="core.system.session_snapshot", + name="Session snapshot", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports current user/SID, OS build, memory, language, host, time, and system-drive data.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("user_identity", "system_configuration"), + default_profiles=("deep",), + timeout_seconds=45, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query bounded session fields and register their JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before session-snapshot collection") + context.report_progress("session_snapshot_started") + command = ( + "$os = Get-CimInstance -ClassName Win32_OperatingSystem; " + "$identity = whoami /user /fo csv /nh | ConvertFrom-Csv; " + "[pscustomobject]@{ " + "ComputerName = $env:COMPUTERNAME; CurrentUser = $identity.'User Name'; Sid = $identity.SID; " + "WindowsBuild = $os.BuildNumber; Version = $os.Version; " + "PhysicalMemoryBytes = $os.TotalVisibleMemorySize * 1KB; " + "VirtualMemoryBytes = $os.TotalVirtualMemorySize * 1KB; " + "FreePhysicalMemoryBytes = $os.FreePhysicalMemory * 1KB; " + "LanguageIds = [System.Globalization.CultureInfo]::CurrentCulture.LCID; " + "UiLanguage = [System.Globalization.CultureInfo]::CurrentUICulture.Name; " + "CollectedAt = [DateTime]::UtcNow.ToString('o'); SystemDrive = $env:SystemDrive " + "} | ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "session-snapshot access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "session-snapshot query failed", errors=(detail,)) + try: + snapshot = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "session-snapshot query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(snapshot, dict): + return CollectorResult(CollectorStatus.FAILED, "session-snapshot query returned an unexpected result") + output = context.workspace / "session_snapshot.json" + output.write_text(json.dumps(snapshot, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("session_snapshot_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("session snapshot collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/system_details.py b/core/system/system_details.py new file mode 100644 index 00000000..e7526d13 --- /dev/null +++ b/core/system/system_details.py @@ -0,0 +1,83 @@ +"""Export the complete read-only Windows systeminfo report as an evidence artifact.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from Windows command output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class SystemDetailsCollector(CoreCollector): + """Capture the full systeminfo report without changing local system state.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated system-details artifact contract.""" + return CollectorMetadata( + id="core.system.system_details", + name="System details", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("text/plain",), + description="Exports the complete Windows systeminfo report as text.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and systeminfo availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("systeminfo") is None: + return ValidationResult(False, reasons=("systeminfo is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run systeminfo and register its complete text output when permitted.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before system-details collection") + context.report_progress("system_details_started") + completed = subprocess.run( + ["systeminfo"], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"systeminfo exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "systeminfo access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "systeminfo failed", errors=(detail,)) + output = context.workspace / "system_details.txt" + output.write_text(completed.stdout, encoding="utf-8", newline="") + artifact = context.artifacts.register_file(output, media_type="text/plain") + line_count = sum(1 for line in completed.stdout.splitlines() if line.strip()) + context.report_progress("system_details_finished", line_count=line_count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("system-details report collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because systeminfo exits before the result is returned.""" diff --git a/core/system/system_diagnostics.py b/core/system/system_diagnostics.py new file mode 100644 index 00000000..57f8645c --- /dev/null +++ b/core/system/system_diagnostics.py @@ -0,0 +1,105 @@ +"""Export Windows hardware, CPU, page-size, and boot-time diagnostics as bounded JSON.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class SystemDiagnosticsCollector(CoreCollector): + """Capture read-only CPU, architecture, page-size, and boot-time diagnostics.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated system-diagnostics JSON artifact contract.""" + return CollectorMetadata( + id="core.system.system_diagnostics", + name="System diagnostics", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports architecture, CPU, page-size, and boot-time diagnostics through Windows CIM.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=45, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query diagnostics and register their bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before system-diagnostics collection") + context.report_progress("system_diagnostics_started") + command = ( + "$ErrorActionPreference = 'Stop'; $os = Get-CimInstance -ClassName Win32_OperatingSystem; " + "$cpu = @(Get-CimInstance -ClassName Win32_Processor | " + "Select-Object Name, Architecture, AddressWidth, NumberOfCores, " + "NumberOfLogicalProcessors, MaxClockSpeed); " + "[pscustomobject]@{ OperatingSystem = $os.Caption; Version = $os.Version; " + "Architecture = $os.OSArchitecture; Machine = $env:PROCESSOR_IDENTIFIER; " + "PageSizeBytes = [Environment]::SystemPageSize; ProcessorCount = $cpu.Count; " + "Processors = $cpu; LastBootUpTime = $os.LastBootUpTime; " + "CollectedAt = [DateTime]::UtcNow.ToString('o') } | ConvertTo-Json -Depth 5" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=40, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "system-diagnostics access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "system-diagnostics query failed", errors=(detail,)) + try: + diagnostics = json.loads(completed.stdout) + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "system-diagnostics query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(diagnostics, dict): + return CollectorResult(CollectorStatus.FAILED, "system-diagnostics query returned an unexpected result") + output = context.workspace / "system_diagnostics.json" + output.write_text(json.dumps(diagnostics, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("system_diagnostics_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("system diagnostics collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/system_info.py b/core/system/system_info.py new file mode 100644 index 00000000..00360425 --- /dev/null +++ b/core/system/system_info.py @@ -0,0 +1,81 @@ +"""Collect a bounded, non-sensitive inventory of the current Windows system.""" + +from __future__ import annotations + +import getpass +import json +import os +import platform +from datetime import UTC, datetime + +from logicytics import ( + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import network_adapter as socket + + +class SystemInfoCollector(CoreCollector): + """Create a JSON system inventory through the v4 artifact boundary.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the collector identity, scope, and bounded output contract.""" + return CollectorMetadata( + id="core.system.system_info", + name="System information", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Collects a bounded operating-system and hardware inventory.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(), + default_profiles=("minimal", "standard", "deep", "offline"), + timeout_seconds=15, + maximum_output_bytes=128 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Confirm that the standard-library system APIs are available.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Write and register the system inventory as a JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before system inventory collection") + context.report_progress("system_inventory_started") + inventory = { + "collected_at": datetime.now(UTC).isoformat(), + "hostname": socket.gethostname(), + "username": getpass.getuser(), + "operating_system": { + "system": platform.system(), + "release": platform.release(), + "version": platform.version(), + "architecture": platform.architecture(), + }, + "hardware": { + "machine": platform.machine(), + "processor": platform.processor() or "unavailable", + "cpu_count": os.cpu_count(), + }, + "python": platform.python_version(), + } + output = context.workspace / "system_info.json" + output.write_text(json.dumps(inventory, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("system_inventory_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded( + "system inventory collected", + (artifact,), + ) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because this collector owns none beyond its workspace.""" diff --git a/core/system/windows_services.py b/core/system/windows_services.py new file mode 100644 index 00000000..f18446e5 --- /dev/null +++ b/core/system/windows_services.py @@ -0,0 +1,99 @@ +"""Export local Windows service metadata as bounded, read-only JSON evidence.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from PowerShell output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class WindowsServicesCollector(CoreCollector): + """Capture read-only service names, states, startup modes, and paths through CIM.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated Windows-services JSON artifact contract.""" + return CollectorMetadata( + id="core.system.windows_services", + name="Windows services", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/json",), + description="Exports local Windows service names, states, start modes, accounts, and executable paths.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("standard", "deep"), + timeout_seconds=60, + maximum_output_bytes=4 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and PowerShell availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("powershell") is None: + return ValidationResult(False, reasons=("PowerShell is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query service CIM metadata and register a bounded JSON evidence artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Windows-services collection") + context.report_progress("windows_services_started") + command = ( + "Get-CimInstance -ClassName Win32_Service | " + "Select-Object Name, DisplayName, State, StartMode, StartName, PathName, ProcessId | ConvertTo-Json -Depth 3" + ) + completed = subprocess.run( + ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], + capture_output=True, + check=False, + text=True, + timeout=55, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"PowerShell exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "Windows-service access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "Windows-service query failed", errors=(detail,)) + try: + services = json.loads(completed.stdout) if completed.stdout.strip() else [] + except json.JSONDecodeError as error: + return CollectorResult( + CollectorStatus.FAILED, + "Windows-service query returned invalid JSON", + errors=(str(error),), + ) + if not isinstance(services, (dict, list)): + return CollectorResult(CollectorStatus.FAILED, "Windows-service query returned an unexpected result") + output = context.workspace / "windows_services.json" + output.write_text(json.dumps(services, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + count = len(services) if isinstance(services, list) else 1 + context.report_progress("windows_services_finished", service_count=count, bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("Windows services collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because PowerShell exits before the result is returned.""" diff --git a/core/system/windows_system_data_backup.py b/core/system/windows_system_data_backup.py new file mode 100644 index 00000000..20e6fe6d --- /dev/null +++ b/core/system/windows_system_data_backup.py @@ -0,0 +1,107 @@ +"""Copy bounded Windows policy, event-log, and security-support evidence.""" + +from __future__ import annotations + +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + EvidenceKind, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter + +MAX_FILE_BYTES = 100 * 1024 * 1024 + + +class WindowsSystemDataBackupCollector(CoreCollector): + """Back up bounded Windows system-data paths into labeled private directories.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the explicit-consent Windows system-data backup contract.""" + return CollectorMetadata( + id="core.system.windows_system_data_backup", + name="Windows system-data backup", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("application/octet-stream",), + description="Copies bounded Group Policy, event-log, and Windows security-support evidence.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.FILESYSTEM_READ, Capability.SENSITIVE_FILES), + sensitive_data_categories=("security_logs", "group_policy", "system_configuration"), + default_profiles=("deep",), + timeout_seconds=180, + maximum_output_bytes=256 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state before system-data copying starts.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Copy supported Windows system-data files and preserve source metadata.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Windows system-data backup") + windows = filesystem_adapter.environment_path("SystemRoot", Path(r"C:\Windows")) + program_data = filesystem_adapter.environment_path("ProgramData", Path(r"C:\ProgramData")) + candidates = [ + ("group_policy", windows / "System32" / "GroupPolicy" / "Machine" / "Registry.pol"), + ("group_policy", windows / "System32" / "GroupPolicy" / "User" / "Registry.pol"), + ("event_logs", windows / "System32" / "winevt" / "Logs" / "System.evtx"), + ("event_logs", windows / "System32" / "winevt" / "Logs" / "Application.evtx"), + ("event_logs", windows / "System32" / "winevt" / "Logs" / "Security.evtx"), + ] + candidates.extend( + ("security_support", path) + for path in filesystem_adapter.glob(program_data / "Microsoft" / "Windows Defender" / "Support", "*.log") + ) + copied: list[Path] = [] + context.report_progress("windows_system_data_backup_started") + for label, source in candidates: + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during Windows system-data backup") + try: + if not source.is_file() or source.is_symlink() or source.stat().st_size > MAX_FILE_BYTES: + continue + destination = context.workspace / "windows_system_data" / label / source.name + destination.parent.mkdir(parents=True, exist_ok=True) + filesystem_adapter.copy_file(source, destination) + except OSError: + continue + if context.is_cancelled: + destination.unlink(missing_ok=True) + for copied_path in copied: + copied_path.unlink(missing_ok=True) + return CollectorResult(CollectorStatus.CANCELLED, "cancelled during Windows system-data backup") + copied.append(destination) + if not copied: + return CollectorResult( + CollectorStatus.SKIPPED, + "no configured Windows system-data files met the bounded backup policy", + ) + artifacts = [] + for path in copied: + if context.is_cancelled: + for unpublished in copied[len(artifacts):]: + unpublished.unlink(missing_ok=True) + return CollectorResult.cancelled("cancelled during system-data registration", tuple(artifacts)) + artifacts.append(context.artifacts.register_file(path, evidence_kind=EvidenceKind.RAW)) + artifact_tuple = tuple(artifacts) + context.report_progress( + "windows_system_data_backup_finished", + copied_files=len(artifact_tuple), + bytes_written=sum(item.size_bytes for item in artifact_tuple), + ) + return CollectorResult.succeeded("Windows system-data backup collected", artifact_tuple) + + def cleanup(self, context: CollectorContext) -> None: + """Leave copied evidence removal to the isolated workspace lifecycle.""" diff --git a/core/system/wmic_inventory.py b/core/system/wmic_inventory.py new file mode 100644 index 00000000..c2e1729c --- /dev/null +++ b/core/system/wmic_inventory.py @@ -0,0 +1,109 @@ +"""Capture a bounded legacy WMIC computer-system inventory when WMIC is installed.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +class WmicInventoryCollector(CoreCollector): + """Preserve the optional WMIC integration without making it a host prerequisite.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare a deep-profile, subprocess-gated text evidence contract.""" + return CollectorMetadata( + id="core.system.wmic_inventory", + name="WMIC computer-system inventory", + version="4.0.0", + specialty=Specialty.SYSTEM, + output_media_types=("text/plain",), + description=( + "Exports bounded computer-system identity and hardware fields through the optional legacy WMIC executable."), + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("system_configuration",), + default_profiles=("manual",), + timeout_seconds=30, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Treat an absent optional Windows feature as an explicit skip condition.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("wmic") is None: + return ValidationResult( + False, + reasons=("WMIC is not installed; enable the optional WMIC capability to collect this view",), + ) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Run one read-only WMIC query and register its normalized text output.""" + if context.is_cancelled: + return CollectorResult( + CollectorStatus.CANCELLED, + "cancelled before WMIC inventory collection", + ) + executable = which("wmic") + if executable is None: + return CollectorResult( + CollectorStatus.SKIPPED, + "WMIC is not installed on this Windows system", + errors=("enable the optional WMIC capability to collect the legacy view",), + ) + context.report_progress("wmic_inventory_started") + completed = subprocess.run( + [ + executable, + "computersystem", + "get", + ("Name,Manufacturer,Model,SystemType,NumberOfProcessors,NumberOfLogicalProcessors,TotalPhysicalMemory"), + "/format:list", + ], + capture_output=True, + check=False, + text=True, + encoding="utf-16", + errors="replace", + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or completed.stdout.strip() + detail = detail or f"WMIC exit code {completed.returncode}" + if "access" in detail.casefold() and "denied" in detail.casefold(): + return CollectorResult( + CollectorStatus.SKIPPED, + "WMIC access was denied for the current account", + errors=(detail,), + ) + return CollectorResult( + CollectorStatus.FAILED, + "WMIC computer-system query failed", + errors=(detail,), + ) + normalized = completed.stdout.replace("\r\n", "\n").replace("\r", "\n").strip() + if not normalized: + return CollectorResult( + CollectorStatus.FAILED, + "WMIC computer-system query returned no data", + ) + output = context.workspace / "wmic_inventory.txt" + output.write_text(normalized + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + context.report_progress("wmic_inventory_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("WMIC computer-system inventory collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because WMIC exits before the result is returned.""" diff --git a/core/usb/usb_storage_inventory.py b/core/usb/usb_storage_inventory.py new file mode 100644 index 00000000..1a8d5899 --- /dev/null +++ b/core/usb/usb_storage_inventory.py @@ -0,0 +1,122 @@ +"""Collect Windows USB storage device metadata from the read-only USBSTOR registry tree.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import registry_adapter as winreg + +_USBSTOR_PATH = r"SYSTEM\CurrentControlSet\Enum\USBSTOR" + + +def _registry_last_write(key: winreg.HKEYType) -> str | None: + """Return one registry key's last-write timestamp in UTC when Windows reports it.""" + return winreg.last_write_time(key) + + +def _enumerate_usb_storage() -> list[dict[str, str | None]]: + """Return USBSTOR device class, instance ID, and friendly name when readable.""" + devices: list[dict[str, str | None]] = [] + with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, _USBSTOR_PATH) as root: + class_index = 0 + while True: + try: + device_class = winreg.EnumKey(root, class_index) + except OSError: + break + class_index += 1 + class_path = f"{_USBSTOR_PATH}\\{device_class}" + with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, class_path) as class_key: + instance_index = 0 + while True: + try: + instance_id = winreg.EnumKey(class_key, instance_index) + except OSError: + break + instance_index += 1 + instance_path = f"{class_path}\\{instance_id}" + friendly_name = instance_id + try: + with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, instance_path) as instance_key: + friendly_name = str(winreg.QueryValueEx(instance_key, "FriendlyName")[0]) + last_write = _registry_last_write(instance_key) + except OSError: + last_write = None + devices.append( + { + "device_class": device_class, + "instance_id": instance_id, + "friendly_name": friendly_name, + "last_write": last_write, + } + ) + return devices + + +class UsbStorageInventoryCollector(CoreCollector): + """Register a bounded JSON inventory of USB storage devices visible in the registry.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the registry-read capability and device-metadata-only output scope.""" + return CollectorMetadata( + id="core.usb.usb_storage_inventory", + name="USB storage inventory", + version="4.0.0", + specialty=Specialty.USB, + output_media_types=("application/json",), + description="Reads USB storage device class, instance ID, friendly name, and last-write time from USBSTOR.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.REGISTRY_READ,), + default_profiles=("deep",), + timeout_seconds=20, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation and read access to the USBSTOR registry root.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + try: + with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, _USBSTOR_PATH): + return ValidationResult(True) + except OSError as error: + return ValidationResult(False, reasons=(f"USBSTOR registry access is unavailable: {error}",)) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Read USB storage metadata and register the resulting JSON artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before USB registry collection") + context.report_progress("usb_storage_inventory_started") + try: + devices = _enumerate_usb_storage() + except OSError as error: + return CollectorResult( + CollectorStatus.SKIPPED, + "USBSTOR registry access was denied during collection", + errors=(str(error),), + ) + report = {"collected_at": datetime.now(UTC).isoformat(), "devices": devices} + output = context.workspace / "usb_storage_inventory.json" + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress( + "usb_storage_inventory_finished", + device_count=len(devices), + bytes_written=artifact.size_bytes, + ) + return CollectorResult.succeeded("USB storage inventory collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because registry handles are closed during enumeration.""" diff --git a/core/wireless/wifi_interfaces.py b/core/wireless/wifi_interfaces.py new file mode 100644 index 00000000..9c8c1ab6 --- /dev/null +++ b/core/wireless/wifi_interfaces.py @@ -0,0 +1,111 @@ +"""Export local Wi-Fi interface state with a normalized interface-name field.""" + +from __future__ import annotations + +import json + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from netsh output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ( + "access" in normalized and "denied" in normalized) or "requires elevation" in normalized + + +def _normalize_interface_name(name: str) -> str: + """Normalize the common Wi-Fi and WiFi spelling variation for joins.""" + return "Wi-Fi" if name.strip().casefold() in {"wifi", "wi-fi"} else name.strip() + + +class WifiInterfacesCollector(CoreCollector): + """Capture Wi-Fi interface metadata without requesting profile key material.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the bounded, subprocess-gated wireless interface artifact contract.""" + return CollectorMetadata( + id="core.wireless.wifi_interfaces", + name="Wi-Fi interfaces", + version="4.0.0", + specialty=Specialty.WIRELESS, + output_media_types=("application/json",), + description="Exports local Wi-Fi interface state and normalized interface names without key material.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("network_configuration",), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=256 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and netsh availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("netsh") is None: + return ValidationResult(False, reasons=("netsh is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Query Wi-Fi interface output and register normalized JSON evidence.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Wi-Fi interface collection") + context.report_progress("wifi_interfaces_started") + completed = subprocess.run( + ["netsh", "wlan", "show", "interfaces"], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + detail = completed.stderr.strip() + if completed.returncode != 0: + message = detail or completed.stdout.strip() or f"netsh exit code {completed.returncode}" + if _is_access_denied(message): + return CollectorResult( + CollectorStatus.SKIPPED, + "Wi-Fi interface access was denied for the current account", + errors=(message,), + ) + return CollectorResult(CollectorStatus.FAILED, "Wi-Fi interface query failed", errors=(message,)) + + interfaces: list[dict[str, str]] = [] + current: dict[str, str] = {} + for raw_line in completed.stdout.splitlines(): + line = raw_line.strip() + if not line or ":" not in line: + continue + key, value = (part.strip() for part in line.split(":", 1)) + key = key.casefold().replace(" ", "_") + if key == "name" and current: + interfaces.append(current) + current = {} + current[key] = value + if current: + interfaces.append(current) + for interface in interfaces: + name = interface.get("name", "") + interface["normalized_name"] = _normalize_interface_name(name) + + report = {"interfaces": interfaces, "raw_output": completed.stdout.strip()} + output = context.workspace / "wifi_interfaces.json" + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="application/json") + context.report_progress("wifi_interfaces_finished", bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("Wi-Fi interfaces collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because netsh exits before the result is returned.""" diff --git a/core/wireless/wifi_profile_keys.py b/core/wireless/wifi_profile_keys.py new file mode 100644 index 00000000..3024cab6 --- /dev/null +++ b/core/wireless/wifi_profile_keys.py @@ -0,0 +1,98 @@ +"""Export saved Windows Wi-Fi profiles with key material after explicit approval.""" + +from __future__ import annotations + +from pathlib import Path + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + EvidenceKind, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import filesystem_adapter, which +from logicytics.platform_adapters import process_adapter as subprocess + + +def _is_access_denied(detail: str) -> bool: + """Recognize common Windows permission-denied or elevation-required wording.""" + normalized = detail.casefold() + return "permission denied" in normalized or ( + "access" in normalized and "denied" in normalized) or "requires elevation" in normalized + + +class WifiProfileKeysCollector(CoreCollector): + """Export saved Wi-Fi profiles and credentials only after sensitive approval.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the explicit-consent wireless credential export contract.""" + return CollectorMetadata( + id="core.wireless.wifi_profile_keys", + name="Saved Wi-Fi profile keys", + version="4.0.0", + specialty=Specialty.WIRELESS, + output_media_types=("application/xml",), + description="Exports saved Wi-Fi profile XML including key material after explicit sensitive-data approval.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS, Capability.SENSITIVE_FILES), + sensitive_data_categories=("wireless_profile_names", "credentials"), + default_profiles=("deep",), + timeout_seconds=60, + maximum_output_bytes=2 * 1024 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and netsh availability before credential export.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("netsh") is None: + return ValidationResult(False, reasons=("netsh is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """Export profile XML into this collector's private workspace and register it.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Wi-Fi profile-key collection") + context.report_progress("wifi_profile_keys_started") + export_directory = context.workspace / "wifi_profiles_with_keys" + export_directory.mkdir() + completed = subprocess.run( + ["netsh", "wlan", "export", "profile", f"folder={export_directory}", "key=clear"], + capture_output=True, + check=False, + text=True, + timeout=55, + ) + detail = completed.stderr.strip() or completed.stdout.strip() + if completed.returncode != 0: + message = detail or f"netsh exit code {completed.returncode}" + if _is_access_denied(message): + return CollectorResult( + CollectorStatus.SKIPPED, + "Wi-Fi profile-key access was denied for the current account", + errors=(message,), + ) + return CollectorResult(CollectorStatus.FAILED, "Wi-Fi profile-key export failed", errors=(message,)) + profiles = sorted(path for path in filesystem_adapter.glob(export_directory, "*.xml") if path.is_file()) + if not profiles: + context.report_progress("wifi_profile_keys_finished", profile_count=0, bytes_written=0) + return CollectorResult.succeeded("no saved Wi-Fi profiles with key material were exported") + artifacts = tuple( + context.artifacts.register_file(Path(profile), media_type="application/xml", evidence_kind=EvidenceKind.RAW) + for profile in profiles + ) + context.report_progress( + "wifi_profile_keys_finished", + profile_count=len(artifacts), + bytes_written=sum(item.size_bytes for item in artifacts), + ) + return CollectorResult.succeeded("saved Wi-Fi profile keys collected", artifacts) + + def cleanup(self, context: CollectorContext) -> None: + """Leave cleanup to the isolated collector workspace lifecycle.""" diff --git a/core/wireless/wifi_profiles.py b/core/wireless/wifi_profiles.py new file mode 100644 index 00000000..233ea9d6 --- /dev/null +++ b/core/wireless/wifi_profiles.py @@ -0,0 +1,84 @@ +"""Export saved Windows Wi-Fi profile names as bounded, read-only text evidence.""" + +from __future__ import annotations + +from logicytics import ( + Capability, + CollectorMetadata, + CollectorResult, + CoreCollector, + Specialty, + ValidationResult, +) +from logicytics.contracts import CollectorContext, CollectorStatus +from logicytics.platform_adapters import process_adapter as subprocess +from logicytics.platform_adapters import which + + +def _is_access_denied(detail: str) -> bool: + """Recognize common permission-denied wording from netsh output.""" + normalized = detail.casefold() + return "permission denied" in normalized or ("access" in normalized and "denied" in normalized) + + +class WifiProfilesCollector(CoreCollector): + """Capture saved wireless profile names without reading profile key material.""" + + @classmethod + def metadata(cls) -> CollectorMetadata: + """Declare the subprocess-gated Wi-Fi profile-name artifact contract.""" + return CollectorMetadata( + id="core.wireless.wifi_profiles", + name="Saved Wi-Fi profiles", + version="4.0.0", + specialty=Specialty.WIRELESS, + output_media_types=("text/plain",), + description="Exports local saved Wi-Fi profile names without retrieving key material.", + author="Logicytics", + supported_platforms=("win32",), + capabilities=(Capability.SUBPROCESS,), + sensitive_data_categories=("wireless_profile_names",), + default_profiles=("deep",), + timeout_seconds=30, + maximum_output_bytes=512 * 1024, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + """Check cancellation state and netsh availability before collection.""" + if context.is_cancelled: + return ValidationResult(False, reasons=("run cancellation was requested",)) + if which("netsh") is None: + return ValidationResult(False, reasons=("netsh is unavailable on this system",)) + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + """List saved Wi-Fi profiles and register their read-only text artifact.""" + if context.is_cancelled: + return CollectorResult(CollectorStatus.CANCELLED, "cancelled before Wi-Fi profile collection") + context.report_progress("wifi_profiles_started") + completed = subprocess.run( + ["netsh", "wlan", "show", "profiles"], + capture_output=True, + check=False, + text=True, + timeout=25, + ) + if completed.returncode != 0: + detail = completed.stderr.strip() or f"netsh exit code {completed.returncode}" + if _is_access_denied(detail): + return CollectorResult( + CollectorStatus.SKIPPED, + "Wi-Fi profile access was denied for the current account", + errors=(detail,), + ) + return CollectorResult(CollectorStatus.FAILED, "Wi-Fi profile query failed", errors=(detail,)) + output = context.workspace / "wifi_profiles.txt" + output.write_text(completed.stdout, encoding="utf-8") + artifact = context.artifacts.register_file(output, media_type="text/plain") + profile_count = sum(1 for line in completed.stdout.splitlines() if " : " in line) + context.report_progress("wifi_profiles_finished", profile_count=profile_count, + bytes_written=artifact.size_bytes) + return CollectorResult.succeeded("saved Wi-Fi profiles collected", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + """Release no resources because netsh exits before the result is returned.""" diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 00000000..b5cb83d8 --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,40 @@ +# Repository architecture + +## Top-level folders + +| Folder | Responsibility | +|---------------|-----------------------------------------------------------------------------------------------------------------------------------| +| `logicytics/` | Public package, CLI, and the internal engine modules for contracts, configuration, discovery, planning, runtime, logging, manifests, packaging, and platform adapters | +| `core/` | Shipped read-only collector implementations grouped by specialty | +| `plugins/` | Opt-in user-owned `PluginCollector` implementations | +| `tests/` | Unit, contract, resilience, security, and Windows integration tests | +| `docs/` | This user, operator, and developer manual | +| `output/` | Local run output; do not commit evidence | +| `.github/` | Contribution templates, security automation, and documentation publishing workflow | + +## Engine modules + +- `module/contracts.py` defines enums, metadata, request, context, artifact, result, and collector interfaces. +- `configuration.py` loads and validates `logicytics.yaml`. +- `discovery.py` finds candidates, performs static checks, probes metadata, and manages the disposable validation cache. +- `planner.py` resolves profiles, selectors, dependencies, capabilities, ordering, and run fingerprints. +- `runtime.py` owns worker processes, quotas, cancellation, retry rules, progress, and terminal statuses. +- `artifacts.py` is the workspace-bound artifact writer. +- `manifest.py` writes the durable run record. +- `packaging.py` builds and verifies ZIP packages and hash sidecars. +- `api.py` exposes planning, collection, run queries, and bounded artifact reads. +- `module/platform_adapters.py` owns the injectable host-process and Windows API boundaries. +- `cli/terminal.py` manages console lifecycle and `cli/virtual_environment.py` validates the supported interpreter environment. +- `cli/commands.py` maps command-line arguments to the public engine operations. + +## Boundary rules + +Collectors may depend on contracts and approved platform adapters. They should not reach into supervisor internals, +write outside their workspace, alter global configuration, or create unbounded output. The runtime owns cleanup and +process termination. The manifest is the source for later inspection; do not infer success from a console line alone. + +## Core source layout + +The core ID follows `core..` and normally maps to `core//.py`. A module contains one +collector class with a PascalCase name ending in `Collector`, a `metadata()` class method, lifecycle methods, and no +import-time collection. The catalog in [Core Collectors](CORE_COLLECTORS.md) is grouped by the live source tree. diff --git a/docs/COMMANDS.md b/docs/COMMANDS.md new file mode 100644 index 00000000..2a2afae5 --- /dev/null +++ b/docs/COMMANDS.md @@ -0,0 +1,170 @@ +# Command reference + +Run commands from the repository root. Except for the installer, use the managed interpreter exactly as shown: + +```text +.\.venv\Scripts\python.exe -m logicytics [global options] [command options] +``` + +An option may appear before or after a command when the command supports it. Repeated options may be supplied more than once. `plan` is the safe checkpoint: it resolves scope without collecting evidence. + +## Global options + +| Flag | What it does | +| --- | --- | +| `--config PATH` | Selects the authoritative YAML configuration. The path is validated before planning. | +| `--usb [DRIVE]` | Targets a Windows installation on removable storage; omit `DRIVE` to scan, or provide a drive letter. | +| `--usage` | Displays local interaction statistics and writes the usage graph. | +| `--modes` | Prints and writes the generated execution-mode inclusion matrix. | +| `--match TEXT` | Suggests the closest documented action from a natural-language description. | +| `-h`, `--help` | Shows parser help without collecting evidence. | + +## Installer + +```powershell +python -m logicytics.cli.installer [--environment PATH] [--overwrite-config] +``` + +The installer is the only entry point intended for a system Python. It creates or reuses the virtual environment and creates `logicytics.yaml` if absent. + +| Flag | What it does | +| --- | --- | +| `--environment PATH` | Uses this virtual-environment directory instead of `.venv`; a relative path is resolved from the repository root. | +| `--overwrite-config` | Replaces the root configuration template. Back up a customized file first. | + +## `preflight` + +```powershell +.\.venv\Scripts\python.exe -m logicytics preflight [flags] +``` + +Validates core and, when enabled, plugin sources before collection. It reports valid collectors, quarantined invalid optional plugins, and blocking selected failures. + +| Flag | What it does | +| --- | --- | +| `--config PATH`, `--usb [DRIVE]` | Applies the global configuration or removable-media source selection. | +| `--profile MODE` | Uses one named mode for selected-plugin validation context. | +| `--include ID` | Repeats to select exact collector IDs; selected invalid plugins become blocking failures. | +| `--exclude ID` | Repeats to remove IDs from the request. | +| `--plugins` | Enables valid `PluginCollector` implementations from `plugins/` for the request. | +| `--invalidate-cache` | Discards cached validation metadata and performs a fresh source/runtime probe. | +| `--workers COUNT` | Supplies the bounded worker count carried into request validation. | +| `--block-capability CAP` | Repeats to prohibit a declared capability. See [Safety](SAFETY.md). | + +## `plan` + +```powershell +.\.venv\Scripts\python.exe -m logicytics plan --mode standard [flags] +``` + +Builds and prints a deterministic, dependency-ordered plan. It starts no collector worker and writes no evidence artifacts. + +| Flag | What it does | +| --- | --- | +| `--mode {quick,balanced,standard,offline,thorough}` | Selects the request mode and its scheduling behavior. | +| `--profile MODE` | Alternative named mode selection. Do not combine it with `--mode`. | +| `--include ID` / `--exclude ID` | Repeats to refine the resolved profile with exact dotted collector IDs. | +| `--plugins` | Allows valid opt-in plugins to be considered; plugins remain opt-in even when their profile matches. | +| `--invalidate-cache` | Revalidates collector metadata before planning. | +| `--workers COUNT` | Sets a positive bounded worker count for the resulting request. | +| `--block-capability CAP` | Rejects a plan containing a collector that declares the capability. | +| `--config PATH`, `--usb [DRIVE]` | Selects configuration or removable-media context. | + +## `run` + +```powershell +.\.venv\Scripts\python.exe -m logicytics run --mode standard --acknowledge-authorization [flags] +``` + +Runs the reviewed plan in isolated workers. Read the final manifest even after an exit code of `0`; a package is not proof that every collector succeeded. + +| Flag | What it does | +| --- | --- | +| `--mode MODE` | Selects a named execution mode. | +| `--default`, `--threaded`, `--minimal`, `--depth` | Historical aliases for `standard`, `balanced`, `quick`, and `thorough`; exactly one mode selector is allowed. | +| `--profile MODE` | Alternative named mode selector; cannot be combined with a mode selector. | +| `--include ID` / `--exclude ID` | Repeats to add or remove exact collector IDs. | +| `--plugins` | Enables valid opt-in plugins for this run. | +| `--workers COUNT` | Caps concurrent isolated workers. Sequential work requires `1`; parallel work requires at least `2`. | +| `--sequential` / `--parallel` | Mutually exclusive scheduling override. Use sequential execution for deterministic diagnosis. | +| `--performance-check` | Forces serial measurement and writes per-collector duration evidence. It cannot be combined with `--parallel`. | +| `--block-capability CAP` | Repeats to prohibit declared access categories. | +| `--rerun-from PATH` | Re-runs explicitly included IDs from a finalized compatible manifest or run directory. | +| `--no-package` | Finalizes the manifest-backed run directory without writing a ZIP package. | +| `--reboot` / `--shutdown` | Mutually exclusive post-run host actions, scheduled only after verified package publication. | +| `--acknowledge-authorization` | Required confirmation that the operator has authority for the reviewed scope. | +| `--interactive` | Keeps the interactive command window open after final status. | +| `--config PATH`, `--usb [DRIVE]` | Selects configuration or removable-media context. | + +## `collector` + +```powershell +.\.venv\Scripts\python.exe -m logicytics collector core.system.system_info --acknowledge-authorization [flags] +``` + +Runs one exact collector ID independently. It is the preferred way to diagnose one source after reviewing its capabilities. + +| Flag | What it does | +| --- | --- | +| `COLLECTOR_ID` | Required positional dotted ID for the one collector to run. | +| `--acknowledge-authorization` | Required authorization confirmation. | +| `--no-package` | Keeps the manifest-backed result without a ZIP. | +| `--interactive` | Keeps an interactive command window open at completion. | +| `--workers COUNT` | Sets the bounded worker limit; the direct collector still resolves one selected source. | +| `--block-capability CAP` | Rejects the collector when it declares a blocked capability. | +| `--config PATH`, `--usb [DRIVE]` | Selects configuration or removable-media context. | + +`collector` cannot be combined with `--profile`, `--include`, `--exclude`, or `--plugins`; its positional ID is the complete selection. + +## `debug` + +```powershell +.\.venv\Scripts\python.exe -m logicytics debug [flags] +``` + +Writes a bounded diagnostic JSON report containing environment, interpreter, configuration, maintenance, optional-tool, and preflight information. Sanitize it before sharing. + +| Flag | What it does | +| --- | --- | +| `--config PATH`, `--usb [DRIVE]` | Uses the selected configuration and target context while gathering diagnostics. | +| `--profile MODE`, `--include ID`, `--exclude ID`, `--plugins` | Applies the request selection used for preflight diagnostics. | +| `--workers COUNT`, `--block-capability CAP` | Carries execution and policy context into the diagnostic request. | + +## `update` + +```powershell +.\.venv\Scripts\python.exe -m logicytics update [--apply] [--launch-action ACTION --new-window] +``` + +Checks Git, the repository, origin configuration, and remote reachability. It never pulls unless `--apply` is explicit. + +| Flag | What it does | +| --- | --- | +| `--apply` | Runs `git pull` after successful repository and remote checks. | +| `--launch-action {preflight,debug,dev}` | Selects an allowlisted maintenance action to launch after update. Requires `--new-window`. | +| `--new-window` | Starts the allowlisted action in a visible separate Windows console. | +| `--config PATH`, `--usb [DRIVE]` | Accepts common context flags; update behavior itself operates on the checkout. | + +## `dev` + +```powershell +.\.venv\Scripts\python.exe -m logicytics dev [--write-manifest --next-version VERSION] [--interactive] +``` + +Runs repository integrity and organization checks. It is a developer maintenance command, not an evidence collection command. + +| Flag | What it does | +| --- | --- | +| `--write-manifest` | Writes the reviewed local integrity manifest; requires `--next-version` unless interactive mode supplies it. | +| `--next-version VERSION` | Provides the semantic version recorded in a newly written manifest. | +| `--interactive` | Shows checks and prompts before an integrity-manifest write. | +| `--config PATH`, `--usb [DRIVE]` | Accepts common context flags. | + +## Exit codes + +| Code | Meaning | Next action | +| --- | --- | --- | +| `0` | Command completed successfully. | Review the manifest or generated report. | +| `1` | A collection finished with non-success status. | Preserve the manifest and inspect partial, skipped, failed, or cancelled records. | +| `2` | Argument, configuration, preflight, planning, or command error. | Read the rendered error and [Error reference](ERRORS.md); correct the cause rather than bypassing validation. | +| `130` | Interrupted. | Locate the durable manifest and determine what safely finalized before retrying. | diff --git a/docs/CONFIGURATION.md b/docs/CONFIGURATION.md new file mode 100644 index 00000000..51a57c5c --- /dev/null +++ b/docs/CONFIGURATION.md @@ -0,0 +1,92 @@ +# Configuration + +`logicytics.yaml` is the authoritative user configuration. It is a strict, mapping-only YAML subset; JSON is accepted +when it is also valid YAML. The parser rejects unknown keys, duplicate keys, unsafe paths, non-finite numbers, +unsupported values, and files larger than 2 MiB before planning. + +## Root keys + +The root keys are `schema_version`, `runtime`, `interaction`, `maintenance`, `logging`, and `collectors`. The parser +currently requires `schema_version: 4`; this is a file-schema identifier, not a release guide. + +| Section | Keys | +|---------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `runtime` | `output_root`, `default_max_workers`, `maximum_workers`, `package_completed_runs`, `maximum_run_output_bytes`, `blocked_capabilities`, `temporary_directory` | +| `interaction` | `history_enabled`, `similarity_threshold`, `model_name`, `model_debug` | +| `maintenance` | `remote_manifest_url`, `remote_manifest_sha256`, `local_manifest_path`, `minimum_python`, `recommended_python`, `sysinternals_enabled`, `sysinternals_download_url` | +| `logging` | `level`, `console_enabled`, `color_enabled`, `file_enabled`, `maximum_bytes`, `delete_previous`, `retention_days` | +| `collectors` | A mapping from collector ID to that collector's declared settings | + +Relative `runtime.output_root` and `maintenance.local_manifest_path` stay inside the project. `temporary_directory` is +`project` or `system`. `blocked_capabilities` is a mapping of capability names to booleans; `true` blocks matching +requests. Remote manifests require HTTPS and a lowercase SHA-256 digest. Optional Sysinternals discovery is controlled +by `sysinternals_enabled` and its download URL. + +## Collector settings + +Core collectors reject settings they do not declare. Extension IDs may define their own settings. The supported core +settings are: + +| Collector | Settings | +|--------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------| +| `core.network.bandwidth_sample` | `sample_count` 1–10; `interval_seconds` 0.1–60 | +| `core.packet.packet_capture` | `packet_count` 1–10000; `timeout_seconds` 1–60; `retry_window_seconds` 0–60; `interface` text | +| `core.filesystem.system_drive_tree` | `max_entries` 1–50000; `max_depth` 1–32 | +| `core.filesystem.system_drive_listing` | `max_entries` 1–50000; `max_depth` 1–32 | +| `core.filesystem.sensitive_file_inventory` | `root` path; `max_directories` 1–50000; `max_matches` 1–5000 | +| `core.process.memory_map` | `max_regions` 1–100000; `output_limit_bytes` 1024–67108864; `disk_safety_margin_bytes` 0–68719476736; `dump_directory` relative path | + +## Complete example + +```json +{ + "schema_version": 4, + "runtime": { + "output_root": "output/data", + "default_max_workers": 4, + "maximum_workers": 16, + "package_completed_runs": true, + "maximum_run_output_bytes": 4294967296, + "temporary_directory": "project", + "blocked_capabilities": {"network": false, "subprocess": false} + }, + "interaction": { + "history_enabled": false, + "similarity_threshold": 0.55, + "model_name": "stdlib-sequence-matcher", + "model_debug": false + }, + "maintenance": { + "remote_manifest_url": null, + "remote_manifest_sha256": null, + "local_manifest_path": "project.manifest.json", + "minimum_python": "3.11", + "recommended_python": "3.11", + "sysinternals_enabled": true, + "sysinternals_download_url": "https://download.sysinternals.com/files/SysinternalsSuite.zip" + }, + "logging": { + "level": "INFO", + "console_enabled": true, + "color_enabled": true, + "file_enabled": true, + "maximum_bytes": 4194304, + "delete_previous": false, + "retention_days": 30 + }, + "collectors": { + "core.network.bandwidth_sample": {"sample_count": 3, "interval_seconds": 1}, + "core.packet.packet_capture": {"packet_count": 100, "timeout_seconds": 10, "retry_window_seconds": 0, "interface": "Ethernet"}, + "core.filesystem.system_drive_tree": {"max_entries": 5000, "max_depth": 4}, + "core.filesystem.system_drive_listing": {"max_entries": 5000, "max_depth": 4}, + "core.filesystem.sensitive_file_inventory": {"root": "C:\\Evidence", "max_directories": 1000, "max_matches": 100}, + "core.process.memory_map": {"max_regions": 5000, "output_limit_bytes": 67108864, "disk_safety_margin_bytes": 104857600, "dump_directory": "memory_maps"} + } +} +``` + +Validate a changed file before collecting: + +```powershell +python -m logicytics preflight --config .\logicytics.yaml +``` diff --git a/docs/CONTRACTS.md b/docs/CONTRACTS.md new file mode 100644 index 00000000..0abe8ec7 --- /dev/null +++ b/docs/CONTRACTS.md @@ -0,0 +1,103 @@ +# Engine and plugin contracts + +`logicytics/module/contracts.py` is the stable vocabulary shared by the CLI, planner, runtime, core collectors, and the only supported extension type: `PluginCollector`. A contract violation is intentionally rejected during construction, preflight, planning, worker execution, artifact registration, or packaging rather than being silently repaired. + +## Identity and ownership + +Core IDs use `core..`; plugin IDs use `plugin.`. IDs are lowercase dotted identifiers, while labels such as profiles and categories are lowercase snake case. A core module must agree with its `core//.py` location. A plugin must agree with its plugin file or `main.py` folder ownership. + +`CollectorKind` has only `core` and `plugin`. Plugins may use a custom lowercase specialty; core collectors must use the closed `Specialty` enum. All metadata construction must be deterministic and side-effect-free. + +## `CollectorMetadata` + +Metadata is policy, not display text. It is evaluated before a collector may enter a plan. + +| Field | Requirement and effect | +|----------------------------------------------------|----------------------------------------------------------------------------------------------------------------| +| `id`, `name`, `description`, `author` | Non-empty single-line strings. `id` must match its owner and location. | +| `version` | Semantic version, for example `1.2.0`. | +| `specialty` | A core `Specialty` value or a plugin's custom lowercase category. | +| `supported_platforms` | Non-empty tuple of lowercase platform labels; a mismatched current platform rejects the plan. | +| `capabilities` | Tuple of declared `Capability` values. Declared access is visible to planning and can be blocked. | +| `privilege_level` | `standard` or `elevated`; it must agree exactly with the `elevated_privileges` capability. | +| `network_access` | `none`, `local`, or `remote`; `network` capability requires local or remote access. | +| `sensitive_data_categories` | Lowercase labels describing sensitive output. Minimal mode cannot include a sensitive collector. | +| `default_profiles` | Non-empty mode-profile membership. Standard mode admits only routine configuration categories. | +| `dependencies` | Unique collector IDs, never the collector's own ID. Dependencies are ordered and policy-checked. | +| `output_media_types` | Non-empty unique MIME types. Every registered artifact must use one of these types. | +| `timeout_seconds` | Positive worker time limit. | +| `maximum_memory_bytes`, `maximum_output_bytes` | Positive per-collector resource ceilings. | +| `maximum_artifact_bytes`, `maximum_artifact_files` | Positive artifact limits; a missing per-artifact byte limit defaults to the output limit and cannot exceed it. | +| `maximum_retries`, `retry_delay_seconds` | Bounded retry policy: retries are 0–3 and delay is finite from 0–30 seconds. | +| `minimum_contract_version` | A valid `major.minor` contract version; the current contract is `4.0`. | +| `parallel_safe`, `resource_class` | Scheduling declaration. A resource class prevents unsafe overlap with conflicting work. | + +### Capability values + +| Capability | Declare it when the collector needs to | +|---------------------------------------------------|--------------------------------------------------------------------------------------| +| `filesystem_read` / `filesystem_write` | Read host files or write outside its supplied private workspace. | +| `registry_read` | Read Windows registry state. | +| `subprocess` | Invoke a host executable or shell-free command. | +| `network` / `packet_capture` | Reach the network or capture packet metadata. | +| `browser_data`, `sensitive_files`, `private_keys` | Handle those sensitive data categories. | +| `elevated_privileges` | Require an administrator context; metadata must also set `privilege_level=elevated`. | + +The planner rejects selected collectors with a request-blocked capability. Offline mode additionally rejects `network` and `packet_capture` collectors. Capability declaration does not grant permission to evade the worker boundary. + +## Collector lifecycle + +```python +class ExamplePluginCollector(PluginCollector): + @classmethod + def metadata(cls) -> CollectorMetadata: ... + def validate(self, context: CollectorContext) -> ValidationResult: ... + @staticmethod + def prepare(context: CollectorContext) -> ValidationResult: ... + def collect(self, context: CollectorContext) -> CollectorResult: ... + @staticmethod + def finalize(context: CollectorContext, result: CollectorResult) -> CollectorResult: ... + @staticmethod + def estimate(context: CollectorContext) -> CollectionEstimate: ... + @classmethod + def dependencies(cls) -> tuple[str, ...]: ... + def cleanup(self, context: CollectorContext) -> None: ... +``` + +`metadata` is declarative. `validate` is side-effect-free and checks prerequisites. `prepare` performs bounded collector-local setup. `collect` does the bounded evidence work. `finalize` returns the publishable result. `cleanup` releases collector-owned resources even when work failed; it must not delete engine-owned run data. `estimate` and `dependencies` are optional hooks when meaningful. + +Preflight also enforces the public class shape. Do not add import-time work, dynamic code loading, hidden public methods, global mutable run state, nested worker pools, `print`-driven control flow, or collector-to-collector calls. + +## Context, results, and progress + +`CollectorContext` supplies the run ID, collector ID, private `workspace`, private `temporary_directory`, configuration settings, artifact writer, event logger, and cancellation state. Use the typed setting helpers (`setting_int`, `setting_float`, and `setting_str`) rather than coercing arbitrary values. Check `context.is_cancelled` before and during every traversal, copy, query, capture, loop, or long subprocess. + +`CollectorResult` uses one terminal `CollectorStatus`: + +| Status | Use it when | +|-------------|---------------------------------------------------------------------------------| +| `succeeded` | The requested work completed and registered its declared output. | +| `partial` | Useful registered evidence exists, but a meaningful portion could not complete. | +| `skipped` | A normal prerequisite, device, feature, or permission is unavailable. | +| `cancelled` | The engine or user cancellation boundary was reached. | +| `failed` | The collector encountered an execution or contract error. | + +Always provide a non-empty summary. Attach structured errors and scalar metrics where they aid review. Do not describe a partial or skipped result as success. + +## Artifact contract + +`context.artifacts.register_file(...)` is the only publication path. The source must be complete, non-empty, beneath the worker workspace, owned by the current collector, and use a declared MIME type. The writer validates its path, content size, SHA-256, file-count and byte quotas, and provenance before placing it under the run-owned artifact store. + +An `Artifact` contains a stable ID, POSIX relative path, lowercase SHA-256, size, MIME type, collector ID, source category, timezone-aware collection timestamp, transformations, evidence kind, safe filename, and `registered` status. Artifact paths cannot escape the owner directory; packages are built only from the finalized manifest catalog. + +## Plugin author checklist + +1. Put one well-named `PluginCollector` implementation under `plugins/`. +2. Make metadata complete, honest, deterministic, and compatible with contract `4.0`. +3. Make validation explain absent prerequisites without collecting data. +4. Bound every loop, file, command, memory allocation, and output path. +5. Use platform adapters and the provided context rather than direct global host access. +6. Register only complete declared artifacts, return a typed result, and clean up collector-owned resources. +7. Add focused tests, run `preflight --plugins`, then inspect the plan before enabling the plugin in a collection. + +See [Plugin Authoring](PLUGIN_AUTHORING.md) for a minimal example and [Errors](ERRORS.md) for remediation of contract failures. diff --git a/docs/CORE_COLLECTORS.md b/docs/CORE_COLLECTORS.md new file mode 100644 index 00000000..17c50a85 --- /dev/null +++ b/docs/CORE_COLLECTORS.md @@ -0,0 +1,100 @@ +# Core collector catalog + +Core collectors are shipped, discovered automatically, and selected by a profile or exact `--include` ID. The table +below is generated from the live `metadata()` declarations. `standard` means routine local inventory; `deep` means +broader or more sensitive collection; `minimal`, `offline`, and `manual` identify narrower or explicitly selected +entries. Capabilities are the access categories to review before authorization. + +## System and hardware + +| ID | Purpose | Profiles | Access | Output | +|------------------------------------------|-------------------------------------------------------------|----------------------------------|----------------------------------|--------| +| `core.system.bios_info` | BIOS manufacturer, name, version, release date | standard, deep | subprocess | HTML | +| `core.system.computer_system` | Computer model, manufacturer, processor count | standard, deep | subprocess | JSON | +| `core.system.defender_status` | Defender engine, signatures, protection state | standard, deep | subprocess | JSON | +| `core.system.environment_posture` | Administrator state, UAC, PowerShell policies | standard, deep | subprocess | JSON | +| `core.system.group_policy` | Group Policy Result summary | standard, deep | subprocess | TXT | +| `core.system.installed_drivers` | Windows driver inventory | standard, deep | subprocess | CSV | +| `core.system.installed_updates` | Installed hotfix IDs, descriptions, dates | standard, deep | subprocess | JSON | +| `core.system.local_accounts` | Local account names, SIDs, state, descriptions, login times | deep | subprocess | JSON | +| `core.system.operating_system` | Detailed Windows operating-system data | standard, deep | subprocess | JSON | +| `core.system.scheduled_tasks` | Bounded scheduled-task names, paths, authors, states | deep | subprocess | JSON | +| `core.system.session_snapshot` | User/SID, OS build, memory, language, host, time, drive | deep | subprocess | JSON | +| `core.system.system_details` | Windows `systeminfo` report | standard, deep | subprocess | TXT | +| `core.system.system_diagnostics` | Architecture, CPU, page size, boot time | standard, deep | subprocess | JSON | +| `core.system.system_info` | Bounded OS and hardware inventory | minimal, standard, deep, offline | _None_ | JSON | +| `core.system.windows_services` | Service names, state, start mode, account, executable | standard, deep | subprocess | JSON | +| `core.system.windows_system_data_backup` | Bounded policy, event-log, and security-support evidence | deep | filesystem_read, sensitive_files | BIN | +| `core.system.wmic_inventory` | Optional legacy WMIC computer-system inventory | manual | subprocess | TXT | +| `core.hardware.battery_status` | Battery status, charge, capacity, runtime | standard, deep | subprocess | JSON | +| `core.hardware.display_adapters` | Display adapter, driver, resolution, memory | standard, deep | subprocess | JSON | +| `core.hardware.windows_features` | Optional-feature names and enabled states | standard, deep | subprocess | JSON | + +## Storage, filesystem, and memory + +| ID | Purpose | Profiles | Access | Output | +|--------------------------------------------|---------------------------------------------------|----------------------------------|----------------------------------|--------| +| `core.storage.logical_drives` | Logical-drive types and aggregate capacity | minimal, standard, deep, offline | _None_ | JSON | +| `core.storage.mounted_volumes` | Volume GUID and mount-point mappings | standard, deep | subprocess | TXT | +| `core.storage.physical_disks` | Disk model, media, interface, size | standard, deep | subprocess | JSON | +| `core.storage.volume_details` | Drive type, filesystem, label, capacity | standard, deep | _None_ | JSON | +| `core.filesystem.startup_folder_entries` | Startup-folder names, locations, size, timestamps | deep | filesystem_read | JSON | +| `core.filesystem.system_drive_listing` | Bounded recursive system-drive listing | deep | filesystem_read | TXT | +| `core.filesystem.system_drive_tree` | Bounded recursive directory tree | deep | filesystem_read | TXT | +| `core.filesystem.sensitive_file_inventory` | Bounded matching sensitive filenames and copies | deep | filesystem_read, sensitive_files | BIN | +| `core.memory.memory_snapshot` | Physical, virtual, and page-file totals | minimal, standard, deep, offline | _None_ | JSON | +| `core.process.memory_map` | Readable virtual-memory regions and RSS | deep | _None_ | JSON | + +## Processes and network + +| ID | Purpose | Profiles | Access | Output | +|-------------------------------------|---------------------------------------------------|----------------|----------------------------------------------|--------------| +| `core.process.running_processes` | Non-verbose task list | standard, deep | subprocess | CSV | +| `core.process.detailed_processes` | Verbose task list | deep | subprocess | CSV | +| `core.process.process_memory` | Working-set, private, virtual counters | deep | subprocess | JSON | +| `core.network.active_connections` | TCP/UDP endpoints, state, owning PID | deep | subprocess | TXT | +| `core.network.adapter_statistics` | Interface bytes, packets, errors, discards | deep | subprocess | JSON | +| `core.network.arp_cache` | Local ARP cache | deep | subprocess | TXT | +| `core.network.bandwidth_sample` | Average and peak interface bandwidth | deep | subprocess | JSON | +| `core.network.connection_processes` | Netstat endpoints correlated with processes | deep | subprocess | CSV | +| `core.network.dns_cache` | Local DNS resolver cache | deep | subprocess | TXT | +| `core.network.firewall_profiles` | Domain, Private, Public firewall settings | standard, deep | subprocess | JSON | +| `core.network.network_adapters` | IP configuration and adapter details | standard, deep | subprocess | TXT | +| `core.network.network_identity` | Hostname and resolver-provided addresses | standard, deep | network | JSON | +| `core.network.network_interfaces` | IPv4, masks, link, speed, duplex | deep | subprocess | JSON | +| `core.network.routing_table` | IPv4 and IPv6 routes | deep | subprocess | TXT | +| `core.packet.connection_graph` | DOT source/destination graph with protocol labels | deep | subprocess | Graphviz DOT | +| `core.packet.packet_capture` | Bounded IPv4 packet metadata, no payloads | deep | network, packet_capture, elevated_privileges | CSV | + +## Wireless, Bluetooth, USB, and registry + +| ID | Purpose | Profiles | Access | Output | +|----------------------------------------|-----------------------------------------------|----------------|--------------------------------------------|------------| +| `core.wireless.wifi_interfaces` | Wi-Fi state and normalized names | deep | subprocess | JSON | +| `core.wireless.wifi_profiles` | Saved Wi-Fi profile names | deep | subprocess | TXT | +| `core.wireless.wifi_profile_keys` | Saved profile XML including key material | deep | subprocess, sensitive_files | XML | +| `core.bluetooth.paired_devices` | Bluetooth PnP device metadata | deep | subprocess | JSON | +| `core.bluetooth.bluetooth_addresses` | Bluetooth names and address-like identifiers | deep | subprocess | JSON | +| `core.bluetooth.bluetooth_history` | Timestamped Bluetooth PnP snapshot | deep | subprocess | JSON | +| `core.usb.usb_storage_inventory` | USB storage class, ID, name, last-write time | deep | registry_read | JSON | +| `core.registry.installed_applications` | Names, versions, publishers, install metadata | standard, deep | registry_read | JSON | +| `core.registry.startup_applications` | User and machine Run/RunOnce entries | standard, deep | registry_read | JSON | +| `core.registry.hklm_backup` | HKLM hive backup | deep | registry_read, subprocess, sensitive_files | `.reg` TXT | + +## Logs, encryption, browser, and integrations + +| ID | Purpose | Profiles | Access | Output | +|----------------------------------------|-----------------------------------------------------|----------------|------------------------------------------------|--------| +| `core.event_log.application_events` | Up to 1,000 Application events | deep | subprocess | CSV | +| `core.event_log.security_events` | Up to 1,000 Security events | deep | subprocess | CSV | +| `core.event_log.system_events` | Up to 1,000 System events | deep | subprocess | CSV | +| `core.encryption.bitlocker_status` | Read-only drive encryption status | standard, deep | subprocess | TXT | +| `core.encryption.bitlocker_volumes` | Read-only BitLocker volume metadata | standard, deep | subprocess | JSON | +| `core.browser.browser_data_backup` | Bounded local Edge, Chrome, Firefox, Opera evidence | deep | filesystem_read, browser_data, sensitive_files | BIN | +| `core.media.media_backup` | Bounded Pictures/Videos JPG, PNG, MP4 copies | deep | filesystem_read, sensitive_files | BIN | +| `core.ssh.ssh_backup` | `.ssh` keys and configuration archive | deep | filesystem_read, sensitive_files, private_keys | ZIP | +| `core.diagnostics.sysinternals_report` | Sysinternals state and available tool output | deep | subprocess | TXT | +| `core.integration.legacy_code_outputs` | Bounded generated evidence from `CODE` | manual | filesystem_read | BIN | + +All entries default to standard privilege and no remote network access except the packet and network-identity +declarations shown above. A missing Windows facility is normally recorded as skipped with an actionable reason. diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md new file mode 100644 index 00000000..b607f42a --- /dev/null +++ b/docs/DEVELOPMENT.md @@ -0,0 +1,29 @@ +# Development guide + +## Setup + +Use the repository's managed environment and read `CONTRIBUTING.md`, `SECURITY.md`, and [Contracts](CONTRACTS.md) before +changing engine or collector code. Keep the standard-library-first installer path intact. + +## Change boundaries + +Engine changes and contract changes belong under `logicytics/module/`; shipped +collection belongs under the matching `core//` directory; user extensions belong in `plugins/`; +documentation changes belong in `docs/`. Do not commit collected evidence, caches, credentials, or local manifests. + +## Adding a core collector + +Choose one specialty and exact ID, implement the lifecycle contract, declare all access and limits, add focused tests, +update the catalog and configuration guide if settings are introduced, and verify preflight plus packaging. Core +collectors are automatically discoverable; do not add ad hoc registration tables unless the architecture requires it. + +## Documentation changes + +Use stable concepts instead of release-specific marketing. Every command, field, capability, collector ID, and output +claim should be traceable to live parser, contract, metadata, or test evidence. The wiki workflow publishes the contents +of `docs/` after a default-branch documentation change. + +## Commits + +Use a focused Conventional Commit such as `docs: replace operator manual` or `ci: publish documentation to wiki`. +Inspect `git diff` and `git status` before staging so unrelated work remains untouched. diff --git a/docs/ENGINE.md b/docs/ENGINE.md new file mode 100644 index 00000000..e4c26e47 --- /dev/null +++ b/docs/ENGINE.md @@ -0,0 +1,57 @@ +# How the engine works + +The engine is a staged pipeline: + +```text +request -> configuration -> preflight -> plan -> isolated workers -> artifacts -> manifest -> package +``` + +## 1. Request + +The CLI or Python API creates an immutable `RunRequest`. It contains the profile, exact includes/excludes, plugin and +plugin enablement, worker limit, blocked capabilities, performance policy, rerun source, output policy, and optional post-run +action. Invalid combinations are rejected before collection. + +## 2. Configuration + +`logicytics.yaml` is parsed as a strict mapping. Unknown keys, duplicate keys, unsafe paths, invalid numbers, +unsupported collector settings, and oversized files are rejected. The validated `AppConfig` produces a configuration +fingerprint used to prevent stale validation data from being reused. + +## 3. Preflight + +Discovery walks the source trees, checks filenames and class shape, validates imports and side effects, and runs a +side-effect-free validation worker. A successful cached probe is accepted only when source hash, relative path, +collector kind, interpreter version, contract version, and full configuration identity still match. Invalid candidates +are never cached as successful. + +## 4. Plan + +The planner selects by profile, exact IDs, dependencies, capability policy, platform, privilege, and resource rules. It +orders dependencies before dependants and chooses bounded parallel or serial execution. A plan has a fingerprint; it is +the stable description of what the run intends to do. + +## 5. Worker execution + +The supervisor starts eligible collectors in separate processes. A worker receives a `CollectorContext`, calls +validation/prepare/collect/finalize, and returns a typed `CollectorResult`. Heartbeats and progress are persisted. +Timeouts, memory limits, cancellation, crashes, malformed results, and capability violations become explicit statuses +rather than false success. + +## 6. Artifact publication + +Collectors cannot publish arbitrary files directly. They call `context.artifacts.register_file(...)`. The writer +enforces workspace ownership, declared MIME types, size and file quotas, safe relative paths, SHA-256 calculation, and +evidence kind. Worker scratch space is removed after terminal publication; durable run evidence remains. + +## 7. Manifest and package + +The manifest records run identity, request, plan, timestamps, configuration identity, collector records, artifacts, +failures, progress, and package metadata. Packaging verifies allowlisted paths, checksums, archive members, and the ZIP +sidecar before reporting a successful package. A manifest-only run is valid when `--no-package` is intentional. + +## Status meanings + +Collector statuses are `succeeded`, `partial`, `skipped`, `cancelled`, and `failed`. Run statuses are `planned`, +`running`, `succeeded`, `partial`, `failed`, and `cancelled`. A skipped collector is not the same as a failed collector; +read its summary and errors for the reason. diff --git a/docs/ERRORS.md b/docs/ERRORS.md new file mode 100644 index 00000000..c5b60d67 --- /dev/null +++ b/docs/ERRORS.md @@ -0,0 +1,122 @@ +# Error reference + +This reference explains every user-facing error family emitted by the CLI, planner, preflight, runtime, artifact writer, package publisher, and public API. The final detail can include a Windows error, file path, collector ID, or exception text from the host; preserve that detail in the manifest or sanitized diagnostic because it identifies the exact failing source. + +## Read an error before retrying + +1. Record the command, exit code, and exact rendered message. +2. For a collection, open the referenced `manifest.json`; a non-success run can retain valid registered artifacts. +3. Read the error family below, correct the underlying input, scope, capability, source, or environment condition. +4. Re-run `preflight` or `plan` before re-authorizing collection. Do not bypass a gate by changing a status or deleting diagnostic files. + +## CLI and request errors + +| Rendered error or family | Meaning | Correction | +|----------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------|---------------------------------------------------------------------------------------------| +| `unrecognized arguments` / parser usage | A flag is unsupported, misspelled, or placed after an incompatible command. | Use the supported tables in [Commands](COMMANDS.md). | +| `--profile cannot be combined with a named collection mode` | Two mode selectors were supplied. | Use one of `--profile`, `--mode`, or a legacy mode alias. | +| `--mode cannot be combined with a legacy mode alias` / `legacy collection mode aliases are mutually exclusive` | Conflicting mode selectors were supplied. | Select exactly one mode. | +| `performance checking requires sequential execution` | `--performance-check` was combined with parallel scheduling. | Remove `--parallel`; performance measurement is serial. | +| `sequential execution requires --workers=1` | A sequential request asks for more than one worker. | Omit `--workers` or set it to `1`. | +| `parallel execution requires at least two configured workers` | Parallel execution cannot run with fewer than two workers. | Increase the configured or explicit worker count. | +| `collector execution cannot combine its ID with profile or selection flags` | Direct `collector ID` selection was mixed with profile/include/exclude/plugin selection. | Run only the positional ID, or use `plan`/`run` for a multi-collector request. | +| `--rerun-from requires at least one explicit --include collector ID` | A rerun has no precise source selection. | Supply one or more `--include` IDs from the original finalized plan. | +| `original run manifest cannot be loaded` | The supplied rerun path is unreadable, not JSON, or not a manifest. | Use the prior run directory or its `manifest.json`; preserve the original file. | +| `original run manifest must ...` / `uses unsupported schema_version` | The rerun manifest is malformed, unfinished, incompatible, or lacks a valid resolved plan. | Rerun from a finalized compatible v4 manifest. | +| `rerun collectors were not present in the original run` | An include ID was not part of the prior resolved plan. | Choose only IDs recorded by that manifest, or make a new reviewed request. | +| `must run inside a virtual environment` | The normal CLI was invoked with system Python. | Run the installer, then use `.\.venv\Scripts\python.exe -m logicytics ...`. | +| `Command cancelled` / exit `130` | The process received an interrupt. | Locate the manifest, inspect finalized records, and retry only if scope remains authorized. | + +## Configuration errors + +Configuration failures are `PlanError` conditions and exit before collection. They protect a single authoritative YAML configuration. + +| Error family | Meaning | Correction | +|-----------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------| +| `configuration file ...` / `cannot read configuration` | The selected file is missing, unreadable, oversized, or not valid UTF-8. | Use an accessible YAML file below the size limit. | +| `configuration must contain ...` / `unknown configuration key` | The root or nested mapping has the wrong shape or an unsupported key. | Compare the file against [Configuration](CONFIGURATION.md); remove typos rather than relying on ignored keys. | +| `duplicate key` | YAML repeats a key, making intent ambiguous. | Keep one value for the key. | +| `unsupported configuration schema_version` | The file is not schema version `4`. | Migrate deliberately to the documented schema. | +| `must be ...` / `must not ...` for runtime, logging, interaction, or maintenance fields | A value has the wrong type, range, label, URL, digest, or path form. | Use the field's documented type and bound; do not coerce booleans into numbers. | +| `path escapes project` / `unsafe path` | A configured relative or absolute path leaves its approved root or uses a forbidden location. | Choose a permitted project-relative output, manifest, dump, or temporary path. | +| `collector settings ...` | A core collector received unknown, missing, or out-of-range settings. | Use only that collector's settings and bounds from [Configuration](CONFIGURATION.md). | + +## Preflight errors + +Preflight combines static source checks and an isolated runtime metadata probe. A core failure always blocks. An invalid plugin is quarantined until explicitly selected or enabled, at which point it blocks. + +| Diagnostic family | Meaning | Correction | +|------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------| +| `collector preflight failed` | One or more selected/core candidates are invalid. The appended path and diagnostics identify them. | Fix each named source, then run `preflight --invalidate-cache`. | +| `unsafe filename`, `forbidden top-level statement`, `import-time ...` | Source layout or import behavior would perform work before isolation. | Use a regular lowercase Python module and move work into lifecycle methods. | +| `collector class requires a docstring` / `metadata must construct one CollectorMetadata` | The source does not expose the required inspectable class shape. | Follow [Contracts](CONTRACTS.md) and [Plugin Authoring](PLUGIN_AUTHORING.md). | +| `plugin metadata must explicitly declare ...` | Plugin metadata omits a required security, quota, or output field. | Declare every named field honestly in `metadata()`. | +| `invalid validation response` / `collector contract version is unsupported` | The isolated probe returned invalid metadata or an unsupported contract. | Correct `metadata()` and use minimum contract version `4.0`. | +| `collector ID must start with its owner kind` / location mismatch | Metadata ID does not match a core/plugin module's ownership. | Rename the module, folder, or ID so they agree. | +| `preflight cache ...` | Cached validation state is unreadable or stale. | Run with `--invalidate-cache`; never edit the cache to force validity. | + +## Planning and policy errors + +| Error family | Meaning | Correction | +|------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------|---------------------------------------------------------------------------------| +| `unknown collection profile` | The request profile is not one of the built-in profiles. | Use the modes shown by `--modes`. | +| `max_workers must be positive` / `request max_workers ...` | Worker count is outside its supported positive bound. | Set a valid count and honor sequential/parallel constraints. | +| `requested collectors are unavailable` | An include ID was not discovered as valid. | Check the exact ID, plugin enablement, preflight result, and configuration. | +| `selected collector depends on ...` | A dependency is unavailable, excluded, or forms a cycle. | Restore the dependency, remove the exclusion, or correct metadata dependencies. | +| `plugin dependency ... must be explicitly selected or plugins enabled` | A selected collector depends on an opt-in plugin that was not enabled. | Review and enable plugins, or explicitly include the dependency. | +| `selected collector policy validation failed` | The detailed rows identify blocked capability, unsupported platform, or offline-policy conflict. | Reduce scope, remove the block only when authorized, or use a supported host. | +| `selected collectors require an administrator account` | Metadata requires elevation and the process is not elevated. | Use the normal approved administrative process, or exclude the collector. | +| `authorization acknowledgement is required` | A state-changing collection request omitted explicit authorization acknowledgement. | Review the plan, then add `--acknowledge-authorization` if authorized. | + +## Metadata and contract errors + +These `ValueError` messages are emitted during metadata, request, artifact, estimate, or result construction. They identify a programmer or plugin-authoring contract failure, not a condition to suppress. + +| Error family | Meaning | Correction | +|--------------------------------------------------------------------------------|---------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------------| +| `metadata ... must be a non-empty single-line string` | Required identity text is missing or contains a newline. | Provide one non-empty single-line value. | +| `metadata id/specialty/version/minimum_contract_version has an invalid schema` | Identity/version format is invalid. | Use supported core/plugin ID, specialty, semantic version, and `major.minor` contract format. | +| `metadata ... must be a tuple ...` / `must not contain duplicates` | A metadata sequence has the wrong type, invalid labels, or duplicates. | Use immutable tuples of valid unique values. | +| `sensitive collectors must not belong to the minimal profile` | Sensitive evidence was placed in the smallest profile. | Remove `minimal` from the profile list. | +| `standard collectors may contain only routine ... categories` | Standard mode was assigned sensitive categories outside its approved set. | Move it to a narrower/deeper explicit profile. | +| `privilege_level must match ...` / `network_access must declare ...` | Access declarations contradict their capability. | Align elevation/network values with capability declarations. | +| `maximum_artifact_bytes must not exceed maximum_output_bytes` | An individual artifact could exceed total collector output. | Lower the artifact ceiling or increase the total bound responsibly. | +| `request include and exclude selections must not overlap` | The same collector was simultaneously selected and removed. | Keep it on one side only. | +| `request ... must be boolean` / `must be a tuple of ...` | A programmatic `RunRequest` uses wrong types. | Construct the immutable request with documented types. | +| `artifact ...` | Artifact identity, hash, timestamp, MIME type, owner, path, or transformations are invalid. | Publish through the supplied writer; do not manually forge artifacts. | +| `collector result ...` / `estimated_...` | A collector returned an invalid status, summary, result shape, or estimate. | Return a typed constructor result with non-empty summary and valid metrics. | + +## Runtime, artifact, and capability errors + +| Error family | Meaning | Correction | +|----------------------------------------------------|-------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------| +| `CAPABILITY_BLOCKED` | A declared capability is prohibited by request or configuration policy. | Keep the collector out of the plan or remove the block only after authorization review. | +| `CAPABILITY_UNDECLARED` | Collector code attempted host access it did not declare. | Correct metadata and implementation; do not add capability declarations merely to silence the failure. | +| `artifact source ...` / `artifact ... escapes ...` | A source, destination, symlink, or owner path violates workspace confinement. | Produce files under the provided workspace and register them through the writer. | +| `artifact source changed during registration` | The file changed while being copied and hashed. | Finish writing before registration; use an atomic collector-local staging pattern. | +| `artifact ... exceeds ... limit` | Per-file, artifact-count, reserved-output, or run-output quota was exceeded. | Reduce collection or define a justified bounded output policy. | +| `collector worker crashed` | The isolated worker raised unexpectedly. | Read the collector record's error and trace; fix the source or prerequisite, then preflight. | +| `collector timed out`, `memory limit`, `cancelled` | Runtime containment ended the work. | Reduce scope, use sequential diagnosis, and correct the bounded collector behavior. | +| `unsupported collector execution type` | Internal worker payload was not a supported collector payload. | This is an engine bug; retain diagnostics and report it with sanitized context. | + +## Manifest, package, and API errors + +| Error family | Meaning | Correction | +|--------------------------------------------------------------------------------------|------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------------------------| +| `run manifest is missing schema_version` / `unsupported run manifest schema_version` | A caller tried to read an incompatible or malformed manifest. | Use a finalized v4 manifest. | +| `no output directory exists for run` | The requested run ID has no matching retained output directory. | Verify the ID and retention location; do not create a replacement directory. | +| `artifact is not registered` / `artifact exceeds maximum read size` | Public API artifact read is not manifest-backed or exceeds its bounded read limit. | Read only catalogued artifacts and request an appropriate bounded limit. | +| `package contents do not match ...` / `package integrity verification failed` | ZIP members, checksums, or catalog differ from the finalized manifest. | Treat the package as invalid; preserve it for investigation and rerun packaging from intact evidence. | +| `package metadata ... invalid` / `packaged metadata does not match ...` | Package metadata cannot be decoded or differs from the manifest. | Do not distribute it; investigate tampering or a packaging defect. | +| `requested performance report is missing` / `performance report escapes ...` | A performance-mode run lacks its required report or the path is unsafe. | Preserve the run and investigate runtime output ownership before retrying. | +| `no unique output fingerprint prefix is available` | Existing output paths exhaust all fingerprint prefixes. | Preserve existing evidence, choose a new output root, and investigate collisions. | + +## Maintenance and update errors + +| Error family | Meaning | Correction | +|---------------------------------------------------------------------------------------|----------------------------------------------------------------------------------|--------------------------------------------------------------------------------------| +| `git is unavailable`, `not a repository`, `origin ...`, `remote ...` | Update/development checks cannot establish a valid checkout or reachable remote. | Install Git, use the intended checkout, configure origin, or resolve network access. | +| `--write-manifest requires --next-version` | A persistent integrity-manifest write has no explicit semantic target. | Provide `--next-version X.Y.Z` or use interactive review. | +| `unsupported new-window action` / `new command windows are supported only on Windows` | An update launch action is not allowlisted or host support is absent. | Use `preflight`, `debug`, or `dev` with `--new-window` on Windows. | + +When an error is not represented verbatim above, keep its exact message and the manifest/debug JSON. Host API, optional Windows tool, and collector-specific error text is intentionally preserved as the final detail needed to diagnose that source. diff --git a/docs/EVIDENCE_REVIEW.md b/docs/EVIDENCE_REVIEW.md new file mode 100644 index 00000000..f4c920a7 --- /dev/null +++ b/docs/EVIDENCE_REVIEW.md @@ -0,0 +1,52 @@ +# Evidence review and handoff + +Treat every completed run directory and ZIP package as potentially confidential. The engine records evidence provenance +and integrity metadata, but the operator still decides whether the scope was appropriate, whether the output is complete +enough for the question, and who may receive it. + +## Find the authoritative run record + +Runs are stored below `runtime.output_root`, normally `output/data`. A completed run owns a fingerprinted directory +containing `manifest.json`, collector artifacts, logs, and reports. Packaging, when enabled, writes a ZIP and a SHA-256 +sidecar under sibling `zip` and `hashes` directories. The exact layout and retention distinction are documented +in [Results](OUTPUTS.md). + +The manifest is the authority for interpreting a run. Do not infer full success from a console message, a populated +directory, or the existence of a ZIP. + +## Review sequence + +Use this order before analysis, export, or handoff: + +1. Confirm `run_id`, overall `status`, request, and plan match the approved collection scope. +2. Count collector records by status. Read every `failed`, `partial`, `skipped`, and cancelled record before relying on + an aggregate conclusion. +3. For each non-success record, capture the summary, structured error details, remediation, and retry-safe indication. + Decide whether it changes the answer to the original question. +4. Follow each registered artifact entry to its path, MIME type, byte count, timestamp, evidence kind, and SHA-256. Use + the manifest catalog instead of browsing arbitrary files in the workspace. +5. If a package was requested, verify the package and its hash metadata before transporting it. Retain the manifest with + the package so its contents remain interpretable later. + +The engine may keep useful evidence from a partial run. Preserve that context; do not discard a partial result merely +because one independent collector failed. Equally, do not call a partial result complete when the failed collector was +needed for the question. + +## Handoff checklist + +Before sharing, record the case or change reference, operator, date and time, host or asset identifier under the +applicable policy, command or plan used, configuration identity, manifest path, package path, and hash value. State the +overall status and each relevant limitation. Keep this record outside public issues and source control. + +Share only the smallest approved artifact set. Raw evidence may contain account names, SIDs, hostnames, paths, +addresses, browser data, saved Wi-Fi material, or private keys. Sanitize a diagnostic excerpt before using it in a +ticket. Never attach an entire `output` tree, raw ZIP, credentials, or key material to a public issue. + +## Retention and deletion + +Completed run evidence, reports, manifests, packages, hashes, and run logs stay together until explicit user removal. +`logging.retention_days` controls application-log rotation only; it does not silently expire run evidence. Follow the +governing retention policy and confirm that the manifest, ZIP, sidecar, and required logs are no longer needed before +removing a run. + +For artifact type conventions, use [Formats](FORMATS.md). For privacy and scope requirements, use [Safety](SAFETY.md). diff --git a/docs/FLOW_MATRIX.md b/docs/FLOW_MATRIX.md new file mode 100644 index 00000000..25c0b69f --- /dev/null +++ b/docs/FLOW_MATRIX.md @@ -0,0 +1,31 @@ +# Execution flow checklist + +| Flow | How to exercise it | +|-----------------------------|-------------------------------------------------------------------------------------| +| Preflight | `python -m logicytics preflight` | +| Plan without collection | `python -m logicytics plan --mode standard` | +| One collector | `python -m logicytics collector ID --acknowledge-authorization` | +| Bounded parallel run | `python -m logicytics run --mode balanced --parallel --acknowledge-authorization` | +| Deterministic run | `python -m logicytics run --mode standard --sequential --acknowledge-authorization` | +| Optional extensions | Add `--plugins` after reviewing source and metadata | +| Performance measurement | Add `--performance-check` to `run` | +| Manifest-only output | Add `--no-package` to `run` or `collector` | +| Rerun selected IDs | `run --include ID --rerun-from PATH` | +| Cancellation | Interrupt an active run and inspect its manifest | +| Permission/optional feature | Inspect the skipped record and remediation | +| Maintenance | `debug`, `update`, or `dev` | + +The executable regression anchors are: + +- `test_typed_mode_registry_maps_every_user_mode_and_legacy_alias` +- `test_explicit_execution_modes_control_isolated_worker_overlap` +- `test_performance_report_is_finalized_before_automatic_packaging` +- `test_dev_writes_explicit_manifest_and_debug_persists_diagnostics` +- `test_update_can_explicitly_launch_an_allowlisted_action_in_a_new_window` +- `test_semantic_flag_matching_history_usage_and_graph_are_local_and_opt_in` +- `test_post_run_actions_are_typed_exclusive_and_require_verified_packaging` +- `test_cancelled_run_writes_a_recoverable_package_and_manifest` +- `test_access_denied_is_skipped_not_failed` + +The automated suite is the authoritative regression evidence for these flows. Use [Verification](VERIFICATION.md) for +the standard commands. diff --git a/docs/FORMATS.md b/docs/FORMATS.md new file mode 100644 index 00000000..7e3305c5 --- /dev/null +++ b/docs/FORMATS.md @@ -0,0 +1,41 @@ +# Reading generated values + +Start with the run's `manifest.json`. It is the index of truth: use its collector IDs and artifact records to find files +rather than assuming a filename exists. + +## Common fields + +An artifact has an ID, collector ID, POSIX-style run-relative path, name, MIME type, evidence kind (`raw` or `derived`), +byte size, collection timestamp, transformations, and SHA-256. JSON values may contain `null` when Windows did not +provide a value. Text and CSV should be read as UTF-8 where possible; preserve raw bytes when a file is marked binary. + +## Format guidance + +- JSON: parse as a mapping or array; retain unknown fields for forward compatibility. +- CSV: use a header-aware reader and do not split on commas manually. +- HTML: open locally and treat it as untrusted data; it may contain host-derived strings. +- XML: parse with a safe XML parser and avoid resolving external entities. +- ZIP: verify the adjacent SHA-256 sidecar before extraction; extract to a new directory and reject path traversal. +- Graphviz DOT: render only in a viewer you trust and treat labels as untrusted text. +- Plain text: preserve line breaks; it may be a command report rather than a key/value document. + +## Using values in a script + +The public API can load a validated `RunSnapshot` and read bounded artifacts. Prefer the API to globbing the output tree +because it revalidates manifest structure and collector ownership. + +```python +from pathlib import Path +from logicytics import open_artifact, query_run + +snapshot = query_run(Path("."), "run-0123456789abcdef0123456789abcdef") +for item in snapshot.artifacts: + print(item.collector_id, item.media_type, item.relative_path) + +data = open_artifact(Path("."), snapshot.run_id, "artifact.0123456789abcdef0123456789abcdef") +with data as stream: + first_bytes = stream.read(1024) +``` + +Keep the run ID and artifact ID from the manifest. Do not trust a path supplied by a user until it has been checked +against the manifest and the expected output root. diff --git a/docs/GETTING_STARTED.md b/docs/GETTING_STARTED.md new file mode 100644 index 00000000..7df63c0c --- /dev/null +++ b/docs/GETTING_STARTED.md @@ -0,0 +1,84 @@ +# Getting started + +This walkthrough makes one ordinary, authorized Windows inventory collection. It does not repair the host or remove +data. It does create evidence files, which can contain hostnames, account information, network details, and other +sensitive local state. If that is not within your scope, stop here and read [Safety](SAFETY.md). + +## 1. Prepare the repository + +Open PowerShell at the repository root. The installer is the only Logicytics entry point intended to run with a system +Python; it creates or reuses `.venv` and writes the default configuration only when it is absent. + +```powershell +python -m logicytics.cli.installer +``` + +Use the managed interpreter for every remaining command. Activation is optional: the explicit path is more reproducible +and also works when PowerShell execution policy blocks `Activate.ps1`. + +```powershell +python -m logicytics --help +python -m logicytics preflight +``` + +`preflight` is read-only with respect to collection: it validates sources and prerequisites before a collector can be +planned. An unavailable optional Windows facility is normally reported as unavailable or skipped; an invalid collector, +unsafe extension, or configuration error must be fixed before proceeding. + +## 2. Build and review a plan + +The plan is the authorization checkpoint. It resolves a mode into exact collector IDs, dependencies, capabilities, and +resource limits without starting workers or gathering evidence. + +```powershell +python -m logicytics plan --mode standard +``` + +Read the plan and answer these questions: + +- Are the selected IDs within the agreed scope? The complete descriptions are + in [Core Collector Catalog](CORE_COLLECTORS.md). +- Do any declared capabilities include sensitive files, browser data, private keys, elevation, packet capture, or + network access? +- Is `runtime.output_root` on storage approved for evidence? See [Configuration](CONFIGURATION.md). +- Should a source be removed, or should a capability be blocked across the request? + +For example, this preserves a standard plan while preventing collectors that need sensitive files or packet capture from +entering it: + +```powershell +python -m logicytics plan --mode standard ` + --block-capability sensitive_files --block-capability packet_capture +``` + +If the plan is broader than necessary, refine it before collection. Selectors are repeatable, and an exact collector run +is often a better diagnostic than a large profile: + +```powershell +python -m logicytics collector core.system.system_info ` + --acknowledge-authorization +``` + +## 3. Run with explicit authorization + +The acknowledgement confirms that the operator reviewed the requested evidence; it is not an automatic permission grant. +Use it only after the preceding review. + +```powershell +python -m logicytics run --mode standard ` + --acknowledge-authorization +``` + +Use `--sequential` for a deterministic troubleshooting run, `--no-package` when you need only the manifest-backed run +directory, and `--interactive` when a console window should remain open at the end. Do not add `--plugins` just to make +a command work: it opts into reviewed user-owned collector code. + +## 4. Decide whether the result answers the question + +Start with the run's `manifest.json`, not the ZIP file. Check the overall status, then every skipped, partial, failed, +or cancelled collector record and its actionable details. A partial run may contain valid artifacts; a ZIP can exist +even when not every collector succeeded. + +The console gives the manifest location and, when packaging succeeded, the ZIP and SHA-256 sidecar. +Follow [Evidence Review](EVIDENCE_REVIEW.md) for the review and handoff workflow, +or [Troubleshooting](TROUBLESHOOTING.md) when a result is not usable. diff --git a/docs/INSTALLATION.md b/docs/INSTALLATION.md new file mode 100644 index 00000000..4dbb5768 --- /dev/null +++ b/docs/INSTALLATION.md @@ -0,0 +1,77 @@ +# Installation and environment + +Logicytics is developed and exercised as a Windows-focused application. The v4 runtime has no required third-party +Python dependency, but it deliberately uses a repository-local virtual environment so commands run with a known +interpreter and do not depend on unrelated packages installed for another project. + +## Requirements + +- A local checkout of this repository. +- Python 3.11 or later, available as `python` for the bootstrap command. +- Windows for live collector behavior and Windows integration checks. +- Enough trusted disk space for the configured output location and any ZIP package. Evidence output is not a cache; plan + its storage accordingly. + +Git is needed for the `update` and `dev` maintenance actions, but not for a normal already-checked-out collection run. + +## Bootstrap or repair the environment + +From the repository root, run the installer: + +```powershell +python -m logicytics.cli.installer +``` + +It creates `.venv` if it does not exist, otherwise it reuses it. It also creates the root `logicytics.yaml` template if +it is absent. To intentionally replace that template, use the explicit destructive configuration option: + +```powershell +python -m logicytics.cli.installer --overwrite-config +``` + +Back up or version-control a customized configuration before using `--overwrite-config`; it replaces the root template. + +## Run the supported interpreter + +Every normal application command requires the managed environment. +The most reliable form is the explicit interpreter path: + +```powershell +.\.venv\Scripts\python.exe -m logicytics preflight +``` + +You may activate the environment for an interactive shell: + +```powershell +.\.venv\Scripts\Activate.ps1 +python -m logicytics preflight +``` + +If activation is blocked by PowerShell policy, do not weaken the policy just for this tool. +Continue using `.\.venv\Scripts\python.exe`; it produces the same managed-environment execution. + +## Confirm the installation + +Run the following in order: + +```powershell +python -m logicytics --help +python -m logicytics preflight +``` + +The first command verifies CLI invocation. +The second verifies the selected collector sources and prerequisites without collecting evidence. +Resolve any configuration or invalid-source result before attempting a plan or run. + +## Common environment mistakes + +| Symptom | Cause and correction | +|-----------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------| +| "Must run inside a virtual environment" | Use `.\.venv\Scripts\python.exe` for the normal CLI or activate via `.\.venv\Scripts\Activate.ps1`. Only the installer runs from the system interpreter. | +| `python` is not recognized | Install a supported Python version or invoke its known executable for the installer; after bootstrap use the local `.venv` path. | +| Activation is denied | Do not rely on activation. Use the explicit interpreter path. | +| A custom config disappeared | The installer was run with `--overwrite-config`; restore the backed-up configuration and preflight it before use. | +| Preflight finds invalid extension code | Correct or remove the named extension source. Extensions are intentionally not trusted just because their files exist. | + +For configuration fields and safe path rules, read [Configuration](CONFIGURATION.md). +For collection scope after installation, continue to [Getting Started](GETTING_STARTED.md). diff --git a/docs/MIGRATION.md b/docs/MIGRATION.md new file mode 100644 index 00000000..405c6f5f --- /dev/null +++ b/docs/MIGRATION.md @@ -0,0 +1,14 @@ +# Moving from older layouts + +Use `logicytics.yaml` as the single settings source. Execution code lives under `logicytics/module/`, command parsing +under `logicytics/cli/`, and the public API under `logicytics/module/api.py`. + +Older mode flags remain compatibility aliases: `--default` selects `standard`, `--threaded` selects `balanced`, +`--minimal` selects `quick`, and `--depth` selects `thorough`. New scripts should use `--mode` or `--profile`. + +Legacy `CODE/config.ini` is read only as a compatibility input. The explicit `core.integration.legacy_code_outputs` +collector imports approved generated output; the engine does not implicitly scan the repository. Extension support is +limited to reviewed `PluginCollector` modules under `plugins/`. Run evidence is owned by its manifest-backed run folder. There are no global `ACCESS/`, `RUNS/`, +`LOGS/`, or `PACKAGES/` stores. + +Translate supported settings using [Configuration](CONFIGURATION.md), then run `preflight` and `plan` before collecting. diff --git a/docs/OPERATIONS.md b/docs/OPERATIONS.md new file mode 100644 index 00000000..6daafe1f --- /dev/null +++ b/docs/OPERATIONS.md @@ -0,0 +1,116 @@ +# Operations guide + +This guide turns the command reference into an operating procedure. It is not a substitute for organizational +authorization, incident procedures, or evidence retention policy. It helps an authorized operator make the collection +request small, reviewable, and reproducible. + +## Operating model + +Every request follows the same path: + +```text +scope decision -> preflight -> plan review -> explicit authorization -> isolated run -> manifest review -> evidence handoff +``` + +The command-line interface does not infer authorization from a profile. The `--acknowledge-authorization` flag is +required for collection because the final scope decision belongs to the operator. It must follow, not replace, a review +of the plan. + +## Choose the smallest useful request + +Use a mode only when its complete membership fits the question. Inspect the live matrix with: + +```powershell +python -m logicytics --modes +``` + +Then create a non-collecting plan. The five modes are `quick`, `balanced`, `standard`, `offline`, and `thorough`. + +```powershell +python -m logicytics plan --mode standard +``` + +Use a direct collector for a single known question, such as host inventory: + +```powershell +python -m logicytics collector core.system.system_info --acknowledge-authorization +``` + +Use repeated `--include` and `--exclude` when the standard profile is close but not exact. Keep an exported plan or the +terminal transcript with the case notes so another reviewer can reproduce the scope. + +## Review capabilities before authorization + +Capabilities describe access a collector may request. They are not just labels: a requested block prevents matching +collectors from entering the plan. The available CLI capability values are: + +- `filesystem_read`, `filesystem_write`, `registry_read`, and `subprocess` +- `network` and `packet_capture` +- `browser_data`, `sensitive_files`, and `private_keys` +- `elevated_privileges` + +For a restricted collection, request explicit blocks and plan again: + +```powershell +python -m logicytics plan --mode thorough ` + --block-capability browser_data ` + --block-capability sensitive_files ` + --block-capability private_keys ` + --block-capability packet_capture +``` + +Configuration can also define persistent blocks. Invocation blocks add to those configured restrictions; they do not +weaken them. Read [Safety](SAFETY.md) and the catalog entry for every sensitive collector you leave selected. + +## Run and observe + +Run the reviewed request with acknowledgement: + +```powershell +python -m logicytics run --mode standard ` + --acknowledge-authorization +``` + +The runtime gives each collector its own worker process, workspace, temporary directory, event channel, and resource +boundary. An independent collector failure is recorded rather than normally terminating unrelated collectors. That is +why an apparently successful run must still be inspected collector by collector. + +For diagnosis, make concurrency explicit: + +```powershell +python -m logicytics run --mode standard --sequential ` + --acknowledge-authorization +``` + +`--performance-check` performs serial measurement and saves per-collector durations; it is a measurement mode, not a +shortcut. `--no-package` leaves the durable run directory and manifest but intentionally omits the ZIP. `--reboot` and +`--shutdown` are high-impact post-run actions: they are mutually exclusive and only occur after successful package +publication. + +## Extensions and removable media + +Plugins are opt-in because they are user-owned collector code. Include them only after their source, metadata, and +capability declarations have been reviewed: + +```powershell +python -m logicytics preflight --plugins +python -m logicytics plan --mode standard --plugins +``` + +Use `--usb [DRIVE]` only when deliberately targeting a Windows installation on removable storage. The engine rejects +unsafe output, cache, and temporary paths on that target. Confirm the chosen drive and keep output on separately +approved storage before authorizing the run. + +## Handle the outcome + +Exit code is a routing signal, not the full result: + +| Exit code | Meaning | Operator action | +|-----------|--------------------------------------------------------|-----------------------------------------------------------------------------------| +| `0` | Requested command completed successfully | Review the manifest and artifacts before handoff. | +| `1` | Collection completed with a non-success run status | Review failed, partial, skipped, or cancelled records; preserve useful artifacts. | +| `2` | Command, configuration, preflight, or planning failure | Correct the indicated input or source; do not bypass the gate. | +| `130` | Interrupted | Locate the manifest and determine what was safely finalized before retrying. | + +Finish every collection with [Evidence Review](EVIDENCE_REVIEW.md). If an error is not self-explanatory, +follow [Troubleshooting](TROUBLESHOOTING.md) and retain only sanitized diagnostic material for support. diff --git a/docs/OUTPUTS.md b/docs/OUTPUTS.md new file mode 100644 index 00000000..8c26c742 --- /dev/null +++ b/docs/OUTPUTS.md @@ -0,0 +1,36 @@ +# Results and output layout + +Each run is owned by a fingerprinted directory below `runtime.output_root` (normally `output/data`): + +```text +output/data/ + run// + manifest.json + artifacts//... + logs/... + reports/... + zip/.zip + hashes/.zip.sha256 + logs/... +``` + +The prefix starts at eight SHA-256 characters and grows only to resolve a collision. A package contains the manifest, +metadata, reports, logs, and evidence under stable `raw` or `derived` sections. A direct collector or `--no-package` run +may have only the manifest-backed run folder. + +## Manifest reading order + +1. Read `status`, `run_id`, request, and plan. +2. Count collector records by status. +3. Read each failed or skipped record's summary, errors, and actionable failure details. +4. Follow `artifacts` entries to exact files. +5. Verify package and hash metadata before sharing or extracting a ZIP. + +## Stable artifact rules + +Artifacts are collector-owned, use POSIX separators, carry a MIME type and SHA-256, and cannot escape their owner +directory. Typical MIME-to-suffix mappings are `application/json` → `.json`, `text/csv` → `.csv`, `text/html` → `.html`, +`text/plain` → `.txt`, `application/xml` → `.xml`, `application/zip` → `.zip`, and `text/vnd.graphviz` → `.dot`. + +Completed evidence, manifests, packages, hashes, reports, and run logs are retained together until explicit user +removal. Application-log rotation is separate and controlled by `logging.retention_days`. diff --git a/docs/PLUGINS.md b/docs/PLUGINS.md new file mode 100644 index 00000000..a427f783 --- /dev/null +++ b/docs/PLUGINS.md @@ -0,0 +1,37 @@ +# Plugin support + +Plugins are the only supported extension mechanism. A plugin is a typed Python `PluginCollector` stored below `plugins/`; it is never selected by a normal core-only request. Script sidecars, batch, PowerShell, and executable extension paths are not supported. + +## Enable a reviewed plugin + +1. Read its source and declared metadata. +2. Run preflight with plugins enabled. +3. Build a plan and review the exact ID, capabilities, sensitivity, limits, and dependencies. +4. Run only with explicit authorization. + +```powershell +.\.venv\Scripts\python.exe -m logicytics preflight --plugins +.\.venv\Scripts\python.exe -m logicytics plan --mode standard --plugins +.\.venv\Scripts\python.exe -m logicytics run --mode standard --plugins --acknowledge-authorization +``` + +For a sensitive plugin, prefer an exact `--include plugin.name` request and document why the capability set is authorized. A blocked capability always overrides a plugin declaration. + +## Discovery and identity + +Plugin modules live under `plugins/`. Each valid module exposes exactly one `PluginCollector` subclass. A plugin ID begins with `plugin.` and must match its file or folder ownership: a normal `plugins/example.py` exposes `plugin.example`; a `main.py` entry point uses its containing folder name. + +Discovery statically validates the source first, then probes metadata in a short-lived restricted process. Invalid unselected plugins are quarantined so an unrelated optional plugin cannot break normal core collection. A selected or globally enabled invalid plugin blocks the request. + +## Contract requirements + +The complete field-by-field contract is in [Contracts](CONTRACTS.md) and a runnable minimal pattern is in [Plugin Authoring](PLUGIN_AUTHORING.md). In short, a plugin must: + +- have no import-time collection, network, registry, subprocess, or output side effects; +- construct deterministic metadata and explicitly declare access, privilege, network reach, sensitive categories, limits, output MIME types, and contract version; +- perform bounded work only in its private workspace and temporary directory; +- register complete artifacts through `context.artifacts`, never by writing directly to an output or package path; +- check cancellation during long work and return typed `succeeded`, `partial`, `skipped`, `cancelled`, or `failed` outcomes; and +- keep all host access within declared capabilities and supported platform adapters. + +Plugins do not gain authority merely because they are enabled. The runtime applies the same isolated worker, quota, logging, artifact, packaging, cancellation, and capability-policy rules used for core collectors. diff --git a/docs/PLUGIN_AUTHORING.md b/docs/PLUGIN_AUTHORING.md new file mode 100644 index 00000000..a303df0f --- /dev/null +++ b/docs/PLUGIN_AUTHORING.md @@ -0,0 +1,76 @@ +# Plugin authoring + +This guide assumes you can write basic Python. A plugin is a user-owned `PluginCollector` placed under `plugins/` and +enabled with `--plugins`. It is deliberately opt-in and is isolated like a core collector. + +## Minimal plugin + +```python +from logicytics.module.contracts import ( + CollectorMetadata, CollectorResult, CollectorStatus, CollectorContext, + PluginCollector, ValidationResult, ArtifactWriter, +) + +class ExamplePluginCollector(PluginCollector): + @classmethod + def metadata(cls) -> CollectorMetadata: + return CollectorMetadata( + id="plugin.example", + name="Example plugin", + version="1.0.0", + specialty="integration", + description="Writes a bounded example report.", + author="Your name", + capabilities=(), + output_media_types=("text/plain",), + default_profiles=("standard",), + timeout_seconds=30, + maximum_output_bytes=1024 * 1024, + maximum_artifact_files=5, + ) + + def validate(self, context: CollectorContext) -> ValidationResult: + return ValidationResult(True) + + def collect(self, context: CollectorContext) -> CollectorResult: + report = context.workspace / "report.txt" + report.write_text("hello from the plugin\n", encoding="utf-8") + artifact = context.artifacts.register_file(report, media_type="text/plain") + return CollectorResult(CollectorStatus.SUCCEEDED, "Example report created", (artifact,)) + + def cleanup(self, context: CollectorContext) -> None: + return None +``` + +The unused `ArtifactWriter` import in the conceptual example may be removed. Keep the actual module self-contained and +import only supported contracts and adapters. + +## Metadata checklist + +Use an ID beginning with `plugin.`; a lowercase name, semantic version, supported specialty, one-line description, +author, output MIME types, profiles, capabilities, sensible timeout and quotas. Declare network, sensitive data, +privilege, dependencies, retry policy, parallel safety, and resource class honestly. Metadata must be deterministic and +side-effect-free. + +## Implementation checklist + +1. Do not perform work while the module is imported. +2. Implement `metadata`, `validate`, `collect`, and `cleanup`; add `prepare`, `finalize`, `estimate`, or `dependencies` + only when useful. +3. Write only inside `context.workspace` or the provided temporary directory. +4. Publish only through `register_file`. +5. Bound loops, files, bytes, subprocess time, and memory. +6. Check `context.is_cancelled` during long work and report progress counters. +7. Return typed results for expected failure and partial states. +8. Test on a machine without optional Windows facilities. + +## Enable and test + +```powershell +python -m logicytics preflight --plugins +python -m logicytics plan --mode standard --plugins +python -m logicytics run --mode standard --plugins --acknowledge-authorization +``` + +If the plugin is sensitive, do not put it in a routine profile. Use an exact `--include plugin.example` selection and +document the authorization decision. diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 00000000..10e9b772 --- /dev/null +++ b/docs/README.md @@ -0,0 +1,58 @@ +# Logicytics documentation + +Logicytics is a Windows-focused, run-oriented evidence collector. It plans a bounded request, gives every selected +collector an isolated workspace, registers the resulting files, and records the outcome in a durable manifest. This +manual is for operators who need a defensible collection workflow and developers who need to preserve the engine's +contracts. + +Start with [Installation](INSTALLATION.md), then follow [Getting Started](GETTING_STARTED.md). For an actual +investigation or administrative collection, keep [Operations](OPERATIONS.md) open beside the terminal and finish +with [Evidence Review](EVIDENCE_REVIEW.md). + +## Choose a path + +| Goal | Read | +|-----------------------------------------------|------------------------------------------------------------------------------------------------| +| I have never used the tool | [Installation](INSTALLATION.md), [Getting Started](GETTING_STARTED.md), [Safety](SAFETY.md) | +| I need to run a collection responsibly | [Operations](OPERATIONS.md), [Safety](SAFETY.md), [Core Collector Catalog](CORE_COLLECTORS.md) | +| I need to review or hand over evidence | [Evidence Review](EVIDENCE_REVIEW.md), [Results](OUTPUTS.md), [Formats](FORMATS.md) | +| I need every command | [Command Reference](COMMANDS.md) | +| I want to understand the engine | [Engine](ENGINE.md), [Architecture](ARCHITECTURE.md) | +| I need to change settings | [Configuration](CONFIGURATION.md) | +| I need to find an evidence source | [Core Collector Catalog](CORE_COLLECTORS.md) | +| I need to read JSON, CSV, HTML, or ZIP output | [Results](OUTPUTS.md) and [Formats](FORMATS.md) | +| I want to run a plugin | [Plugins](PLUGINS.md) | +| I want to write a collector | [Plugin Authoring](PLUGIN_AUTHORING.md), [Contracts](CONTRACTS.md) | +| Something went wrong | [Troubleshooting](TROUBLESHOOTING.md), [Error Reference](ERRORS.md) | +| I am maintaining the repository | [Development](DEVELOPMENT.md), [Verification](VERIFICATION.md) | + +## Safety rule of thumb + +Logicytics can collect sensitive local evidence. Only run it on systems and data you are authorized to examine. Start +with `preflight`, review the plan, use the smallest profile that answers the question, and inspect the selected +capabilities before authorizing a run. + +## A normal operator workflow + +1. Install or repair the managed environment with the installer. +2. Run `preflight`; resolve invalid sources before collecting. +3. Create a `plan` for the exact mode and selectors you intend to use. +4. Review the collector IDs, declared capabilities, output location, and any sensitive sources. Add exclusions or + `--block-capability` where needed. +5. Run only after adding `--acknowledge-authorization` deliberately. +6. Read `manifest.json` before relying on or sharing the output. A package is evidence delivery, not a success signal by + itself. + +The engine intentionally records skipped, partial, failed, and cancelled work instead of silently treating it as +success. Those statuses are useful evidence: read them before retrying or expanding scope. + +## Documentation conventions + +- A **collector ID** is the exact dotted identifier such as `core.system.system_info`. +- A **profile** selects a documented group of collectors; `--include` and `--exclude` refine it. +- A **capability** is an access category a collector must declare before it can request that access. +- A **run** is one planned, isolated execution with a manifest and optional package. +- Paths in examples are Windows paths. Keep the managed interpreter path exactly as shown when using PowerShell. + +The repository workflow publishes this directory to the GitHub repository wiki after documentation changes land on the +default branch. diff --git a/docs/SAFETY.md b/docs/SAFETY.md new file mode 100644 index 00000000..6e790fbc --- /dev/null +++ b/docs/SAFETY.md @@ -0,0 +1,50 @@ +# Safety, privacy, and authorization + +## Consent and scope + +You are responsible for authorization. Do not collect another person's machine, private keys, saved Wi-Fi keys, browser +profiles, or sensitive files without a clear legal and organizational basis. Treat the run directory and ZIP package as +potentially confidential. + +Before a run: + +1. Read the plan. +2. Check the collector IDs and profile. +3. Check declared capabilities and sensitive-data categories. +4. Check output location and available disk space. +5. Add `--block-capability` for access you do not want used. +6. Add `--acknowledge-authorization` only after those checks. + +## Isolation model + +Every selected collector gets an independent worker process, a private workspace, a private temporary directory, an +event channel, a cancellation boundary, and declared resource limits. One failure is recorded for that collector and +does not normally stop independent collectors. The supervisor terminates only the worker process tree it owns when a +timeout or memory limit is exceeded. + +Core collectors are discovered from `core/` and selected by profiles. Plugin collectors in `plugins/` never run unless +explicitly enabled. A plugin must pass static and runtime preflight before it can enter a plan. + +## Capability gates + +Capabilities include `filesystem_read`, `filesystem_write`, `registry_read`, `subprocess`, `network`, `packet_capture`, +`browser_data`, `sensitive_files`, `private_keys`, and `elevated_privileges`. A declaration describes what a collector +may need; it does not grant permission to exceed the worker boundary. A blocked capability wins over a declaration. + +## Data handling + +Application messages are redacted before they reach console or file sinks. Evidence itself is not a log message: it is +written through the artifact writer and recorded in the manifest with path, MIME type, size, timestamp, and SHA-256. Do +not upload raw evidence to an issue. Sanitize hostnames, usernames, paths, addresses, account identifiers, tokens, and +private data before sharing diagnostics. + +The engine does not silently expire completed run evidence. Delete a run only after confirming that its manifest, +package, hash, and logs are no longer needed. + +## High-impact options + +- `--reboot` and `--shutdown` are mutually exclusive and are scheduled only after durable package publication. +- `--performance-check` forces serial measurement and is intended for measurement, not fastest collection. +- `--plugins` executes additional user-owned collector code. +- `--usb` changes the source Windows installation and rejects unsafe output/cache/temp placement on that disk. +- `--update --apply` performs `git pull`; use it only when repository state and remote are understood. diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md new file mode 100644 index 00000000..0e8da987 --- /dev/null +++ b/docs/TROUBLESHOOTING.md @@ -0,0 +1,88 @@ +# Troubleshooting + +Start with the command's exit code and the run `manifest.json`. A console summary is useful for navigation, but the +manifest records the plan, terminal statuses, artifact catalog, and actionable collector-level failure details. + +## The CLI says it must run in a virtual environment + +The normal CLI deliberately rejects a system interpreter. Bootstrap or repair the environment with the installer, then +use the explicit managed interpreter: + +```powershell +python -m logicytics.cli.installer # or .\.venv\Scripts\Activate.ps1 to activate if already installed +python -m logicytics preflight +``` + +If PowerShell blocks `.\.venv\Scripts\Activate.ps1`, activation is optional. +Keep using `.\.venv\Scripts\python.exe` rather than changing execution policy. + +## Preflight reports invalid collectors or extensions + +Read the named source and diagnostic row. Then run a fresh validation after the source is corrected: + +```powershell +python -m logicytics preflight --invalidate-cache +``` + +Do not bypass preflight or manually mark a source valid. Syntax and import errors, unsafe names, missing or +nondeterministic metadata, unsupported platforms, undeclared capabilities, and import-time side effects are contract +violations. An optional Windows facility that is merely unavailable should be recorded distinctly from an invalid +source. + +## The plan is empty or a collector is not selected + +First confirm the exact dotted ID in [Core Collector Catalog](CORE_COLLECTORS.md) and inspect the selected mode with +`--modes`. Then rebuild the plan with the same mode, profile, includes, excludes, extension switches, and capability +blocks as the intended run. A collector can be excluded by a selector, not belong to the mode, be invalid in preflight, +or require a capability you blocked. + +Plugins require `--plugins`. Their absence from a normal core-only plan is intentional. Review their source and metadata +before enabling them. + +## A collector is skipped, partial, failed, or canceled + +Read its manifest record's summary, structured error details, remediation, and retry-safe indication. Skipped commonly +means a required command, optional Windows feature, device, permission, or prerequisite was unavailable. Partial means +some registered artifacts may still be valuable. A failed or canceled collector does not normally erase independent +collector results. + +To narrow a reproducible problem, run the exact collector or use sequential execution: + +```powershell +python -m logicytics collector core.system.system_info ` + --acknowledge-authorization + +python -m logicytics run --mode standard --sequential ` + --acknowledge-authorization +``` + +Do not retry a sensitive or high-impact collector until its scope is still authorized. Elevation, if required by the +collector, must be granted through your normal administrative process. + +## Nothing was written or packaging is absent + +Verify the configured `runtime.output_root`, project permissions, free disk space, and `maximum_run_output_bytes`. +Relative output paths are kept inside the project and unsafe paths are rejected. A `--no-package` request intentionally +omits a ZIP; the manifest-backed run directory remains the result. + +Use the diagnostic command for a bounded environment report: + +```powershell +python -m logicytics debug +``` + +Inspect the manifest before deleting or rerunning anything. Packaging is not a success proxy: a ZIP can coexist with +partial results. + +## Timeout, memory, or output limits + +The runtime records the termination reason. Reduce the request to the smallest needed collector, make scheduling +deterministic with `--sequential`, and inspect the collector's documented bounds. Adjust collector settings only when +the larger collection is authorized and the destination can safely hold the output. Do not raise limits blindly for a +source that may traverse, copy, or expose sensitive data. + +## Safe support material + +Use `debug` and a sanitized manifest excerpt. Remove credentials, key material, browser files, usernames, hostnames, +paths, addresses, SIDs, and raw evidence. Never attach an entire output directory or package to a public issue. +See [Evidence Review](EVIDENCE_REVIEW.md) for the handoff checklist. diff --git a/docs/VERIFICATION.md b/docs/VERIFICATION.md new file mode 100644 index 00000000..c47729ef --- /dev/null +++ b/docs/VERIFICATION.md @@ -0,0 +1,18 @@ +# Verification guide + +Run from the repository root with the managed interpreter: + +```powershell +python -m unittest discover -v +python -m compileall -q logicytics core tests +python -m logicytics preflight +git diff --check +``` + +For a documentation-only change, at minimum run the documentation tests, compile check, and diff check. For engine or +collector changes, run the full suite and preflight. For Windows behavior, run the relevant integration tests on the +target host; a non-Windows pass is not equivalent evidence. + +The documentation tests verify the configuration example, guide links, collector specialty coverage, modes, output MIME +types, and required entry points. The workflow itself should be reviewed as YAML and tested by a real Actions run after +it is enabled. diff --git a/logicytics.yaml b/logicytics.yaml new file mode 100644 index 00000000..9b5a10e5 --- /dev/null +++ b/logicytics.yaml @@ -0,0 +1,28 @@ +# Logicytics user configuration +schema_version: 4 +runtime: + output_root: output/data + default_max_workers: 4 + maximum_workers: 16 + package_completed_runs: true + temporary_directory: project +interaction: + history_enabled: true + similarity_threshold: 0.55 + model_name: stdlib-sequence-matcher + model_debug: false +maintenance: + local_manifest_path: project.manifest.json + minimum_python: "3.11" + recommended_python: "3.11" + sysinternals_enabled: true + sysinternals_download_url: https://download.sysinternals.com/files/SysinternalsSuite.zip +logging: + level: INFO + console_enabled: true + color_enabled: true + file_enabled: true + maximum_bytes: 4194304 + delete_previous: true + retention_days: 30 +collectors: { } diff --git a/logicytics/.gitkeep b/logicytics/.gitkeep new file mode 100644 index 00000000..e69de29b diff --git a/logicytics/__init__.py b/logicytics/__init__.py new file mode 100644 index 00000000..a38da61d --- /dev/null +++ b/logicytics/__init__.py @@ -0,0 +1,108 @@ +"""Logicytics v4 public contracts and lazily loaded application API.""" + +import importlib +import sys +from typing import TYPE_CHECKING + +from logicytics.module.contracts import ( + CONTRACT_VERSION, + Capability, + CollectionEstimate, + Collector, + CollectorMetadata, + CollectorResult, + CoreCollector, + EstimatedCost, + EvidenceKind, + NetworkAccess, + OutputPolicy, + PluginCollector, + PostRunAction, + PrivilegeLevel, + ResourceClass, + RunRequest, + RunStatus, + Specialty, + ValidationResult, +) + +if TYPE_CHECKING: + from logicytics.module.api import ( + CollectorFailureSnapshot, + CollectorSnapshot, + RunSnapshot, + load_configuration, + open_artifact, + plan_run, + query_run, + read_artifact, + run_collection, + ) + +_APPLICATION_EXPORTS = frozenset( + { + "CollectorFailureSnapshot", + "CollectorSnapshot", + "RunSnapshot", + "load_configuration", + "open_artifact", + "plan_run", + "query_run", + "read_artifact", + "run_collection", + } +) + +# Keep the former import paths working for shipped collectors and independently +# authored plugins while the implementation lives in its domain subpackages. +for _legacy_name, _module_name in { + "contracts": "logicytics.module.contracts", + "platform_adapters": "logicytics.module.platform_adapters", +}.items(): + sys.modules.setdefault(f"{__name__}.{_legacy_name}", importlib.import_module(_module_name)) + + +def __getattr__(name: str): + """Load application services and global host infrastructure on explicit access.""" + if name in _APPLICATION_EXPORTS: + from logicytics.module import api + + return getattr(api, name) + if name == "ctypes_collector": + return importlib.import_module("logicytics.global.ctypes_collector") + global_infrastructure = importlib.import_module("logicytics.global.ctypes_collector") + if hasattr(global_infrastructure, name): + return getattr(global_infrastructure, name) + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") + + +__all__ = [ + "CONTRACT_VERSION", + "Capability", + "CollectionEstimate", + "Collector", + "CollectorFailureSnapshot", + "CollectorMetadata", + "CollectorResult", + "CollectorSnapshot", + "CoreCollector", + "EstimatedCost", + "EvidenceKind", + "NetworkAccess", + "OutputPolicy", + "PluginCollector", + "PostRunAction", + "PrivilegeLevel", + "ResourceClass", + "RunRequest", + "RunSnapshot", + "RunStatus", + "Specialty", + "ValidationResult", + "load_configuration", + "open_artifact", + "plan_run", + "query_run", + "read_artifact", + "run_collection", +] diff --git a/logicytics/__main__.py b/logicytics/__main__.py new file mode 100644 index 00000000..482339a8 --- /dev/null +++ b/logicytics/__main__.py @@ -0,0 +1,5 @@ +"""Enable `python -m logicytics`.""" + +from logicytics.cli import main + +raise SystemExit(main()) diff --git a/logicytics/cli/__init__.py b/logicytics/cli/__init__.py new file mode 100644 index 00000000..6e73ecad --- /dev/null +++ b/logicytics/cli/__init__.py @@ -0,0 +1,46 @@ +"""User-facing Logicytics command entry points.""" + +from __future__ import annotations + +import sys +from pathlib import Path + +from logicytics.cli.commands import cli_methods, CLI +from logicytics.module.terminal import terminal_lifecycle +from logicytics.module.virtual_environment import ( + is_running_in_virtual_environment, + render_virtual_environment_error, +) + +__all__ = ["CLI", "cli_methods", "main"] + + +def _project_root() -> Path: + """Return the repository root without importing the normal CLI implementation.""" + return Path(__file__).resolve().parents[2] + + +def main(argv: list[str] | None = None) -> int: + """Guard the interpreter before loading the full CLI implementation.""" + try: + with terminal_lifecycle(): + if not is_running_in_virtual_environment(): + render_virtual_environment_error(sys.stderr, _project_root()) + return 2 + from logicytics.cli.commands import main as command_main + + return command_main(argv) + except KeyboardInterrupt: + from logicytics.module.logging import ApplicationLogger + + ApplicationLogger.render_section(sys.stderr, "Command cancelled", ("Interrupted by user.",)) + return 130 + + +def __getattr__(name: str) -> object: + """Load the normal CLI API only when a caller asks for it.""" + if name in {"CLI", "cli_methods"}: + from logicytics.cli.commands import CLI, cli_methods + + return {"CLI": CLI, "cli_methods": cli_methods}[name] + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") diff --git a/logicytics/cli/__main__.py b/logicytics/cli/__main__.py new file mode 100644 index 00000000..f8991e1e --- /dev/null +++ b/logicytics/cli/__main__.py @@ -0,0 +1,6 @@ +"""Run the normal Logicytics CLI with ``python -m logicytics.cli``.""" + +from logicytics.cli import main + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/logicytics/cli/commands.py b/logicytics/cli/commands.py new file mode 100644 index 00000000..69e96950 --- /dev/null +++ b/logicytics/cli/commands.py @@ -0,0 +1,1453 @@ +"""Command-line orchestration for v4 planning and supervised execution.""" + +from __future__ import annotations + +import argparse +import contextlib +import importlib.util +import json +import os +import platform +import subprocess +import sys +from dataclasses import replace +from pathlib import Path +from time import perf_counter +from typing import Literal + +from logicytics.module.configuration import AppConfig, load_config +from logicytics.module.contracts import Capability, OutputPolicy, PostRunAction, RunRequest +from logicytics.module.discovery import preflight +from logicytics.module.environment import inspect_environment +from logicytics.module.errors import LogicyticsError +from logicytics.module.interaction import ( + load_history, + match_flag, + record_command, + record_match, + usage_statistics, + write_usage_graph, +) +from logicytics.module.logging import ( + ApplicationLogger, + HumanArgumentParser, + collector_log_source, + get_application_logger, +) +from logicytics.module.maintenance import ( + build_manifest, + compare_files, + developer_checks, + load_local_manifest, + local_version, + maintenance_diagnostics, + write_local_manifest, +) +from logicytics.module.manifest import MANIFEST_SCHEMA_VERSION +from logicytics.module.modes import ( + EXECUTION_MODES, + LEGACY_MODE_ALIASES, + ExecutionStrategy, + mode_matrix, + resolve_execution_mode, +) +from logicytics.module.output_layout import ensure_output_layout +from logicytics.module.planner import build_plan +from logicytics.module.platform_adapters import process_adapter +from logicytics.module.runtime import RunSupervisor +from logicytics.module.sysinternals import ensure_sysinternals +from logicytics.module.terminal import terminal_lifecycle +from logicytics.module.usb import ensure_usb_storage, find_windows_installation +from logicytics.module.virtual_environment import ( + is_running_in_virtual_environment, + virtual_environment_details, + virtual_environment_error, +) + + +class CLI: + """Translate command-line arguments into validated application requests.""" + + @staticmethod + def render_preflight( + logger: ApplicationLogger, + validation: dict[str, list[dict[str, object]]], + sysinternals: dict[str, str], + ) -> None: + """Present validated collectors and diagnostics without exposing internal JSON.""" + valid = validation["valid"] + invalid = validation["invalid"] + quarantined = validation["quarantined"] + logger.box( + "Preflight summary", + ( + f"Valid collectors: {len(valid)}", + f"Quarantined extensions: {len(quarantined)}", + f"Blocking failures: {len(invalid)}", + f"Sysinternals: {sysinternals['status']}", + ), + ) + for item in (*invalid, *quarantined): + raw_diagnostics = item.get("diagnostics", []) + diagnostics = raw_diagnostics if isinstance(raw_diagnostics, list) else [] + details = "; ".join( + str(diagnostic.get("message", "invalid collector")) for diagnostic in diagnostics if + isinstance(diagnostic, dict) + ) + logger.event( + "ERROR" if item in invalid else "WARNING", + details or "collector validation failed", + source="logicytics.cli.commands", + collector=str(item["id"]), + ) + + @staticmethod + def project_root() -> Path: + """Return the repository root from the relocated CLI package.""" + return Path(__file__).resolve().parents[2] + + @staticmethod + def request( + arguments: argparse.Namespace, + default_workers: int, + configured_blocked_capabilities: tuple[Capability, ...] = (), + ) -> RunRequest: + """Build an immutable run request while enforcing mode and rerun conflicts.""" + legacy_flags = {flag: getattr(arguments, flag, False) for flag in LEGACY_MODE_ALIASES} + + selected_mode = resolve_execution_mode(getattr(arguments, "mode", None), legacy_flags) + profile_mode = resolve_execution_mode(getattr(arguments, "profile", None), {}) + + if selected_mode is not None and profile_mode is not None: + raise ValueError("--profile cannot be combined with a named collection mode") + + mode = selected_mode or profile_mode + profile = mode.profile if mode is not None else "standard" + + explicit_sequential = getattr(arguments, "sequential", False) + explicit_parallel = getattr(arguments, "parallel", False) + + performance_check = bool(getattr(arguments, "performance_check", False)) + + if performance_check and explicit_parallel: + raise ValueError("performance checking requires sequential execution") + + if mode is not None: + if explicit_sequential and mode.strategy is ExecutionStrategy.PARALLEL: + if legacy_flags["threaded"]: + raise ValueError("sequential execution conflicts with legacy --threaded mode") + if not performance_check: + raise ValueError(f"{mode.name} mode requires configured parallel execution") + + if explicit_parallel and mode.strategy is ExecutionStrategy.SEQUENTIAL: + if legacy_flags["default_mode"]: + raise ValueError("parallel execution conflicts with sequential default mode") + raise ValueError(f"{mode.name} mode requires sequential execution") + + sequential = performance_check or explicit_sequential or ( + mode is not None and mode.strategy is ExecutionStrategy.SEQUENTIAL) + parallel = not performance_check and ( + explicit_parallel or (mode is not None and mode.strategy is ExecutionStrategy.PARALLEL)) + + requested_workers = getattr(arguments, "workers", None) + + if sequential and requested_workers is not None and requested_workers != 1: + raise ValueError("sequential execution requires --workers=1") + + worker_count = 1 if sequential else requested_workers or default_workers + + if parallel and worker_count < 2: + raise ValueError("parallel execution requires at least two configured workers") + + parent_run_id: str | None = None + + rerun_value = getattr(arguments, "rerun_from", None) + + if rerun_value is not None: + if not isinstance(rerun_value, (str, os.PathLike)): + raise TypeError(f"rerun_from must be a path-like value, got {type(rerun_value).__name__}") + + rerun_path = Path(rerun_value) + + if not arguments.include: + raise ValueError("--rerun-from requires at least one explicit --include collector ID") + + manifest_path = rerun_path / "manifest.json" if rerun_path.is_dir() else rerun_path + + try: + previous = json.loads(manifest_path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise ValueError(f"original run manifest cannot be loaded: {error}") from error + + if not isinstance(previous, dict) or not isinstance( + previous.get("run_id"), + str, + ): + raise ValueError("original run manifest must contain a valid run_id") + + manifest_schema_version = previous.get("manifest_schema_version") + + if ( + not isinstance(manifest_schema_version, int) + or isinstance(manifest_schema_version, bool) + or manifest_schema_version != MANIFEST_SCHEMA_VERSION + ): + schema_version_display = "None" if manifest_schema_version is None else repr(manifest_schema_version) + + raise ValueError( + f"original run manifest uses unsupported schema_version {schema_version_display}; expected {MANIFEST_SCHEMA_VERSION}" + ) + + if previous.get("status") not in { + "succeeded", + "partial", + "failed", + "cancelled", + }: + raise ValueError("original run manifest must describe a finalized run") + + resolved = previous.get("resolved_plan") + + if not isinstance(resolved, list) or not all(isinstance(item, str) for item in resolved): + raise ValueError("original run manifest must contain a valid resolved_plan") + + unknown = sorted(set(arguments.include) - set(resolved)) + + if unknown: + raise ValueError(f"rerun collectors were not present in the original run: {', '.join(unknown)}") + + parent_run_id = previous["run_id"] + + selection_only = getattr(arguments, "command", None) == "collector" + include_arguments = tuple(getattr(arguments, "include", ()) or ()) + exclude_arguments = tuple(getattr(arguments, "exclude", ()) or ()) + plugins_enabled = bool(getattr(arguments, "plugins", False)) + if selection_only and ( + include_arguments or exclude_arguments or arguments.profile is not None or plugins_enabled): + raise ValueError("collector execution cannot combine its ID with profile or selection flags") + + includes = (arguments.collector_id,) if selection_only else include_arguments + + requested_blocked_capabilities = tuple( + Capability(value) for value in getattr(arguments, "block_capability", ())) + blocked_capabilities = tuple(dict.fromkeys((*configured_blocked_capabilities, *requested_blocked_capabilities))) + + return RunRequest( + profile=profile, + include=includes, + exclude=exclude_arguments, + selection_only=selection_only, + enable_plugins=plugins_enabled, + max_workers=worker_count, + acknowledge_authorization=getattr( + arguments, + "acknowledge_authorization", + False, + ), + blocked_capabilities=blocked_capabilities, + performance_check=performance_check, + rerun_from=parent_run_id, + output_policy=( + OutputPolicy.MANIFEST_ONLY if getattr(arguments, "no_package", False) else OutputPolicy.PACKAGE), + post_run_action=( + PostRunAction.REBOOT + if getattr(arguments, "reboot", False) + else PostRunAction.SHUTDOWN + if getattr(arguments, "shutdown", False) + else PostRunAction.NONE + ), + ) + + @staticmethod + def parser() -> argparse.ArgumentParser: + """Create the complete parser for maintenance, planning, and collection commands.""" + parser = HumanArgumentParser(description="Logicytics v4 run-oriented evidence framework") + parser.add_argument( + "--config", + type=Path, + help="Path to the authoritative Logicytics YAML configuration file", + ) + parser.add_argument( + "--usb", + nargs="?", + const="", + metavar="DRIVE", + help="Run from removable storage and locate Windows by scanning A through Z; optionally select its drive letter.", + ) + parser.add_argument( + "--usage", + action="store_true", + help="Show local interaction statistics and create a usage graph.", + ) + parser.add_argument("--modes", action="store_true", help="Show the typed execution-mode inclusion matrix.") + parser.add_argument( + "--match", + metavar="TEXT", + help="Suggest the closest documented action for natural-language input.", + ) + subcommands = parser.add_subparsers(dest="command", parser_class=HumanArgumentParser) + for command in ("preflight", "debug", "update", "dev", "plan", "run", "collector"): + subparser = subcommands.add_parser(command, help=f"Run the {command} action.") + subparser.add_argument( + "--config", + type=Path, + default=argparse.SUPPRESS, + help="Path to the authoritative Logicytics YAML configuration file", + ) + subparser.add_argument( + "--usb", + nargs="?", + const="", + default=argparse.SUPPRESS, + metavar="DRIVE", + help="Run from removable storage and locate Windows; optionally select its drive letter.", + ) + subparser.add_argument( + "--profile", + default=None, + choices=tuple(EXECUTION_MODES), + help="Named user-facing collection mode.", + ) + if command in {"preflight", "plan"}: + subparser.add_argument( + "--invalidate-cache", + action="store_true", + help="Discard cached preflight metadata before validation.", + ) + subparser.add_argument( + "--include", + action="append", + default=[], + metavar="COLLECTOR_ID", + help="Include a collector by exact dotted ID; repeat for multiple collectors.", + ) + subparser.add_argument( + "--exclude", + action="append", + default=[], + metavar="COLLECTOR_ID", + help="Exclude a collector by exact dotted ID; repeat for multiple collectors.", + ) + subparser.add_argument( + "--plugins", + action="store_true", + help="Enable all valid opt-in plugin collectors for the selected profile.", + ) + subparser.add_argument( + "--workers", + type=int, + metavar="COUNT", + help="Bound concurrent isolated workers to this positive count.", + ) + subparser.add_argument( + "--block-capability", + action="append", + default=[], + choices=[capability.value for capability in Capability], + help="Block a declared capability for the selected collectors; repeat as needed.", + ) + if command == "plan": + subparser.add_argument( + "--mode", + choices=tuple(EXECUTION_MODES), + help="Select one user-facing typed execution mode for the plan.", + ) + if command == "run": + subparser.add_argument( + "--rerun-from", + type=Path, + help="Rerun explicitly included collector IDs from a finalized run manifest or directory.", + ) + execution = subparser.add_mutually_exclusive_group() + execution.add_argument( + "--sequential", + action="store_true", + help="Explicitly run isolated collectors one at a time for deterministic debugging.", + ) + execution.add_argument( + "--parallel", + action="store_true", + help="Explicitly run isolated collectors with the configured bounded worker limit.", + ) + mode = subparser.add_mutually_exclusive_group() + mode.add_argument( + "--mode", + choices=tuple(EXECUTION_MODES), + help="Select one user-facing typed execution mode.", + ) + mode.add_argument( + "--default", + dest="default_mode", + action="store_true", + help="Run the standard built-in profile.", + ) + mode.add_argument( + "--threaded", + action="store_true", + help="Run the standard built-in profile with configured parallel workers.", + ) + mode.add_argument("--minimal", action="store_true", help="Run the minimal built-in profile.") + mode.add_argument("--depth", action="store_true", help="Run the deep built-in profile.") + subparser.add_argument( + "--performance-check", + action="store_true", + help="Measure the selected mode serially and save per-collector duration measurements.", + ) + subparser.add_argument( + "--no-package", + action="store_true", + help="Finalize the run manifest without creating a ZIP package.", + ) + power_action = subparser.add_mutually_exclusive_group() + power_action.add_argument( + "--reboot", + action="store_true", + help="Schedule a reboot only after the run is packaged successfully.", + ) + power_action.add_argument( + "--shutdown", + action="store_true", + help="Schedule a shutdown only after the run is packaged successfully.", + ) + subparser.add_argument( + "--acknowledge-authorization", + action="store_true", + help="Confirm you are authorized to collect the selected evidence.", + ) + subparser.add_argument( + "--interactive", + action="store_true", + help="Pause at the final status so an interactive command window remains visible.", + ) + if command == "collector": + subparser.add_argument("collector_id", help="Exact dotted ID of the collector to run independently.") + subparser.add_argument( + "--acknowledge-authorization", + action="store_true", + help="Confirm you are authorized to collect this collector's evidence.", + ) + subparser.add_argument( + "--no-package", + action="store_true", + help="Finalize the direct collector manifest without creating a ZIP package.", + ) + subparser.add_argument( + "--interactive", + action="store_true", + help="Pause at the final status so an interactive command window remains visible.", + ) + if command == "update": + subparser.add_argument( + "--apply", + action="store_true", + help="Explicitly run git pull after repository checks.", + ) + subparser.add_argument( + "--launch-action", + choices=("preflight", "debug", "dev"), + help="Select a safe maintenance action to launch after a successful update check.", + ) + subparser.add_argument( + "--new-window", + action="store_true", + help="Launch --launch-action in a visible, separate Windows command window.", + ) + if command == "dev": + subparser.add_argument( + "--write-manifest", + action="store_true", + help=("Write the reviewed local integrity manifest; requires --next-version unless interactive."), + ) + subparser.add_argument( + "--next-version", + help=("Semantic version to validate and record in a newly written integrity manifest."), + ) + subparser.add_argument( + "--interactive", + action="store_true", + help=("Show contribution checks and prompt before changing the local integrity manifest."), + ) + return parser + + @staticmethod + def write_json(path: Path, payload: object) -> Path: + """Atomically write a human-readable JSON diagnostic artifact.""" + path.parent.mkdir(parents=True, exist_ok=True) + temporary = path.with_suffix(path.suffix + ".tmp") + temporary.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8") + temporary.replace(path) + return path + + @staticmethod + def launch_action_window(root: Path, action: str) -> int: + """Launch one allowlisted maintenance action in a separate visible Windows console.""" + if sys.platform != "win32": + raise OSError("new command windows are supported only on Windows") + if action not in {"preflight", "debug", "dev"}: + raise ValueError(f"unsupported new-window action: {action}") + command = [sys.executable, "-m", "logicytics", action] + process = process_adapter.popen( + command, + cwd=root, + shell=False, + creationflags=process_adapter.create_new_console, + close_fds=True, + ) + return process.pid + + @staticmethod + def repository_status(root: Path) -> dict[str, bool | int | str | None]: + """Check the local Git worktree, origin declaration, and remote reachability.""" + status: dict[str, bool | int | str | None] = { + "git_available": False, + "git_version": None, + "is_repository": False, + "origin_configured": False, + "remote_reachable": False, + "repository_returncode": None, + "origin_returncode": None, + "reachability_returncode": None, + } + try: + git = process_adapter.run( + ["git", "--version"], + capture_output=True, + check=False, + text=True, + ) + except OSError: + return status + + git_output = git.stdout if isinstance(git.stdout, str) else "" + status["git_available"] = git.returncode == 0 + status["git_version"] = git_output.strip() or None + if git.returncode != 0: + return status + + try: + repository = process_adapter.run( + ["git", "rev-parse", "--is-inside-work-tree"], + cwd=root, + capture_output=True, + check=False, + text=True, + ) + except OSError: + return status + status["repository_returncode"] = repository.returncode + repository_output = repository.stdout if isinstance(repository.stdout, str) else "" + status["is_repository"] = repository.returncode == 0 and repository_output.strip().casefold() == "true" + if not status["is_repository"]: + return status + + try: + origin = process_adapter.run( + ["git", "remote", "get-url", "origin"], + cwd=root, + capture_output=True, + check=False, + text=True, + ) + except OSError: + return status + status["origin_returncode"] = origin.returncode + origin_output = origin.stdout if isinstance(origin.stdout, str) else "" + status["origin_configured"] = origin.returncode == 0 and bool(origin_output.strip()) + if not status["origin_configured"]: + return status + + try: + reachability = process_adapter.run( + ["git", "ls-remote", "--exit-code", "origin", "HEAD"], + cwd=root, + capture_output=True, + check=False, + text=True, + timeout=15, + ) + except (OSError, subprocess.TimeoutExpired): + return status + status["reachability_returncode"] = reachability.returncode + status["remote_reachable"] = reachability.returncode == 0 + return status + + def run_developer_action( + self, + root: Path, + configuration: AppConfig, + arguments: argparse.Namespace, + debug_logs: Path, + logger: ApplicationLogger, + repository: dict[str, bool | int | str | None], + ) -> int: + """Run read-only contribution checks and an explicitly confirmed manifest update.""" + settings = configuration.maintenance + logger.event( + "INFO", + "development_checks_started", + source="logicytics.cli.commands", + manifest_write_requested=arguments.write_manifest, + interactive=arguments.interactive, + ) + existing = load_local_manifest(root, settings) + + comparison = ( + compare_files(root, settings, existing) + if existing + else { + "missing": [], + "modified": [], + "extra": [], + "unchanged": [], + } + ) + + checks = developer_checks(root, settings) + next_version = arguments.next_version + write_requested = arguments.write_manifest + logger.event( + "INFO", + "development_integrity_reviewed", + source="logicytics.cli.commands", + manifest_available=existing is not None, + missing_files=len(comparison["missing"]), + modified_files=len(comparison["modified"]), + extra_files=len(comparison["extra"]), + ) + + if arguments.interactive: + organization_checks: tuple[ + Literal[ + "naming_violations", + "misplaced_python", + "missing_module_docstrings", + "crowded_modules", + ], + ..., + ] = ( + "naming_violations", + "misplaced_python", + "missing_module_docstrings", + "crowded_modules", + ) + logger.box( + "Contribution and repository organization checks", + ( + *( + f"{name.title()}: {len(comparison[name])}" + for name in ( + "missing", + "modified", + "extra", + "unchanged", + ) + ), + *(f"{name.replace('_', ' ').title()}: {len(checks[name])}" for name in organization_checks), + ), + ) + + if next_version is None: + current_version = local_version(root) + entered = input(f"Next semantic version [{current_version}]: ").strip() + next_version = entered or current_version + + answer = input("Write the reviewed integrity manifest? [y/N]: ").strip().casefold() + + write_requested = answer in {"y", "yes"} + + if arguments.write_manifest and next_version is None: + raise ValueError("--write-manifest requires --next-version unless --interactive is used") + + manifest_path: str | None = None + + if write_requested: + if next_version is None: + raise ValueError("a semantic next version is required to write the integrity manifest") + + logger.event( + "INFO", + "development_manifest_write_started", + source="logicytics.cli.commands", + next_version=next_version, + ) + manifest = build_manifest(root, settings, next_version) + written_manifest = write_local_manifest( + root, + settings, + manifest, + ) + + manifest_path = str(written_manifest) + logger.event( + "INFO", + "development_manifest_written", + source="logicytics.cli.commands", + manifest_path=manifest_path, + ) + + payload = { + "checks": checks, + "comparison": comparison, + "repository": repository, + "current_version": local_version(root), + "existing_manifest_version": (existing.version if existing else None), + "manifest_written": manifest_path, + "next_version": next_version, + } + + development_path = debug_logs / "development.json" + self.write_json(development_path, payload) + organization_issues = sum( + len(checks[name]) + for name in ( + "naming_violations", + "misplaced_python", + "missing_module_docstrings", + "crowded_modules", + ) + ) + logger.event( + "INFO" if organization_issues == 0 else "WARNING", + "development_structure_reviewed", + source="logicytics.cli.commands", + organization_issues=organization_issues, + github_reachable=bool(repository["remote_reachable"]), + ) + logger.box( + "Development summary", + ( + f"Integrity: {len(comparison['missing'])} missing, " + f"{len(comparison['modified'])} modified, {len(comparison['extra'])} extra", + f"Organization issues: {organization_issues}", + f"GitHub: {'reachable' if repository['remote_reachable'] else 'unreachable'}", + f"Manifest: {'written' if manifest_path else 'not changed'}", + *((f"Manifest path: {manifest_path}",) if manifest_path else ()), + f"Diagnostic report: {development_path}", + ), + ) + return 0 if organization_issues == 0 and repository["remote_reachable"] else 1 + + +def main(argv: list[str] | None = None) -> int: + """Run the selected preflight, planning, or supervised execution command.""" + if not is_running_in_virtual_environment(): + ApplicationLogger.render_alert( + sys.stderr, + "Logicytics startup error", + virtual_environment_error(CLI.project_root()), + ) + return 2 + cli_parser = cli_methods.parser() + try: + arguments = cli_parser.parse_args(argv) + except KeyboardInterrupt: + ApplicationLogger.render_section(sys.stderr, "Command cancelled", ("Interrupted by user.",)) + return 130 + + standalone_actions = sum( + bool(value) + for value in ( + arguments.usage, + arguments.match, + arguments.modes, + ) + ) + + if standalone_actions > 1: + cli_parser.error("--usage, --match, and --modes are mutually exclusive") + + if arguments.command is not None and standalone_actions: + cli_parser.error("--usage, --match, and --modes are standalone actions") + + if arguments.usage: + arguments.command = "usage" + elif arguments.match: + arguments.command = "match" + elif arguments.modes: + arguments.command = "modes" + + if arguments.command is None: + if arguments.config is None: + cli_parser.print_help() + return 0 + arguments.command = "config" + + root = cli_methods.project_root() + application_logger = None + command_started_at: float | None = None + + try: + configuration = load_config(root, arguments.config) + cache_directory = root / ".cache" + preflight_cache_directory: Path | None = None + preflight_temporary_directory: Path | None = None + usb_value = getattr(arguments, "usb", None) + if usb_value is not None: + usb_context = find_windows_installation(usb_value or None) + ensure_usb_storage( + usb_context, + project_root=root, + output_root=configuration.runtime.output_root, + cache_directory=cache_directory, + temporary_directory=root / ".temp", + ) + configuration = replace( + configuration, + runtime=replace(configuration.runtime, temporary_directory="project"), + ) + preflight_cache_directory = cache_directory / "preflight" + preflight_temporary_directory = root / ".temp" / "preflight" + + layout = ensure_output_layout(configuration.runtime.output_root) + + application_logger = get_application_logger( + layout.application_log, + configuration.logging, + ) + + application_logger.event( + "INFO", + "command_started", + source="logicytics.cli.commands", + command=arguments.command, + **( + { + "usb_windows_drive": usb_context.windows_drive, + "usb_drive_scan": ", ".join(usb_context.scanned_drives), + } + if usb_value is not None + else {} + ), + ) + started_at = perf_counter() + command_started_at = started_at + + def finish_command( + exit_code: int, + *, + status: str | None = None, + **fields: float | str, + ) -> int: + """Record one command's terminal status and elapsed time.""" + application_logger.event( + "INFO" if exit_code == 0 else "WARNING", + "command_finished", + source="logicytics.cli.commands", + command=str(arguments.command), + exit_code=exit_code, + status=status or ("succeeded" if exit_code == 0 else "failed"), + duration_seconds=round(perf_counter() - started_at, 3), + **fields, + ) + return exit_code + + cache_directory.mkdir(parents=True, exist_ok=True) + history_path = cache_directory / "interaction_history.json.gz" + + if configuration.interaction.history_enabled and arguments.command not in {"match", "usage"}: + record_command( + history_path, + str(arguments.command), + mode=getattr(arguments, "mode", None) or getattr(arguments, "profile", None), + ) + + if arguments.command == "config": + selected_path = arguments.config.resolve() + application_logger.box( + "Configuration", + ( + "Configuration validated successfully.", + f"Path: {selected_path}", + f"Schema version: {configuration.schema_version}", + ), + ) + return finish_command(0, status="configuration_validated", configuration_path=str(selected_path)) + + last_preflight_progress_at = 0.0 + + def report_preflight_progress(phase: str, checked: int, total: int, current: str) -> None: + """Keep long sequential collector validation visibly alive without flooding output.""" + nonlocal last_preflight_progress_at + now = perf_counter() + should_report = ( + last_preflight_progress_at == 0.0 or + now - last_preflight_progress_at >= 1.0 or + (phase == "checked" and checked == total) + ) + if should_report: + application_logger.event( + "INFO", + "preflight_progress", + console=False, + source="logicytics.cli.commands", + phase="validating collectors", + checked=checked, + total=total, + current=current, + ) + application_logger.progress( + "Preflight", checked, total, current + ) + last_preflight_progress_at = now + + if arguments.command == "match": + history = load_history(history_path) + + match = match_flag( + arguments.match, + threshold=(configuration.interaction.similarity_threshold), + model_name=configuration.interaction.model_name, + history=history, + ) + + if configuration.interaction.history_enabled: + record_match(history_path, match) + + application_logger.box( + "Match result", + ( + f"Input: {match.input}", + f"Matched flag: {match.matched_flag or 'none'}", + f"Confidence: {match.accuracy:.1%}", + f"Match source: {match.source.replace('_', ' ')}", + f"History persisted: {'yes' if configuration.interaction.history_enabled else 'no'}", + *( + (f"Model: {match.model_name} (threshold {configuration.interaction.similarity_threshold:.1%})",) + if configuration.interaction.model_debug + else () + ), + ), + ) + + exit_code = 0 if match.matched_flag is not None else 1 + return finish_command( + exit_code, + status="matched" if exit_code == 0 else "not_matched", + matched_flag=match.matched_flag or "none", + ) + + if arguments.command == "usage": + statistics = usage_statistics(load_history(history_path)) + raw_average_accuracy = statistics.get("average_accuracy", 0.0) + average_accuracy = float(raw_average_accuracy) if isinstance(raw_average_accuracy, (int, float)) else 0.0 + + graph_path = write_usage_graph( + cache_directory / "flag_usage.svg", + statistics, + ) + + frequencies = statistics.get("per_flag_frequency", {}) + frequency_rows = ( + tuple(f"{flag}: {count}" for flag, count in sorted(frequencies.items())) + if isinstance(frequencies, dict) and frequencies + else ("none",) + ) + application_logger.box( + "Interaction usage", + ( + f"Total interactions: {statistics['total_interactions']}", + f"Average confidence: {average_accuracy:.1%}", + f"Common device: {statistics['common_device'] or 'none'}", + f"Common input: {statistics['common_input'] or 'none'}", + "Flag frequency:", + *(f" {row}" for row in frequency_rows), + f"Usage graph: {graph_path}", + ), + ) + + return finish_command(0, status="usage_written") + + if arguments.command == "modes": + modes_configuration_hash = configuration.fingerprint() + application_logger.event( + "INFO", + "preflight_started", + source="logicytics.cli.commands", + configuration_hash=modes_configuration_hash, + purpose="mode_matrix", + ) + report = preflight( + root, + configuration_hash=modes_configuration_hash, + progress=report_preflight_progress, + cache_directory=preflight_cache_directory, + temporary_directory=preflight_temporary_directory, + ) + application_logger.event( + "INFO", + "preflight_finished", + source="logicytics.cli.commands", + valid_collectors=len(report.valid), + invalid_collectors=len(report.invalid), + purpose="mode_matrix", + ) + + matrix_payload = mode_matrix((*report.valid, *report.invalid)) + matrix_path = layout.debug_logs / "modes.json" + cli_methods.write_json(matrix_path, matrix_payload) + mode_rows = [] + for item in matrix_payload["modes"]: + aliases = ", ".join(item["legacy_aliases"]) or "none" + mode_rows.append( + f"{item['name']}: {item['description']} ({len(item['collector_ids'])} collectors; aliases: {aliases})") + application_logger.box( + "Execution modes", + ( + *mode_rows, + f"Valid collectors: {len(report.valid)}", + f"Invalid collectors: {len(report.invalid)}", + f"Machine-readable matrix: {matrix_path}", + ), + ) + + return finish_command( + 0, + status="mode_matrix_written", + valid_collectors=len(report.valid), + invalid_collectors=len(report.invalid), + ) + + application_logger.event( + "INFO", + "preflight_started", + source="logicytics.cli.commands", + configuration_hash=configuration.fingerprint(), + ) + + report = preflight( + root, + configuration_hash=configuration.fingerprint(), + progress=report_preflight_progress, + invalidate_cache=getattr(arguments, "invalidate_cache", False), + cache_directory=preflight_cache_directory, + temporary_directory=preflight_temporary_directory, + ) + application_logger.event( + "INFO", + "preflight_finished", + source="logicytics.cli.commands", + valid_collectors=len(report.valid), + invalid_collectors=len(report.invalid), + ) + + if arguments.command == "preflight": + validation = report.to_dict( + selected_plugins=tuple(arguments.include), + enable_plugins=arguments.plugins, + ) + + sysinternals = ensure_sysinternals(root, configuration.maintenance).to_dict() + cli_methods.render_preflight(application_logger, validation, sysinternals) + + exit_code = 0 if not validation["invalid"] else 2 + return finish_command( + exit_code, + status="validated" if exit_code == 0 else "invalid_collectors", + valid_collectors=len(validation["valid"]), + invalid_collectors=len(validation["invalid"]), + ) + + if arguments.command == "debug": + payload: dict[str, object] = { + "configuration": (configuration.to_manifest_dict()), + "environment": inspect_environment().to_dict(), + "python": { + "executable": sys.executable, + "implementation": (platform.python_implementation()), + "version": platform.python_version(), + "prefix": sys.prefix, + "virtual_environment": virtual_environment_details(), + "psutil_available": (importlib.util.find_spec("psutil") is not None), + "cpu_count": os.cpu_count(), + }, + "sysinternals": (ensure_sysinternals(root, configuration.maintenance).to_dict()), + "preflight": { + "valid_collectors": len(report.valid), + "invalid_collectors": len(report.invalid), + }, + "maintenance": maintenance_diagnostics( + root, + configuration.maintenance, + ), + } + + debug_path = layout.debug_logs / "debug.json" + + payload["debug_log"] = str(debug_path) + + cli_methods.write_json( + debug_path, + payload, + ) + + application_logger.box( + "Diagnostics", + ( + f"Valid collectors: {len(report.valid)}", + f"Invalid collectors: {len(report.invalid)}", + f"Virtual environment: {'yes' if is_running_in_virtual_environment() else 'no'}", + f"Diagnostic report: {debug_path}", + ), + ) + + exit_code = 0 if not report.invalid else 2 + return finish_command( + exit_code, + status="diagnostics_written" if exit_code == 0 else "invalid_collectors", + valid_collectors=len(report.valid), + invalid_collectors=len(report.invalid), + ) + + if arguments.command == "update": + if arguments.new_window != (arguments.launch_action is not None): + raise ValueError("--new-window and --launch-action must be provided together") + + application_logger.event( + "INFO", + "update_started", + source="logicytics.cli.commands", + action="apply update" if arguments.apply else "check for updates", + ) + repository = cli_methods.repository_status(root) + payload: dict[str, object] = { + **repository, + "applied": False, + } + application_logger.event( + "INFO" if repository["remote_reachable"] else "WARNING", + "update_repository_checked", + source="logicytics.cli.commands", + git_available=bool(repository["git_available"]), + repository_ready=bool(repository["is_repository"]), + origin_configured=bool(repository["origin_configured"]), + github_reachable=bool(repository["remote_reachable"]), + ) + + pull_returncode: int | None = None + + if not repository["remote_reachable"]: + update_path = layout.debug_logs / "update.json" + cli_methods.write_json(update_path, payload) + if not repository["git_available"]: + next_step = "Install Git, then run the update command again." + elif not repository["is_repository"]: + next_step = "Run the update command from a Git repository." + elif not repository["origin_configured"]: + next_step = "Configure the repository's origin remote, then retry." + else: + next_step = "Restore GitHub connectivity, then retry the update command." + application_logger.event( + "WARNING", + "update_not_applied", + source="logicytics.cli.commands", + reason="repository remote is unavailable", + diagnostic_report=str(update_path), + ) + application_logger.box( + "Update summary", + ( + "Result: update was not applied.", + f"Next step: {next_step}", + f"Diagnostic report: {update_path}", + ), + ) + return finish_command( + 2, + status="repository_unreachable", + git_available=str(repository["git_available"]), + remote_reachable=str(repository["remote_reachable"]), + ) + + if arguments.apply: + application_logger.event( + "INFO", + "update_apply_started", + source="logicytics.cli.commands", + command="git pull", + ) + pulled = process_adapter.run( + ["git", "pull"], + cwd=root, + capture_output=True, + check=False, + text=True, + ) + + pull_returncode = pulled.returncode + + payload.update( + { + "applied": True, + "returncode": pulled.returncode, + "stdout": pulled.stdout, + "stderr": pulled.stderr, + } + ) + application_logger.event( + "INFO" if pulled.returncode == 0 else "ERROR", + "update_apply_finished", + source="logicytics.cli.commands", + exit_code=pulled.returncode, + ) + + update_succeeded = not arguments.apply or pull_returncode == 0 + + if arguments.new_window and update_succeeded: + launched_process_id = cli_methods.launch_action_window( + root, + arguments.launch_action, + ) + + payload["launched_action"] = arguments.launch_action + payload["launched_process_id"] = launched_process_id + + update_path = layout.debug_logs / "update.json" + cli_methods.write_json(update_path, payload) + update_lines = [ + f"Repository: {'ready' if repository['is_repository'] else 'not detected'}", + "GitHub: reachable", + f"Action: {'git pull completed' if arguments.apply else 'connectivity check only'}", + f"Result: {'succeeded' if update_succeeded else 'failed'}", + ] + if arguments.apply: + pull_exit_code = pull_returncode if pull_returncode is not None else "none" + update_lines.append(f"Git pull exit code: {pull_exit_code}") + if not update_succeeded: + update_lines.append("Next step: inspect the diagnostic report, resolve Git's error, then retry.") + if payload.get("launched_action"): + update_lines.extend( + ( + f"Launched action: {payload['launched_action']}", + f"Launched process id: {payload['launched_process_id']}", + ) + ) + update_lines.append(f"Diagnostic report: {update_path}") + application_logger.event( + "INFO" if update_succeeded else "ERROR", + "update_finished", + source="logicytics.cli.commands", + action="applied" if arguments.apply else "checked", + status="succeeded" if update_succeeded else "failed", + ) + application_logger.box("Update summary", tuple(update_lines)) + + return finish_command( + 0 if update_succeeded else 1, + status="updated" if update_succeeded else "update_failed", + applied=arguments.apply, + ) + + if arguments.command == "dev": + repository = cli_methods.repository_status(root) + exit_code = cli_methods.run_developer_action( + root, + configuration, + arguments, + layout.debug_logs, + application_logger, + repository, + ) + return finish_command( + exit_code, + status="checks_completed" if exit_code == 0 else "checks_failed", + ) + + request = cli_methods.request( + arguments, + configuration.runtime.default_max_workers, + configuration.runtime.blocked_capabilities, + ) + application_logger.event( + "INFO", + "plan_requested", + source="logicytics.cli.commands", + profile=request.profile, + include_count=len(request.include), + exclude_count=len(request.exclude), + blocked_capabilities=len(request.blocked_capabilities), + enable_plugins=request.enable_plugins, + max_workers=request.max_workers, + ) + plan = build_plan( + report, + request, + ) + application_logger.event( + "INFO", + "plan_created", + source="logicytics.cli.commands", + collectors=len(plan.collectors), + fingerprint=plan.fingerprint, + ) + + if arguments.command == "plan": + application_logger.box( + "Collection plan", + tuple(candidate.metadata.id for candidate in plan.collectors if candidate.metadata is not None) + or ("No collectors were selected.",), + ) + application_logger.event( + "INFO", + "plan_rendered", + source="logicytics.cli.commands", + collectors=len(plan.collectors), + ) + + return finish_command(0, status="plan_rendered", collectors=len(plan.collectors)) + + outcome = RunSupervisor( + root, + configuration, + ).run(plan) + + status_counts: dict[str, int] = {} + for record in outcome.manifest.collectors: + status_counts[record.status] = status_counts.get(record.status, 0) + 1 + + result_level = { + "succeeded": "INFO", + "partial": "WARNING", + "cancelled": "WARNING", + "failed": "ERROR", + }.get(outcome.manifest.status.value, "CRITICAL") + application_logger.event( + result_level, + "collection_result", + source="logicytics.cli.commands", + status=outcome.manifest.status.value, + collectors=len(outcome.manifest.collectors), + succeeded_count=status_counts.get("succeeded", 0), + skipped_count=status_counts.get("skipped", 0), + failed_count=status_counts.get("failed", 0), + cancelled_count=status_counts.get("cancelled", 0), + ) + + for record in outcome.manifest.collectors: + if record.status == "succeeded": + continue + severity = "ERROR" if record.status == "failed" else "WARNING" + if record.failure is not None: + application_logger.event( + severity, + "collector_result", + source=collector_log_source(record.id) or "logicytics.cli.commands", + collector_id=record.id, + status=record.status, + summary=record.summary or "not-finished", + duration_seconds=record.duration_seconds or 0.0, + operation=str(record.failure.get("operation", "unknown")), + platform_error=str(record.failure.get("platform_error", "unknown")), + remediation=str(record.failure.get("remediation", "none")), + ) + continue + application_logger.event( + severity, + "collector_result", + source=collector_log_source(record.id) or "logicytics.cli.commands", + collector_id=record.id, + status=record.status, + summary=record.summary or "not-finished", + duration_seconds=record.duration_seconds or 0.0, + ) + + if outcome.manifest.status.value == "succeeded": + result_lines = [f"Collectors: {len(outcome.manifest.collectors)}"] + if outcome.manifest.package and "path" in outcome.manifest.package: + package_sha_path = outcome.manifest.package.get("sha256_path", "unavailable") + result_lines.extend( + f"Package: {outcome.manifest.package['path']}\nSHA-256: {package_sha_path}".splitlines() + ) + result_lines.append(f"Run: {outcome.manifest_path}") + application_logger.box("Collection result", result_lines) + + exit_code = 0 if outcome.manifest.status.value == "succeeded" else 1 + + if arguments.interactive: + try: + input("Press Enter to exit...") + except EOFError: + pass + + return finish_command( + exit_code, + status=outcome.manifest.status.value, + run_id=outcome.manifest.run_id, + collectors=len(outcome.manifest.collectors), + ) + + except KeyboardInterrupt: + if application_logger is not None: + with contextlib.suppress(OSError): + application_logger.incomplete_progress("Preflight") + + with contextlib.suppress(OSError): + application_logger.event( + "WARNING", + "command_cancelled", + source="logicytics.cli.commands", + command=str(arguments.command), + exit_code=130, + console=False, + ) + + application_logger.box( + "Command cancelled", + ("Interrupted by user.",), + ) + + else: + ApplicationLogger.render_section( + sys.stderr, + "Command cancelled", + ("Interrupted by user.",), + ) + + return 130 + except ( + LogicyticsError, + OSError, + PermissionError, + ValueError, + ) as error: + error_title = "Configuration validation failed" if arguments.command == "config" else "Command error" + if application_logger is not None: + with contextlib.suppress(OSError): + application_logger.event( + "EXCEPTION", + str(error), + source="logicytics.cli.commands", + command=str(arguments.command), + error_type=type(error).__name__, + ) + + if command_started_at is not None: + application_logger.event( + "ERROR", + "command_finished", + source="logicytics.cli.commands", + command=str(arguments.command), + exit_code=2, + status="failed", + duration_seconds=round(perf_counter() - command_started_at, 3), + error_type=type(error).__name__, + ) + + if application_logger is not None: + application_logger.box(error_title, (str(error),)) + else: + ApplicationLogger.render_section( + sys.stderr, + error_title, + (str(error),), + ) + return 2 + + +cli_methods = CLI() +if __name__ == "__main__": + try: + with terminal_lifecycle(): + raise SystemExit(main()) + except KeyboardInterrupt: + ApplicationLogger.render_section(sys.stderr, "Command cancelled", ("Interrupted by user.",)) + raise SystemExit(130) diff --git a/logicytics/cli/installer.py b/logicytics/cli/installer.py new file mode 100644 index 00000000..bc2687ff --- /dev/null +++ b/logicytics/cli/installer.py @@ -0,0 +1,52 @@ +"""Bootstrap the supported virtual environment and root YAML configuration.""" + +from __future__ import annotations + +import sys +import venv +from pathlib import Path + +from logicytics.module.configuration import write_default_configuration +from logicytics.module.logging import ApplicationLogger, HumanArgumentParser +from logicytics.module.terminal import terminal_lifecycle + + +def project_root() -> Path: + """Return the repository root without importing the normal application CLI.""" + return Path(__file__).resolve().parents[2] + + +def main(argv: list[str] | None = None) -> int: + """Create the local virtual environment and initialize YAML configuration.""" + parser = HumanArgumentParser(description="Prepare a Logicytics installation.") + parser.add_argument("--environment", type=Path, default=Path(".venv"), help="Virtual environment directory.") + parser.add_argument("--overwrite-config", action="store_true", help="Replace the root YAML template.") + arguments = parser.parse_args(argv) + root = project_root() + environment = arguments.environment if arguments.environment.is_absolute() else root / arguments.environment + environment_status: str + if not environment.exists(): + venv.EnvBuilder(with_pip=True).create(environment) + environment_status = f"Created virtual environment: {environment}" + else: + environment_status = f"Using existing virtual environment: {environment}" + configuration = write_default_configuration(root, overwrite=arguments.overwrite_config) + ApplicationLogger.render_section( + sys.stdout, + "Logicytics installation", + ( + environment_status, + f"Configuration ready: {configuration}", + f"Next step: {environment / 'Scripts' / 'python.exe'} -m logicytics preflight", + ), + ) + return 0 + + +if __name__ == "__main__": + try: + with terminal_lifecycle(): + raise SystemExit(main()) + except KeyboardInterrupt: + ApplicationLogger.render_section(sys.stderr, "Command cancelled", ("Interrupted by user.",)) + raise SystemExit(130) diff --git a/logicytics/cli/tests.py b/logicytics/cli/tests.py new file mode 100644 index 00000000..c46d8fce --- /dev/null +++ b/logicytics/cli/tests.py @@ -0,0 +1,168 @@ +"""Discover and execute the complete Logicytics test suite.""" + +from __future__ import annotations + +import contextlib +import io +import sys +import unittest +from collections.abc import Callable +from pathlib import Path +from typing import TYPE_CHECKING, TypeAlias, cast + +if TYPE_CHECKING: + from unittest.runner import _ResultClassType + +from logicytics.module.terminal import isolated_terminal_lifecycle, terminal_lifecycle +from logicytics.module.virtual_environment import ( + is_running_in_virtual_environment, + render_virtual_environment_error, +) + +ProgressCallback: TypeAlias = Callable[[int, int, str], None] + + +def _progress_result_class(total: int, progress: ProgressCallback) -> type[unittest.TextTestResult]: + """Create a result class that advances one dependency-free test progress bar.""" + + class ProgressResult(unittest.TextTestResult): + """Report each completed test without changing unittest result semantics.""" + + completed = 0 + current = "" + + def startTest(self, test: unittest.case.TestCase) -> None: + self.current = test.id() + progress(self.completed, total, self.current) + super().startTest(test) + + def stopTest(self, test: unittest.case.TestCase) -> None: + super().stopTest(test) + self.completed += 1 + progress(self.completed, total, self.current) + + return ProgressResult + + +def _failure_details(result: unittest.TestResult) -> tuple[tuple[str, str, str], ...]: + """Extract category, test ID, and final actionable line for every failed test.""" + details: list[tuple[str, str, str]] = [] + for category, failures in (("Failure", result.failures), ("Error", result.errors)): + for test, traceback_text in failures: + test_name = test.id() if hasattr(test, "id") else str(test) + tail = next((line.strip() for line in reversed(traceback_text.splitlines()) if line.strip()), + "No detail provided") + details.append((category, test_name, tail)) + return tuple(details) + + +def _failure_rows(result: unittest.TestResult) -> tuple[str, ...]: + """Summarize every failed test without echoing a full unittest traceback wall.""" + return tuple( + item + for category, test_name, reason in _failure_details(result) + for item in (f"{category}: {test_name}", f"Reason: {reason}") + ) + + +def project_root() -> Path: + """Return the repository root containing the dynamically discovered test package.""" + return Path(__file__).resolve().parents[2] + + +def main(argv: list[str] | None = None) -> int: + """Dynamically discover tests and return a CI-appropriate result code.""" + root = project_root() + if not is_running_in_virtual_environment(): + render_virtual_environment_error(sys.stderr, root) + return 2 + from logicytics.module.configuration import load_config + from logicytics.module.errors import PlanError + from logicytics.module.logging import ( + ApplicationLogger, + HumanArgumentParser, + get_application_logger, + ) + from logicytics.module.output_layout import ensure_output_layout + + parser = HumanArgumentParser(description="Run all discovered Logicytics tests.") + parser.add_argument("--verbosity", type=int, choices=(0, 1, 2), default=2) + arguments = parser.parse_args(argv) + try: + configuration = load_config(root) + layout = ensure_output_layout(configuration.runtime.output_root) + except (OSError, PlanError) as error: + ApplicationLogger.render_section( + sys.stderr, + "Test runner error", + (f"Unable to prepare test presentation: {error}",), + ) + return 2 + logger = get_application_logger(layout.application_log, configuration.logging) + debug = str(configuration.logging.level).upper() == "DEBUG" + transcript = io.StringIO() + suite = unittest.defaultTestLoader.discover(str(root / "tests"), top_level_dir=str(root)) + total = suite.countTestCases() + logger.event( + "INFO", + "test_suite_started", + source="logicytics.cli.tests", + tests=total, + mode="debug" if debug else "normal", + ) + with isolated_terminal_lifecycle(): + if debug: + result = unittest.TextTestRunner(stream=sys.stderr, verbosity=arguments.verbosity).run(suite) + else: + def progress(checked: int, count: int, current: str) -> None: + """Render the current normal-mode test position without a third-party dependency.""" + logger.progress("Tests", checked, count, current) + + result_class = _progress_result_class(total, progress) + with contextlib.redirect_stdout(transcript), contextlib.redirect_stderr(transcript): + result = unittest.TextTestRunner( + stream=transcript, + verbosity=0, + # The stdlib decorates its stream before invoking this result factory. + resultclass=cast("_ResultClassType", result_class), + ).run(suite) + summary = ( + f"Tests: {result.testsRun}", + f"Failures: {len(result.failures)}", + f"Errors: {len(result.errors)}", + f"Skipped: {len(result.skipped)}", + f"Result: {'passed' if result.wasSuccessful() else 'failed'}", + ) + if not result.wasSuccessful(): + logger.event( + "ERROR", + "test_suite_failed", + source="logicytics.cli.tests", + failures=len(result.failures), + errors=len(result.errors), + ) + for category, test_name, reason in _failure_details(result): + logger.event( + "ERROR", + "test_case_failed", + source="logicytics.cli.tests", + category=category, + test=test_name, + reason=reason, + ) + logger.box("Test failures", _failure_rows(result) or ("The test runner did not provide failure details.",)) + else: + logger.event("INFO", "test_suite_finished", source="logicytics.cli.tests", tests=result.testsRun) + logger.box("Test suite", summary) + return 0 if result.wasSuccessful() else 1 + + +if __name__ == "__main__": + try: + with terminal_lifecycle(): + raise SystemExit(main()) + except KeyboardInterrupt: + from logicytics.module.logging import ApplicationLogger + + ApplicationLogger.render_section(sys.stderr, "Command cancelled", ("Interrupted by user.",)) + raise SystemExit(130) diff --git a/logicytics/global/__init__.py b/logicytics/global/__init__.py new file mode 100644 index 00000000..22912bab --- /dev/null +++ b/logicytics/global/__init__.py @@ -0,0 +1,6 @@ +"""Globally available, host-bound infrastructure loaded only by explicit consumers. + +The directory name follows the product architecture. Because ``global`` is a +Python keyword, consumers load submodules through the public root exports or +``importlib.import_module("logicytics.global.")``. +""" diff --git a/logicytics/global/ctypes_collector.py b/logicytics/global/ctypes_collector.py new file mode 100644 index 00000000..13dc09ec --- /dev/null +++ b/logicytics/global/ctypes_collector.py @@ -0,0 +1,499 @@ +"""Centralized, typed Windows ctypes bindings used by Logicytics.""" + +from __future__ import annotations + +import ctypes +import sys +from ctypes import wintypes +from typing import Final, TypeAlias + +if sys.platform != "win32": + raise RuntimeError("ctypes_collector is only supported on Windows") + +# ctypes aliases + +DWORD: TypeAlias = wintypes.DWORD +HANDLE: TypeAlias = wintypes.HANDLE +BOOL: TypeAlias = wintypes.BOOL +ULONG: TypeAlias = ctypes.c_ulong +SIZE_T: TypeAlias = ctypes.c_size_t +ULARGE_INTEGER: TypeAlias = ctypes.c_ulonglong +FILETIME: TypeAlias = wintypes.FILETIME +LPVOID: TypeAlias = wintypes.LPVOID +WORD: TypeAlias = wintypes.WORD + +LPDWORD = ctypes.POINTER(DWORD) +LPULARGE_INTEGER = ctypes.POINTER(ULARGE_INTEGER) + +# Windows constants + +PROCESS_QUERY_LIMITED_INFORMATION: Final = 0x1000 +STILL_ACTIVE: Final = 259 + + +# Windows structures + + +class ProcessMemoryCounters(ctypes.Structure): + """Windows PROCESS_MEMORY_COUNTERS structure.""" + + _fields_ = [ + ("cb", DWORD), + ("PageFaultCount", DWORD), + ("PeakWorkingSetSize", SIZE_T), + ("WorkingSetSize", SIZE_T), + ("QuotaPeakPagedPoolUsage", SIZE_T), + ("QuotaPagedPoolUsage", SIZE_T), + ("QuotaPeakNonPagedPoolUsage", SIZE_T), + ("QuotaNonPagedPoolUsage", SIZE_T), + ("PagefileUsage", SIZE_T), + ("PeakPagefileUsage", SIZE_T), + ] + + +class MemoryBasicInformation(ctypes.Structure): + """Windows MEMORY_BASIC_INFORMATION structure.""" + + _fields_ = [ + ("BaseAddress", LPVOID), + ("AllocationBase", LPVOID), + ("AllocationProtect", DWORD), + ("PartitionId", WORD), + ("RegionSize", SIZE_T), + ("State", DWORD), + ("Protect", DWORD), + ("Type", DWORD), + ] + + +class MemoryStatus(ctypes.Structure): + """Windows MEMORYSTATUSEX structure.""" + + _fields_ = [ + ("dwLength", ULONG), + ("dwMemoryLoad", ULONG), + ("ullTotalPhys", ULARGE_INTEGER), + ("ullAvailPhys", ULARGE_INTEGER), + ("ullTotalPageFile", ULARGE_INTEGER), + ("ullAvailPageFile", ULARGE_INTEGER), + ("ullTotalVirtual", ULARGE_INTEGER), + ("ullAvailVirtual", ULARGE_INTEGER), + ("ullAvailExtendedVirtual", ULARGE_INTEGER), + ] + + +# DLL handles + +kernel32 = ctypes.WinDLL("kernel32", use_last_error=True) +advapi32 = ctypes.WinDLL("advapi32", use_last_error=True) +psapi = ctypes.WinDLL("psapi", use_last_error=True) + +# kernel32 bindings + +_OpenProcess = kernel32.OpenProcess +_OpenProcess.argtypes = [ + DWORD, + BOOL, + DWORD, +] +_OpenProcess.restype = HANDLE + +_GetExitCodeProcess = kernel32.GetExitCodeProcess +_GetExitCodeProcess.argtypes = [ + HANDLE, + LPDWORD, +] +_GetExitCodeProcess.restype = BOOL + +_CloseHandle = kernel32.CloseHandle +_CloseHandle.argtypes = [ + HANDLE, +] +_CloseHandle.restype = BOOL + +_GetCurrentProcess = kernel32.GetCurrentProcess +_GetCurrentProcess.argtypes = [] +_GetCurrentProcess.restype = HANDLE + +_GetLogicalDrives = kernel32.GetLogicalDrives +_GetLogicalDrives.argtypes = [] +_GetLogicalDrives.restype = DWORD + +_GetDriveTypeW = kernel32.GetDriveTypeW +_GetDriveTypeW.argtypes = [ + wintypes.LPCWSTR, +] +_GetDriveTypeW.restype = wintypes.UINT + +_GetVolumeInformationW = kernel32.GetVolumeInformationW +_GetVolumeInformationW.argtypes = [ + wintypes.LPCWSTR, + wintypes.LPWSTR, + DWORD, + LPDWORD, + LPDWORD, + LPDWORD, + wintypes.LPWSTR, + DWORD, +] +_GetVolumeInformationW.restype = BOOL + +_GetDiskFreeSpaceExW = kernel32.GetDiskFreeSpaceExW +_GetDiskFreeSpaceExW.argtypes = [ + wintypes.LPCWSTR, + LPULARGE_INTEGER, + LPULARGE_INTEGER, + LPULARGE_INTEGER, +] +_GetDiskFreeSpaceExW.restype = BOOL + +_GlobalMemoryStatusEx = kernel32.GlobalMemoryStatusEx +_GlobalMemoryStatusEx.argtypes = [ + ctypes.POINTER(MemoryStatus), +] +_GlobalMemoryStatusEx.restype = BOOL + +_VirtualQuery = kernel32.VirtualQuery +_VirtualQuery.argtypes = [ + LPVOID, + ctypes.POINTER(MemoryBasicInformation), + SIZE_T, +] +_VirtualQuery.restype = SIZE_T + +# advapi32 bindings + +_RegQueryInfoKeyW = advapi32.RegQueryInfoKeyW +_RegQueryInfoKeyW.argtypes = [ + wintypes.HKEY, + wintypes.LPWSTR, + LPDWORD, + LPDWORD, + LPDWORD, + LPDWORD, + LPDWORD, + LPDWORD, + LPDWORD, + LPDWORD, + LPDWORD, + ctypes.POINTER(FILETIME), +] +_RegQueryInfoKeyW.restype = wintypes.LONG + +# psapi bindings + +_GetProcessMemoryInfo = psapi.GetProcessMemoryInfo +_GetProcessMemoryInfo.argtypes = [ + HANDLE, + ctypes.POINTER(ProcessMemoryCounters), + DWORD, +] +_GetProcessMemoryInfo.restype = BOOL + +_GetMappedFileNameW = psapi.GetMappedFileNameW +_GetMappedFileNameW.argtypes = [ + HANDLE, + LPVOID, + wintypes.LPWSTR, + DWORD, +] +_GetMappedFileNameW.restype = DWORD + + +# Process wrappers + + +def open_process( + process_id: int, + *, + access: int = PROCESS_QUERY_LIMITED_INFORMATION, + inherit_handle: bool = False, +) -> HANDLE | None: + """Open a Windows process and return its handle, or None on failure.""" + + handle = _OpenProcess( + access, + inherit_handle, + process_id, + ) + + if not handle: + return None + + return handle + + +def get_current_process() -> HANDLE: + """Return the pseudo-handle for the current process.""" + + return _GetCurrentProcess() + + +def get_exit_code_process(handle: HANDLE) -> int | None: + """Return a process exit code, or None if querying it failed.""" + + exit_code = DWORD() + + if not _GetExitCodeProcess( + handle, + ctypes.byref(exit_code), + ): + return None + + return int(exit_code.value) + + +def close_handle(handle: HANDLE) -> bool: + """Close a Windows kernel handle.""" + + return bool(_CloseHandle(handle)) + + +def get_process_memory_info( + process: HANDLE, + counters: ProcessMemoryCounters, +) -> bool: + """Populate memory counters for a process.""" + + counters.cb = ctypes.sizeof(counters) + + return bool( + _GetProcessMemoryInfo( + process, + ctypes.byref(counters), + ctypes.sizeof(counters), + ) + ) + + +# Virtual-memory wrappers + + +def virtual_query( + address: int, + memory: MemoryBasicInformation, +) -> int: + """Query the virtual-memory region containing an address.""" + + return int( + _VirtualQuery( + ctypes.c_void_p(address), + ctypes.byref(memory), + ctypes.sizeof(memory), + ) + ) + + +def get_mapped_file_name( + process: HANDLE, + address: int, + buffer_size: int = 32_768, +) -> str | None: + """Return the mapped filename associated with a process address.""" + + buffer = ctypes.create_unicode_buffer(buffer_size) + + length = _GetMappedFileNameW( + process, + ctypes.c_void_p(address), + buffer, + buffer_size, + ) + + if not length: + return None + + return buffer.value[: int(length)] + + +def pointer_value(pointer: LPVOID) -> int | None: + """Return the integer value represented by a Windows pointer.""" + + return ctypes.cast( + pointer, + ctypes.c_void_p, + ).value + + +# Drive and volume wrappers + + +def get_logical_drives() -> int: + """Return the bitmask identifying available Windows logical drives.""" + + mask = _GetLogicalDrives() + + if mask == 0: + raise ctypes.WinError(ctypes.get_last_error()) + + return int(mask) + + +def get_drive_type(root: str) -> int: + """Return the Windows drive type code for a filesystem root.""" + + return int(_GetDriveTypeW(root)) + + +def get_volume_information( + root: str, + label: ctypes.Array[ctypes.c_wchar], + serial: DWORD, + maximum_component_length: DWORD, + flags: DWORD, + filesystem: ctypes.Array[ctypes.c_wchar], +) -> bool: + """Populate volume metadata for a filesystem root.""" + + return bool( + _GetVolumeInformationW( + root, + label, + len(label), + ctypes.byref(serial), + ctypes.byref(maximum_component_length), + ctypes.byref(flags), + filesystem, + len(filesystem), + ) + ) + + +def get_disk_free_space( + root: str, + available: ULARGE_INTEGER, + total: ULARGE_INTEGER, + free: ULARGE_INTEGER, +) -> bool: + """Populate disk space counters for a filesystem root.""" + + return bool( + _GetDiskFreeSpaceExW( + root, + ctypes.byref(available), + ctypes.byref(total), + ctypes.byref(free), + ) + ) + + +# Memory wrappers + + +def global_memory_status() -> MemoryStatus: + """Return current Windows physical and virtual memory statistics.""" + + status = MemoryStatus() + status.dwLength = ctypes.sizeof(status) + + if not _GlobalMemoryStatusEx(ctypes.byref(status)): + raise ctypes.WinError(ctypes.get_last_error()) + + return status + + +# Registry wrappers + + +def query_registry_key_info( + key_handle: int, + timestamp: FILETIME, +) -> int: + """Query registry key metadata and populate its last-write timestamp.""" + + return int( + _RegQueryInfoKeyW( + wintypes.HKEY(key_handle), + None, + None, + None, + None, + None, + None, + None, + None, + None, + None, + ctypes.byref(timestamp), + ) + ) + + +# ctypes factories + + +def create_unicode_buffer(size: int) -> ctypes.Array[ctypes.c_wchar]: + """Create a mutable Windows Unicode buffer.""" + + return ctypes.create_unicode_buffer(size) + + +def dword(value: int = 0) -> DWORD: + """Create a Windows DWORD value.""" + + return DWORD(value) + + +def ulong(value: int = 0) -> ULONG: + """Create an unsigned long ctypes value.""" + + return ULONG(value) + + +def ularge_integer(value: int = 0) -> ULARGE_INTEGER: + """Create an unsigned 64-bit Windows integer.""" + + return ULARGE_INTEGER(value) + + +def filetime() -> FILETIME: + """Create an empty Windows FILETIME structure.""" + + return FILETIME() + + +__all__ = [ + # Types + "BOOL", + "DWORD", + "FILETIME", + "HANDLE", + "LPDWORD", + "LPULARGE_INTEGER", + "LPVOID", + "MemoryBasicInformation", + "MemoryStatus", + "ProcessMemoryCounters", + "SIZE_T", + "ULARGE_INTEGER", + "ULONG", + "WORD", + # Constants + "PROCESS_QUERY_LIMITED_INFORMATION", + "STILL_ACTIVE", + # Process API + "close_handle", + "get_current_process", + "get_exit_code_process", + "get_process_memory_info", + "open_process", + # Virtual memory API + "get_mapped_file_name", + "pointer_value", + "virtual_query", + # Drive and volume API + "get_disk_free_space", + "get_drive_type", + "get_logical_drives", + "get_volume_information", + # Memory API + "global_memory_status", + # Registry API + "query_registry_key_info", + # ctypes factories + "create_unicode_buffer", + "dword", + "filetime", + "ularge_integer", + "ulong", +] diff --git a/logicytics/module/__init__.py b/logicytics/module/__init__.py new file mode 100644 index 00000000..c6ac7fee --- /dev/null +++ b/logicytics/module/__init__.py @@ -0,0 +1 @@ +"""Application services that plan, execute, package, and maintain Logicytics runs.""" diff --git a/logicytics/module/api.py b/logicytics/module/api.py new file mode 100644 index 00000000..2b72d1c9 --- /dev/null +++ b/logicytics/module/api.py @@ -0,0 +1,432 @@ +"""Small, side-effect-free public application API for v4 collection runs.""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +import sys +from dataclasses import dataclass, replace +from datetime import datetime +from math import isfinite +from pathlib import Path, PurePosixPath +from typing import Any + +from logicytics.module.contracts import Artifact, CollectorStatus, RunRequest, RunStatus +from logicytics.module.configuration import AppConfig, load_config +from logicytics.module.discovery import preflight +from logicytics.module.errors import ArtifactError, PlanError +from logicytics.module.manifest import MANIFEST_SCHEMA_VERSION +from logicytics.module.output_layout import locate_run_directory +from logicytics.module.planner import RunPlan, build_plan +from logicytics.module.runtime import RunOutcome, RunSupervisor + +_RUN_ID = re.compile(r"run-[0-9a-f]{32}") +_ARTIFACT_ID = re.compile(r"artifact\.[0-9a-f]{32}") +_COLLECTOR_ID = re.compile(r"(?:core|plugin)\.[a-z][a-z0-9_]*(?:\.[a-z][a-z0-9_]*)?") +_MAXIMUM_ARTIFACT_READ_BYTES = 64 * 1024 * 1024 +_DEFAULT_ARTIFACT_READ_BYTES = 16 * 1024 * 1024 + + +@dataclass(frozen=True, slots=True) +class CollectorFailureSnapshot: + """Immutable actionable failure details from one persisted collector result.""" + + collector_id: str + operation: str + platform_error: str + remediation: str + retry_safe: bool + + +@dataclass(frozen=True, slots=True) +class CollectorSnapshot: + """Immutable public lifecycle details for one collector in a persisted run.""" + + collector_id: str + status: str + started_at: str | None + finished_at: str | None + duration_seconds: float | None + summary: str | None + errors: tuple[str, ...] + failure: CollectorFailureSnapshot | None + + +@dataclass(frozen=True, slots=True) +class RunSnapshot: + """Immutable, manifest-validated view of one current or completed run.""" + + run_id: str + status: RunStatus + run_directory: Path + manifest_path: Path + collectors: tuple[CollectorSnapshot, ...] + artifacts: tuple[Artifact, ...] + finished_at: str | None + + +def _unique_manifest_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + """Reject ambiguous persisted manifest keys instead of replacing prior values.""" + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate run manifest key {key!r}") + result[key] = value + return result + + +def load_configuration(project_root: Path | str, config_path: Path | str | None = None) -> AppConfig: + """Load strictly validated configuration without creating collection output.""" + root = Path(project_root).resolve() + selected = None if config_path is None else Path(config_path) + if selected is not None and not selected.is_absolute(): + selected = root / selected + return load_config(root, selected) + + +def _configuration( + project_root: Path | str, + configuration: AppConfig | None, + config_path: Path | str | None, +) -> tuple[Path, AppConfig]: + """Resolve one unambiguous, typed configuration source.""" + root = Path(project_root).resolve() + if configuration is not None and config_path is not None: + raise PlanError("provide either configuration or config_path, not both") + if configuration is not None and not isinstance(configuration, AppConfig): + raise PlanError("configuration must be a validated AppConfig instance") + return root, configuration if configuration is not None else load_configuration(root, config_path) + + +def plan_run( + project_root: Path | str, + request: RunRequest, + *, + configuration: AppConfig | None = None, + config_path: Path | str | None = None, +) -> RunPlan: + """Preflight and resolve a strict run without starting collectors or writing output.""" + root, settings = _configuration(project_root, configuration, config_path) + if not isinstance(request, RunRequest): + raise PlanError("request must be an immutable RunRequest instance") + blocked_capabilities = tuple(dict.fromkeys((*settings.runtime.blocked_capabilities, *request.blocked_capabilities))) + request = replace(request, blocked_capabilities=blocked_capabilities) + if request.max_workers > settings.runtime.maximum_workers: + raise PlanError("requested workers exceed configured maximum_workers") + return build_plan(preflight(root, configuration_hash=settings.fingerprint()), request) + + +def run_collection( + project_root: Path | str, + request: RunRequest, + *, + configuration: AppConfig | None = None, + config_path: Path | str | None = None, +) -> RunOutcome: + """Execute only a strictly validated, authorized, independently isolated run.""" + root, settings = _configuration(project_root, configuration, config_path) + plan = plan_run(root, request, configuration=settings) + return RunSupervisor(root, settings).run(plan) + + +def _artifact(raw: object, collector_id: str) -> Artifact: + """Reconstruct one strict artifact while enforcing collector-owned paths.""" + if not isinstance(raw, dict): + raise PlanError("run manifest contains a malformed registered artifact") + values = dict(raw) + transformations = values.get("transformations") + if not isinstance(transformations, list): + raise PlanError("run manifest artifact transformations must be an array") + try: + artifact = Artifact.from_dict(values) + except (TypeError, ValueError) as error: + raise PlanError(f"run manifest contains an invalid registered artifact: {error}") from error + relative = PurePosixPath(artifact.relative_path) + owner = collector_id.replace(".", "_") + if ( + artifact.collector_id != collector_id + or "\\" in artifact.relative_path + or relative.is_absolute() + or ".." in relative.parts + or relative.as_posix() != artifact.relative_path + or len(relative.parts) < 2 + or relative.parts[0] != owner + or artifact.name != relative.name + ): + raise PlanError("run manifest artifact violates collector ownership or path boundaries") + return artifact + + +def _manifest_timestamp(value: object, field: str, *, optional: bool = False) -> str | None: + """Require timezone-aware persisted timestamps, allowing null only when declared.""" + if value is None and optional: + return None + if not isinstance(value, str): + raise PlanError(f"run manifest {field} must be a timezone-aware timestamp") + try: + timestamp = datetime.fromisoformat(value) + except ValueError as error: + raise PlanError(f"run manifest {field} must be a timezone-aware timestamp") from error + if timestamp.tzinfo is None: + raise PlanError(f"run manifest {field} must be a timezone-aware timestamp") + return value + + +def _collector_snapshot(record: dict[str, Any], collector_id: str, status: str) -> CollectorSnapshot: + """Validate and freeze persisted per-collector timing and failure information.""" + started_at = _manifest_timestamp(record.get("started_at"), "collector started_at", optional=True) + finished_at = _manifest_timestamp(record.get("finished_at"), "collector finished_at", optional=True) + duration = record.get("duration_seconds") + if duration is not None and ( + not isinstance(duration, (int, float)) or isinstance(duration, bool) or not isfinite( + duration) or duration < 0 + ): + raise PlanError("run manifest collector duration_seconds must be a finite non-negative number or null") + summary = record.get("summary") + if summary is not None and (not isinstance(summary, str) or not summary.strip()): + raise PlanError("run manifest collector summary must be a non-empty string or null") + raw_errors = record.get("errors") + if not isinstance(raw_errors, list) or not all(isinstance(error, str) and error.strip() for error in raw_errors): + raise PlanError("run manifest collector errors must be an array of non-empty strings") + terminal = status in {item.value for item in CollectorStatus} + if terminal and (finished_at is None or summary is None): + raise PlanError("finalized collector records require finished_at and summary") + if not terminal and (finished_at is not None or duration is not None): + raise PlanError("planned or running collector records cannot contain terminal timing") + if started_at is None and duration is not None: + raise PlanError("collector duration_seconds requires a started_at timestamp") + + raw_failure = record.get("failure") + failure: CollectorFailureSnapshot | None = None + if raw_failure is not None: + required = {"collector_id", "operation", "platform_error", "remediation", "retry_safe"} + if not isinstance(raw_failure, dict) or set(raw_failure) != required: + raise PlanError("run manifest collector failure must contain the complete actionable failure contract") + actionable_fields = ("operation", "platform_error", "remediation") + + has_valid_collector_id = raw_failure.get("collector_id") == collector_id + has_valid_actionable_details = all( + isinstance(raw_failure.get(field), str) and raw_failure[field].strip() for field in actionable_fields + ) + has_valid_retry_flag = isinstance(raw_failure.get("retry_safe"), bool) + + if not (has_valid_collector_id and has_valid_actionable_details and has_valid_retry_flag): + raise PlanError("run manifest collector failure contains invalid actionable details") + failure = CollectorFailureSnapshot( + collector_id=collector_id, + operation=raw_failure["operation"], + platform_error=raw_failure["platform_error"], + remediation=raw_failure["remediation"], + retry_safe=raw_failure["retry_safe"], + ) + if (status == CollectorStatus.FAILED.value) != (failure is not None): + raise PlanError("run manifest collector failure details must match failed status") + return CollectorSnapshot( + collector_id=collector_id, + status=status, + started_at=started_at, + finished_at=finished_at, + duration_seconds=None if duration is None else float(duration), + summary=summary, + errors=tuple(raw_errors), + failure=failure, + ) + + +def query_run( + project_root: Path | str, + run_id: str, + *, + configuration: AppConfig | None = None, + config_path: Path | str | None = None, +) -> RunSnapshot: + """Read a run-owned manifest and return immutable, ownership-validated status.""" + _, settings = _configuration(project_root, configuration, config_path) + if not isinstance(run_id, str) or _RUN_ID.fullmatch(run_id) is None: + raise PlanError("run_id must be a canonical run identifier") + output_root = settings.runtime.output_root.resolve() + try: + run_directory = locate_run_directory(output_root / "run", run_id) + manifest_path = run_directory / "manifest.json" + if run_directory.resolve(strict=True) != run_directory: + raise PlanError("run directory escapes its configured output root") + if manifest_path.resolve(strict=True) != manifest_path or not manifest_path.is_file(): + raise PlanError("run manifest escapes its run-owned directory") + payload = json.loads(manifest_path.read_text(encoding="utf-8"), object_pairs_hook=_unique_manifest_object) + except (OSError, UnicodeDecodeError, ValueError) as error: + raise PlanError(f"unable to read run manifest for {run_id}: {error}") from error + if not isinstance(payload, dict) or payload.get("run_id") != run_id: + raise PlanError("run manifest identity does not match its owned directory") + manifest_schema_version = payload.get("manifest_schema_version") + if manifest_schema_version is None: + raise PlanError("run manifest is missing schema_version") + if ( + not isinstance(manifest_schema_version, int) + or isinstance(manifest_schema_version, bool) + or manifest_schema_version != MANIFEST_SCHEMA_VERSION + ): + raise PlanError( + f"unsupported run manifest schema_version {manifest_schema_version!r}; expected {MANIFEST_SCHEMA_VERSION}") + + _manifest_timestamp(payload.get("requested_at"), "requested_at") + if not isinstance(payload.get("status"), str): + raise PlanError("run manifest contains an unsupported run status") + try: + status = RunStatus(payload.get("status")) + except ValueError as error: + raise PlanError("run manifest contains an unsupported run status") from error + records = payload.get("collectors") + catalog = payload.get("artifact_catalog") + if not isinstance(records, list) or not isinstance(catalog, list): + raise PlanError("run manifest must contain collector and artifact catalogs") + + collectors: list[CollectorSnapshot] = [] + artifacts: list[Artifact] = [] + collector_ids: set[str] = set() + artifact_ids: set[str] = set() + expected_catalog: list[dict[str, Any]] = [] + valid_collector_statuses = {"planned", "running", *(item.value for item in CollectorStatus)} + for record in records: + if not isinstance(record, dict): + raise PlanError("run manifest contains a malformed collector record") + collector_id = record.get("id") + collector_status = record.get("status") + if ( + not isinstance(collector_id, str) + or _COLLECTOR_ID.fullmatch(collector_id) is None + or collector_id in collector_ids + or not isinstance(collector_status, str) + or collector_status not in valid_collector_statuses + or not isinstance(record.get("artifacts"), list) + ): + raise PlanError("run manifest contains an invalid or duplicate collector record") + collector_ids.add(collector_id) + collectors.append(_collector_snapshot(record, collector_id, collector_status)) + for raw in record["artifacts"]: + artifact = _artifact(raw, collector_id) + if artifact.id in artifact_ids: + raise PlanError("run manifest contains duplicate registered artifact IDs") + artifact_ids.add(artifact.id) + artifacts.append(artifact) + expected_catalog.append({**raw, "producer_status": collector_status}) + if payload.get("resolved_plan") != [item.collector_id for item in collectors]: + raise PlanError("run manifest collector records do not match the resolved plan") + if catalog != expected_catalog: + raise PlanError("run manifest artifact catalog does not match collector-owned evidence") + if status not in {RunStatus.PLANNED, RunStatus.RUNNING} and payload.get("total_artifact_bytes") != sum( + artifact.size_bytes for artifact in artifacts + ): + raise PlanError("run manifest total_artifact_bytes does not match registered evidence") + finished_at = _manifest_timestamp(payload.get("finished_at"), "finished_at", optional=True) + if status not in {RunStatus.PLANNED, RunStatus.RUNNING} and finished_at is None: + raise PlanError("finalized run manifest must contain finished_at") + return RunSnapshot( + run_id=run_id, + status=status, + run_directory=run_directory, + manifest_path=manifest_path, + collectors=tuple(collectors), + artifacts=tuple(artifacts), + finished_at=finished_at, + ) + + +def read_artifact( + project_root: Path | str, + run_id: str, + artifact_id: str, + *, + maximum_bytes: int = _DEFAULT_ARTIFACT_READ_BYTES, + configuration: AppConfig | None = None, + config_path: Path | str | None = None, +) -> bytes: + """Read bounded, registered evidence only after ownership and SHA-256 verification.""" + if not isinstance(artifact_id, str) or _ARTIFACT_ID.fullmatch(artifact_id) is None: + raise ArtifactError("artifact_id must be a canonical registered artifact identifier") + if not isinstance(maximum_bytes, int) or isinstance(maximum_bytes, + bool) or not 1 <= maximum_bytes <= _MAXIMUM_ARTIFACT_READ_BYTES: + raise ArtifactError("maximum_bytes must be an integer from 1 to 67108864") + snapshot = query_run(project_root, run_id, configuration=configuration, config_path=config_path) + artifact = next((item for item in snapshot.artifacts if item.id == artifact_id), None) + if artifact is None: + raise ArtifactError("artifact is not registered in the selected run manifest") + if artifact.size_bytes > maximum_bytes: + raise ArtifactError("registered artifact exceeds the requested bounded read limit") + root = snapshot.run_directory / "artifacts" + source = root.joinpath(*PurePosixPath(artifact.relative_path).parts) + owner = root / artifact.collector_id.replace(".", "_") + try: + if root.resolve(strict=True) != root or owner.resolve(strict=True) != owner: + raise ValueError("artifact ownership roots must not be redirected") + resolved = source.resolve(strict=True) + resolved.relative_to(owner) + if not source.is_file(): + raise ArtifactError("registered artifact must be a regular run-owned file") + with source.open("rb") as stream: + contents = stream.read(maximum_bytes + 1) + except (OSError, ValueError) as error: + raise ArtifactError("registered artifact escapes its collector-owned store or cannot be read") from error + if len(contents) > maximum_bytes or len(contents) != artifact.size_bytes or hashlib.sha256( + contents).hexdigest() != artifact.sha256: + raise ArtifactError("registered artifact failed manifest size or SHA-256 verification") + return contents + + +def open_artifact( + project_root: Path | str, + run_id: str, + artifact_id: str, + *, + configuration: AppConfig | None = None, + config_path: Path | str | None = None, +) -> Path: + """Verify and open one registered artifact with the platform's associated application.""" + if not isinstance(artifact_id, str) or _ARTIFACT_ID.fullmatch(artifact_id) is None: + raise ArtifactError("artifact_id must be a canonical registered artifact identifier") + + snapshot = query_run( + project_root, + run_id, + configuration=configuration, + config_path=config_path, + ) + + artifact = next( + (item for item in snapshot.artifacts if item.id == artifact_id), + None, + ) + if artifact is None: + raise ArtifactError("artifact is not registered in the selected run manifest") + + root = snapshot.run_directory / "artifacts" + source = root.joinpath(*PurePosixPath(artifact.relative_path).parts) + owner = root / artifact.collector_id.replace(".", "_") + + try: + if root.resolve(strict=True) != root or owner.resolve(strict=True) != owner: + raise ValueError("artifact ownership roots must not be redirected") + + resolved = source.resolve(strict=True) + resolved.relative_to(owner) + + if not source.is_file() or source.stat().st_size != artifact.size_bytes: + raise ArtifactError("registered artifact size does not match its manifest") + + digest = hashlib.sha256() + with source.open("rb") as stream: + for block in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(block) + + except (OSError, ValueError) as error: + raise ArtifactError("registered artifact escapes its collector-owned store or cannot be opened") from error + + if digest.hexdigest() != artifact.sha256: + raise ArtifactError("registered artifact failed manifest SHA-256 verification") + + if sys.platform != "win32": + raise OSError("opening artifacts requires Windows") + + os.startfile(source) + return source diff --git a/logicytics/module/artifacts.py b/logicytics/module/artifacts.py new file mode 100644 index 00000000..8efc1d11 --- /dev/null +++ b/logicytics/module/artifacts.py @@ -0,0 +1,222 @@ +"""Evidence artifact registration with workspace-bound path validation.""" + +from __future__ import annotations + +import fnmatch +import hashlib +import os +import re +import shutil +from datetime import UTC, datetime +from pathlib import Path +from threading import RLock +from uuid import uuid4 + +from logicytics.module.contracts import Artifact, ArtifactWriter, EvidenceKind +from logicytics.module.errors import ArtifactError + + +def _is_within(path: Path, parent: Path) -> bool: + """Return whether a resolved path remains below its owning root.""" + try: + path.resolve().relative_to(parent.resolve()) + except ValueError: + return False + return True + + +def sha256_file(path: Path) -> str: + """Hash a file incrementally so large evidence does not fill memory.""" + digest = hashlib.sha256() + with path.open("rb") as stream: + for block in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(block) + return digest.hexdigest() + + +class WorkspaceArtifactWriter(ArtifactWriter): + """Copies approved files from one collector workspace to the run artifact tree.""" + + def __init__( + self, + collector_id: str, + workspace: Path, + artifact_root: Path, + maximum_output_bytes: int, + maximum_artifact_files: int, + *, + source_category: str | None = None, + maximum_artifact_bytes: int | None = None, + run_output_budget_bytes: int | None = None, + cancellation_file: Path | None = None, + allowed_relative_paths: tuple[str, ...] | None = None, + allowed_media_types: tuple[str, ...] | None = None, + ) -> None: + """Initialize a workspace-bound writer with collector and run output quotas.""" + self._collector_id = collector_id + collector_parts = collector_id.split(".", 2) + resolved_category = source_category if source_category is not None else ( + collector_parts[1] if len(collector_parts) > 1 else "") + if not isinstance(resolved_category, str) or not resolved_category.strip(): + raise ArtifactError("artifact source_category must be a non-empty string") + self._source_category = resolved_category + self._workspace = workspace.resolve() + self._artifact_root = artifact_root.resolve() + self._maximum_output_bytes = maximum_output_bytes + self._maximum_artifact_bytes = maximum_output_bytes if maximum_artifact_bytes is None else maximum_artifact_bytes + if ( + not isinstance(self._maximum_artifact_bytes, int) + or isinstance( + self._maximum_artifact_bytes, + bool, + ) + or not 1 <= self._maximum_artifact_bytes <= maximum_output_bytes + ): + raise ArtifactError("maximum_artifact_bytes must be a positive integer within maximum_output_bytes") + self._maximum_artifact_files = maximum_artifact_files + self._run_output_budget_bytes = run_output_budget_bytes + if run_output_budget_bytes is not None and ( + not isinstance(run_output_budget_bytes, int) or isinstance(run_output_budget_bytes, + bool) or run_output_budget_bytes < 1 + ): + raise ArtifactError("run_output_budget_bytes must be a positive integer") + self._cancellation_file = cancellation_file + self._allowed_relative_paths = allowed_relative_paths + self._allowed_media_types = allowed_media_types + self._registration_lock = RLock() + self._bytes_registered = 0 + self._artifacts: list[Artifact] = [] + + @property + def artifacts(self) -> tuple[Artifact, ...]: + """Return artifacts registered by this worker in registration order.""" + with self._registration_lock: + return tuple(self._artifacts) + + def register_file( + self, + source: Path, + *, + media_type: str = "application/octet-stream", + evidence_kind: EvidenceKind = EvidenceKind.DERIVED, + transformations: tuple[str, ...] = (), + ) -> Artifact: + """Serialize destination allocation, quota checks, publication, and catalog updates.""" + with self._registration_lock: + return self._register_file( + source, + media_type=media_type, + evidence_kind=evidence_kind, + transformations=transformations, + ) + + def _register_file( + self, + source: Path, + *, + media_type: str, + evidence_kind: EvidenceKind, + transformations: tuple[str, ...], + ) -> Artifact: + """Copy one validated source into the run artifact tree and catalog it.""" + if not isinstance(media_type, str) or not re.fullmatch( + r"[A-Za-z0-9][A-Za-z0-9!#$&^_.+-]*/[A-Za-z0-9][A-Za-z0-9!#$&^_.+-]*", + media_type, + ): + raise ArtifactError("artifact media_type must be a valid MIME type") + if self._allowed_media_types is not None and media_type not in self._allowed_media_types: + raise ArtifactError(f"artifact media type is outside the collector output contract: {media_type}") + if not isinstance(evidence_kind, EvidenceKind): + raise ArtifactError("artifact evidence_kind must be an EvidenceKind value") + if not isinstance(transformations, tuple) or any( + not isinstance(step, str) or not step.strip() for step in transformations): + raise ArtifactError("artifact transformations must be a tuple of non-empty strings") + self._check_cancellation() + source = source.resolve() + if not source.is_file() or not _is_within(source, self._workspace): + raise ArtifactError("artifacts must be regular files inside the collector workspace") + source_stat = source.stat() + size_bytes = source_stat.st_size + self._check_output_limits(size_bytes) + if len(self._artifacts) >= self._maximum_artifact_files: + raise ArtifactError("collector artifact count exceeds its declared maximum_artifact_files") + + relative_source = source.relative_to(self._workspace) + if self._allowed_relative_paths is not None and not any( + fnmatch.fnmatchcase(relative_source.as_posix(), pattern) for pattern in self._allowed_relative_paths + ): + relative_path = relative_source.as_posix() + raise ArtifactError(f"artifact path is outside the collector output contract: {relative_path}") + safe_collector_id = self._collector_id.replace(".", "_") + destination = self._artifact_root / safe_collector_id / relative_source + if not _is_within(destination, self._artifact_root): + raise ArtifactError("artifact destination must remain inside the run artifact store") + destination.parent.mkdir(parents=True, exist_ok=True) + if not _is_within(destination, self._artifact_root): + raise ArtifactError("artifact destination must remain inside the run artifact store") + if destination.exists(): + destination = destination.with_name(f"{destination.stem}-{uuid4().hex[:8]}{destination.suffix}") + digest = self._copy_artifact(source, destination, size_bytes, source_stat.st_mtime_ns) + artifact = Artifact( + id=f"artifact.{uuid4().hex}", + relative_path=destination.relative_to(self._artifact_root).as_posix(), + sha256=digest, + size_bytes=size_bytes, + media_type=media_type, + collector_id=self._collector_id, + source_category=self._source_category, + collected_at=datetime.now(UTC).isoformat(), + transformations=(*transformations, "copied into run artifact store"), + evidence_kind=evidence_kind, + name=destination.name, + ) + self._bytes_registered += size_bytes + self._artifacts.append(artifact) + return artifact + + def _check_cancellation(self) -> None: + """Raise when the run cancellation marker requests cooperative shutdown.""" + if self._cancellation_file is not None and self._cancellation_file.exists(): + raise ArtifactError("artifact registration was cancelled") + + def _check_output_limits(self, size_bytes: int) -> None: + """Reject publication that would exceed collector or run output budgets.""" + if size_bytes > self._maximum_artifact_bytes: + raise ArtifactError("collector artifact exceeds its declared maximum_artifact_bytes") + if self._bytes_registered + size_bytes > self._maximum_output_bytes: + raise ArtifactError("collector output exceeds its declared maximum_output_bytes") + if self._run_output_budget_bytes is not None and ( + self._bytes_registered + size_bytes > self._run_output_budget_bytes): + raise ArtifactError("run output exceeds configured maximum_run_output_bytes") + + def _copy_artifact( + self, + source: Path, + destination: Path, + expected_size: int, + expected_modified_at: int, + ) -> str: + """Stream bounded evidence to an atomic destination while observing cancellation.""" + temporary = destination.with_name(f".{destination.name}.{uuid4().hex}.tmp") + digest = hashlib.sha256() + copied_bytes = 0 + try: + with source.open("rb") as source_stream, temporary.open("xb") as destination_stream: + while block := source_stream.read(1024 * 1024): + self._check_cancellation() + copied_bytes += len(block) + self._check_output_limits(copied_bytes) + if copied_bytes > expected_size: + raise ArtifactError("artifact source changed during registration") + destination_stream.write(block) + digest.update(block) + final_stat = source.stat() + if copied_bytes != expected_size or final_stat.st_size != expected_size or final_stat.st_mtime_ns != expected_modified_at: + raise ArtifactError("artifact source changed during registration") + shutil.copystat(source, temporary) + self._check_cancellation() + os.replace(temporary, destination) + return digest.hexdigest() + finally: + if temporary.exists(): + temporary.unlink() diff --git a/logicytics/module/command_runner.py b/logicytics/module/command_runner.py new file mode 100644 index 00000000..f63267ac --- /dev/null +++ b/logicytics/module/command_runner.py @@ -0,0 +1,49 @@ +"""Safe, reusable command execution and structured output parsing for collectors.""" + +from __future__ import annotations + +from collections.abc import Iterable +from dataclasses import dataclass + +from logicytics.module.platform_adapters import process_adapter + + +@dataclass(frozen=True, slots=True) +class CommandResult: + """Captured result of a non-shell command invocation.""" + + command: tuple[str, ...] + returncode: int + stdout: str + stderr: str + + +def run_command(command: Iterable[str], *, timeout_seconds: float = 30) -> CommandResult: + """Run an explicit command without a shell and return decoded captured streams.""" + normalized = tuple(str(argument) for argument in command) + if not normalized: + raise ValueError("command must contain at least one argument") + if timeout_seconds <= 0: + raise ValueError("timeout_seconds must be positive") + completed = process_adapter.run(normalized, capture_output=True, check=False, text=True, timeout=timeout_seconds) + return CommandResult(normalized, completed.returncode, completed.stdout, completed.stderr) + + +def parse_level_messages(output: str) -> tuple[tuple[str, str], ...]: + """Parse `LEVEL: message` lines into normalized structured entries.""" + messages: list[tuple[str, str]] = [] + for line in output.splitlines(): + level, separator, message = line.partition(":") + normalized_level = level.strip().upper() + if not separator or normalized_level not in { + "DEBUG", + "INFO", + "WARNING", + "ERROR", + "CRITICAL", + "INTERNAL", + "EXCEPTION", + }: + continue + messages.append((normalized_level, message.strip())) + return tuple(messages) diff --git a/logicytics/module/configuration.py b/logicytics/module/configuration.py new file mode 100644 index 00000000..53b914e9 --- /dev/null +++ b/logicytics/module/configuration.py @@ -0,0 +1,653 @@ +"""Typed local configuration for v4 runs.""" + +from __future__ import annotations + +import hashlib +import json +import re +from collections.abc import Mapping +from dataclasses import asdict, dataclass, field +from math import isfinite +from pathlib import Path +from types import MappingProxyType +from typing import Any + +from logicytics.module.contracts import Capability +from logicytics.module.errors import PlanError +from logicytics.module.redaction import redact_mapping + +SCHEMA_VERSION = 4 +MAXIMUM_CONFIGURATION_BYTES = 2 * 1024 * 1024 +DEFAULT_MAXIMUM_RUN_OUTPUT_BYTES = 4 * 1024 * 1024 * 1024 +MAXIMUM_RUN_OUTPUT_BYTES = 64 * 1024 * 1024 * 1024 +_COLLECTOR_ID = re.compile(r"^(?:core\.[a-z][a-z0-9_]*\.[a-z][a-z0-9_]*|plugin\.[a-z][a-z0-9_]*)$") +_SETTING_NAME = re.compile(r"^[a-z][a-z0-9_]*$") +_ROOT_FIELDS = frozenset( + { + "schema_version", + "runtime", + "interaction", + "maintenance", + "logging", + "collectors", + } +) +_RUNTIME_FIELDS = frozenset( + { + "output_root", + "default_max_workers", + "maximum_workers", + "package_completed_runs", + "maximum_run_output_bytes", + "blocked_capabilities", + "temporary_directory", + } +) +_INTERACTION_FIELDS = frozenset({"history_enabled", "similarity_threshold", "model_name", "model_debug"}) +_MAINTENANCE_FIELDS = frozenset( + { + "remote_manifest_url", + "remote_manifest_sha256", + "local_manifest_path", + "minimum_python", + "recommended_python", + "sysinternals_enabled", + "sysinternals_download_url", + } +) +_LOGGING_FIELDS = frozenset( + { + "level", + "console_enabled", + "color_enabled", + "file_enabled", + "maximum_bytes", + "delete_previous", + "retention_days", + } +) +_LOG_LEVELS = frozenset({"DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL", "INTERNAL", "EXCEPTION"}) +DEFAULT_CONFIGURATION_FILENAME = "logicytics.yaml" +DEFAULT_SYSINTERNALS_DOWNLOAD_URL = "https://download.sysinternals.com/files/SysinternalsSuite.zip" + +CollectorSettingValue = str | int | float +CollectorSettings = Mapping[str, CollectorSettingValue] + + +@dataclass(frozen=True, slots=True) +class CollectorSettingRule: + """One immutable collector configuration field and its safe value boundary.""" + + kind: str + minimum: int | float = 0 + maximum: int | float = 0 + + +_COLLECTOR_SETTING_SCHEMAS: Mapping[ + str, + Mapping[str, CollectorSettingRule], +] = MappingProxyType( + { + "core.network.bandwidth_sample": { + "sample_count": CollectorSettingRule("integer", 1, 10), + "interval_seconds": CollectorSettingRule("number", 0.1, 60), + }, + "core.packet.packet_capture": { + "packet_count": CollectorSettingRule("integer", 1, 10_000), + "timeout_seconds": CollectorSettingRule("number", 1, 60), + "retry_window_seconds": CollectorSettingRule("number", 0, 60), + "interface": CollectorSettingRule("text"), + }, + "core.filesystem.system_drive_tree": { + "max_entries": CollectorSettingRule("integer", 1, 50_000), + "max_depth": CollectorSettingRule("integer", 1, 32), + }, + "core.filesystem.system_drive_listing": { + "max_entries": CollectorSettingRule("integer", 1, 50_000), + "max_depth": CollectorSettingRule("integer", 1, 32), + }, + "core.filesystem.sensitive_file_inventory": { + "root": CollectorSettingRule("absolute_path"), + "max_directories": CollectorSettingRule("integer", 1, 50_000), + "max_matches": CollectorSettingRule("integer", 1, 5_000), + }, + "core.process.memory_map": { + "max_regions": CollectorSettingRule("integer", 1, 100_000), + "output_limit_bytes": CollectorSettingRule( + "integer", + 1_024, + 64 * 1024 * 1024, + ), + "disk_safety_margin_bytes": CollectorSettingRule( + "integer", + 0, + MAXIMUM_RUN_OUTPUT_BYTES, + ), + "dump_directory": CollectorSettingRule("workspace_path"), + }, + } +) + + +def _positive_integer(value: object, *, minimum: int, maximum: int) -> bool: + """Whether a JSON value is a bounded integer rather than a boolean or coercion.""" + return isinstance(value, int) and not isinstance(value, bool) and minimum <= value <= maximum + + +def _bounded_number(value: object, *, minimum: float, maximum: float) -> bool: + """Whether a JSON value is a bounded finite numeric setting.""" + return isinstance(value, (int, float)) and not isinstance(value, bool) and minimum <= value <= maximum + + +def _configured_capabilities(value: object, setting_name: str) -> tuple[Capability, ...]: + """Parse capabilities disabled by configuration without accepting ambiguous values.""" + if value is None: + return () + if isinstance(value, dict): + if not all(isinstance(name, str) and isinstance(enabled, bool) for name, enabled in value.items()): + raise PlanError(f"{setting_name} must map capability names to booleans") + names = [name for name, enabled in value.items() if enabled] + elif isinstance(value, (list, tuple)): + names = list(value) + else: + raise PlanError(f"{setting_name} must be a capability list or a capability-to-boolean mapping") + try: + capabilities = tuple(Capability(name) for name in names) + except (TypeError, ValueError) as error: + raise PlanError(f"{setting_name} contains an unsupported capability: {error}") from error + if len(set(capabilities)) != len(capabilities): + raise PlanError(f"{setting_name} must not contain duplicate capabilities") + return capabilities + + +def _validate_collector_settings(settings: Mapping[str, Mapping[str, Any]]) -> None: + """Validate collector identities and all documented fields before worker launch.""" + for collector_id, values in settings.items(): + if not isinstance(collector_id, str) or not _COLLECTOR_ID.fullmatch(collector_id): + raise PlanError(f"collectors configuration contains an invalid collector ID: {collector_id!r}") + for key in values: + if not isinstance(key, str) or not _SETTING_NAME.fullmatch(key): + raise PlanError(f"{collector_id} contains an invalid setting name: {key!r}") + schema = _COLLECTOR_SETTING_SCHEMAS.get(collector_id) + if schema is None: + if collector_id.startswith("core.") and values: + raise PlanError(f"{collector_id} does not declare configurable settings") + continue + unknown = sorted(set(values) - set(schema)) + if unknown: + raise PlanError(f"{collector_id} contains unsupported settings: {', '.join(unknown)}") + for key, value in values.items(): + rule = schema[key] + label = f"{collector_id}.{key}" + if rule.kind == "integer": + if not _positive_integer(value, minimum=int(rule.minimum), maximum=int(rule.maximum)): + raise PlanError(f"{label} must be an integer from {rule.minimum} to {rule.maximum}") + elif rule.kind == "number": + if not _bounded_number(value, minimum=rule.minimum, maximum=rule.maximum): + raise PlanError(f"{label} must be a number from {rule.minimum} to {rule.maximum}") + elif not isinstance(value, str) or not value.strip() or "\x00" in value: + raise PlanError(f"{label} must be a non-empty path or string") + elif rule.kind == "absolute_path" and not Path(value).is_absolute(): + raise PlanError(f"{label} must be an absolute filesystem path") + elif rule.kind == "workspace_path": + path = Path(value) + if path.is_absolute() or path.drive or ".." in path.parts: + raise PlanError(f"{label} must remain a relative collector-workspace path") + + +def _unique_json_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + """Reject ambiguous duplicate configuration keys instead of silently replacing them.""" + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate configuration key {key!r}") + result[key] = value + return result + + +def _reject_json_constant(value: str) -> None: + """Reject non-finite nonstandard JSON values before settings can reach workers.""" + raise ValueError(f"non-finite configuration number {value!r}") + + +def _finite_json_number(value: str) -> float: + """Reject syntactically valid decimals that overflow the local float range.""" + result = float(value) + if not isfinite(result): + raise ValueError(f"non-finite configuration number {value!r}") + return result + + +def _yaml_scalar(value: str, *, line_number: int) -> object: + """Parse the safe scalar subset supported by the root configuration format.""" + if not value: + return {} + if value in {"null", "Null", "NULL", "~"}: + return None + if value in {"true", "True", "TRUE"}: + return True + if value in {"false", "False", "FALSE"}: + return False + if re.fullmatch(r"{\s*}", value): + return {} + if value.startswith(('"', "'")): + if not value.endswith(value[0]): + raise PlanError(f"invalid YAML string at line {line_number}") + if value[0] == '"': + try: + return json.loads(value) + except json.JSONDecodeError as error: + raise PlanError(f"invalid YAML string at line {line_number}: {error.msg}") from error + return value[1:-1].replace("''", "'") + if value.startswith(("[", "{", "&", "*", "|", ">", "!")): + raise PlanError(f"unsupported YAML value at line {line_number}; use a scalar or indented mapping") + if re.fullmatch(r"[-+]?\d+", value): + return int(value) + if re.fullmatch(r"[-+]?(?:\d+\.\d*|\d*\.\d+)(?:[eE][-+]?\d+)?", value): + parsed = float(value) + if not isfinite(parsed): + raise PlanError(f"non-finite YAML number at line {line_number}") + return parsed + return value + + +def _load_yaml_mapping(payload: bytes) -> dict[str, Any]: + """Load strict, dependency-free mapping YAML for deterministic user settings.""" + try: + text = payload.decode("utf-8-sig") + except UnicodeDecodeError as error: + raise PlanError(f"invalid YAML configuration encoding: {error}") from error + if text.lstrip().startswith("{"): + try: + parsed = json.loads( + text, + object_pairs_hook=_unique_json_object, + parse_constant=_reject_json_constant, + parse_float=_finite_json_number, + ) + except ValueError as error: + raise PlanError(f"invalid YAML configuration: {error}") from error + if not isinstance(parsed, dict): + raise PlanError("YAML configuration root must be a mapping") + return parsed + records: list[tuple[int, int, str, str]] = [] + for number, raw_line in enumerate(text.splitlines(), start=1): + if "\t" in raw_line: + raise PlanError(f"invalid YAML indentation at line {number}; use spaces") + line = raw_line.split("#", 1)[0].rstrip() + if not line.strip(): + continue + indent = len(line) - len(line.lstrip(" ")) + if indent % 2: + raise PlanError(f"invalid YAML indentation at line {number}; use multiples of two spaces") + content = line.strip() + if content.startswith("-") or ":" not in content: + raise PlanError(f"invalid YAML mapping entry at line {number}") + key, value = content.split(":", 1) + key = key.strip() + if not key or key.startswith(('"', "'")) or any(character.isspace() for character in key): + raise PlanError(f"invalid YAML key at line {number}") + records.append((indent, number, key, value.strip())) + if not records: + raise PlanError("YAML configuration must contain a root mapping") + + def parse_mapping(index: int, indentation: int) -> tuple[dict[str, Any], int]: + """Parse one indentation-delimited YAML mapping level.""" + result: dict[str, Any] = {} + while index < len(records): + indent, number, key, value = records[index] + if indent < indentation: + break + if indent != indentation: + raise PlanError(f"invalid YAML nesting at line {number}") + if key in result: + raise PlanError(f"duplicate YAML key {key!r} at line {number}") + index += 1 + if not value and index < len(records) and records[index][0] > indentation: + result[key], index = parse_mapping(index, records[index][0]) + else: + result[key] = _yaml_scalar(value, line_number=number) + return result, index + + root, consumed = parse_mapping(0, 0) + if consumed != len(records): + raise PlanError("invalid YAML configuration nesting") + return root + + +def default_configuration_yaml() -> str: + """Return the YAML template written by installer and repair flows.""" + return "\n".join( + ( + "# Logicytics user configuration", + "schema_version: 4", + "runtime:", + " output_root: output/data", + " default_max_workers: 4", + " maximum_workers: 16", + " package_completed_runs: true", + " blocked_capabilities: {}", + " temporary_directory: project", + "interaction:", + " history_enabled: false", + " similarity_threshold: 0.55", + " model_name: stdlib-sequence-matcher", + " model_debug: false", + "maintenance:", + " local_manifest_path: project.manifest.json", + ' minimum_python: "3.11"', + ' recommended_python: "3.11"', + " sysinternals_enabled: true", + " sysinternals_download_url: https://download.sysinternals.com/files/SysinternalsSuite.zip", + "logging:", + " level: INFO", + " console_enabled: true", + " color_enabled: true", + " file_enabled: true", + " maximum_bytes: 4194304", + " delete_previous: false", + " retention_days: 30", + "collectors: {}", + "", + ) + ) + + +def write_default_configuration(project_root: Path, *, overwrite: bool = False) -> Path: + """Atomically create the authoritative root YAML configuration template.""" + path = project_root / DEFAULT_CONFIGURATION_FILENAME + if path.exists() and not overwrite: + return path + temporary = path.with_suffix(path.suffix + ".tmp") + temporary.write_text(default_configuration_yaml(), encoding="utf-8") + temporary.replace(path) + return path + + +@dataclass(frozen=True, slots=True) +class RuntimeSettings: + """Engine-wide limits that apply before a collector is started.""" + + output_root: Path + default_max_workers: int = 4 + maximum_workers: int = 16 + package_completed_runs: bool = True + maximum_run_output_bytes: int = DEFAULT_MAXIMUM_RUN_OUTPUT_BYTES + blocked_capabilities: tuple[Capability, ...] = () + temporary_directory: str = "project" + + +@dataclass(frozen=True, slots=True) +class InteractionSettings: + """Local-only matching, diagnostics, and optional history policy.""" + + history_enabled: bool = False + similarity_threshold: float = 0.55 + model_name: str = "stdlib-sequence-matcher" + model_debug: bool = False + + +@dataclass(frozen=True, slots=True) +class MaintenanceSettings: + """Optional integrity sources and supported Python policy.""" + + remote_manifest_url: str | None = None + remote_manifest_sha256: str | None = None + local_manifest_path: Path = Path("project.manifest.json") + minimum_python: str = "3.11" + recommended_python: str = "3.11" + sysinternals_enabled: bool = True + sysinternals_download_url: str = DEFAULT_SYSINTERNALS_DOWNLOAD_URL + + +@dataclass(frozen=True, slots=True) +class LoggingSettings: + """Process-wide human log and console presentation policy.""" + + level: str = "INFO" + console_enabled: bool = True + color_enabled: bool = True + file_enabled: bool = True + maximum_bytes: int = 4 * 1024 * 1024 + delete_previous: bool = False + retention_days: int = 30 + + +@dataclass(frozen=True, slots=True) +class AppConfig: + """Validated settings loaded exclusively from the root YAML configuration file.""" + + schema_version: int + runtime: RuntimeSettings + interaction: InteractionSettings = field(default_factory=InteractionSettings) + maintenance: MaintenanceSettings = field(default_factory=MaintenanceSettings) + logging: LoggingSettings = field(default_factory=LoggingSettings) + collector_settings: Mapping[str, CollectorSettings] = field(default_factory=dict) + migrated_from_schema: int | None = None + + def settings_for(self, collector_id: str) -> CollectorSettings: + """Return the isolated settings declared for one collector.""" + return self.collector_settings.get(collector_id, {}) + + def to_manifest_dict(self) -> dict[str, Any]: + """Return a JSON-safe, non-secret configuration snapshot.""" + data = asdict(self) + data["runtime"]["output_root"] = str(self.runtime.output_root) + data["maintenance"]["local_manifest_path"] = str(self.maintenance.local_manifest_path) + return redact_mapping(data) + + def fingerprint(self) -> str: + """Hash the complete validated configuration without persisting its secrets.""" + data = asdict(self) + data["runtime"]["output_root"] = str(self.runtime.output_root.resolve()) + data["maintenance"]["local_manifest_path"] = str(self.maintenance.local_manifest_path) + serialized = json.dumps(data, sort_keys=True, separators=(",", ":"), ensure_ascii=True) + return hashlib.sha256(serialized.encode("utf-8")).hexdigest() + + +def default_config(project_root: Path) -> AppConfig: + """Create safe defaults rooted at the checked-out project.""" + return AppConfig( + schema_version=SCHEMA_VERSION, + runtime=RuntimeSettings(output_root=project_root / "output" / "data"), + ) + + +def load_config(project_root: Path, config_path: Path | None = None) -> AppConfig: + """Load and validate the single authoritative root YAML configuration file.""" + path = config_path or project_root / DEFAULT_CONFIGURATION_FILENAME + if not path.is_absolute(): + path = project_root / path + if not path.exists(): + if config_path is not None: + raise PlanError(f"configuration file does not exist: {path}") + return default_config(project_root) + if path.suffix.casefold() not in {".yaml", ".yml"}: + raise PlanError("configuration must be a YAML file") + try: + payload = path.read_bytes() + except OSError as error: + raise PlanError(f"invalid YAML configuration file {path}: {error}") from error + if len(payload) > MAXIMUM_CONFIGURATION_BYTES: + raise PlanError("YAML configuration exceeds the 2 MiB limit") + raw = _load_yaml_mapping(payload) + if not isinstance(raw, dict): + raise PlanError("YAML configuration root must be a mapping") + schema_version = raw.get("schema_version", SCHEMA_VERSION) + if not isinstance(schema_version, int) or isinstance(schema_version, bool): + raise PlanError("configuration schema_version must be an integer") + if schema_version != SCHEMA_VERSION: + raise PlanError(f"unsupported configuration schema_version {schema_version}; expected {SCHEMA_VERSION}") + unknown_root = sorted(set(raw) - _ROOT_FIELDS) + if unknown_root: + raise PlanError(f"configuration contains unsupported root settings: {', '.join(unknown_root)}") + + runtime_raw = raw.get("runtime", {}) + if not isinstance(runtime_raw, dict): + raise PlanError("runtime configuration must be an object") + unknown_runtime = sorted(set(runtime_raw) - _RUNTIME_FIELDS) + if unknown_runtime: + raise PlanError(f"runtime configuration contains unsupported settings: {', '.join(unknown_runtime)}") + output_root_value = runtime_raw.get("output_root", project_root / "output" / "data") + if not isinstance(output_root_value, (str, Path)) or not str(output_root_value).strip(): + raise PlanError("runtime output_root must be a non-empty path string") + output_root = Path(output_root_value) + if not output_root.is_absolute(): + output_root = project_root / output_root + default_workers = runtime_raw.get("default_max_workers", 4) + maximum_workers = runtime_raw.get("maximum_workers", 16) + if not _positive_integer(default_workers, minimum=1, maximum=64) or not _positive_integer( + maximum_workers, + minimum=1, + maximum=64, + ): + raise PlanError("worker limits must be integers") + if not 1 <= default_workers <= maximum_workers <= 64: + raise PlanError("worker limits must satisfy 1 <= default <= maximum <= 64") + package_completed_runs = runtime_raw.get("package_completed_runs", True) + if not isinstance(package_completed_runs, bool): + raise PlanError("runtime package_completed_runs must be boolean") + maximum_run_output_bytes = runtime_raw.get( + "maximum_run_output_bytes", + DEFAULT_MAXIMUM_RUN_OUTPUT_BYTES, + ) + if not _positive_integer( + maximum_run_output_bytes, + minimum=1, + maximum=MAXIMUM_RUN_OUTPUT_BYTES, + ): + raise PlanError("runtime maximum_run_output_bytes must be an integer from 1 to 68719476736") + blocked_capabilities = _configured_capabilities( + runtime_raw.get("blocked_capabilities", {}), + "runtime blocked_capabilities", + ) + temporary_directory = runtime_raw.get("temporary_directory", "project") + if not isinstance(temporary_directory, str) or temporary_directory not in {"project", "system"}: + raise PlanError("runtime temporary_directory must be project or system") + + interaction_raw = raw.get("interaction", {}) + if not isinstance(interaction_raw, dict): + raise PlanError("interaction configuration must be an object") + unknown_interaction = sorted(set(interaction_raw) - _INTERACTION_FIELDS) + if unknown_interaction: + unsupported_settings = ", ".join(unknown_interaction) + raise PlanError(f"interaction configuration contains unsupported settings: {unsupported_settings}") + history_enabled = interaction_raw.get("history_enabled", False) + model_debug = interaction_raw.get("model_debug", False) + similarity_threshold = interaction_raw.get("similarity_threshold", 0.55) + model_name = interaction_raw.get("model_name", "stdlib-sequence-matcher") + if not isinstance(history_enabled, bool) or not isinstance(model_debug, bool): + raise PlanError("interaction history_enabled and model_debug must be boolean") + if not _bounded_number(similarity_threshold, minimum=0, maximum=1): + raise PlanError("interaction similarity_threshold must be a number from 0 to 1") + if not isinstance(model_name, str) or not model_name.strip() or any(char in model_name for char in "\r\n"): + raise PlanError("interaction model_name must be a non-empty single-line string") + + maintenance_raw = raw.get("maintenance", {}) + if not isinstance(maintenance_raw, dict): + raise PlanError("maintenance configuration must be an object") + unknown_maintenance = sorted(set(maintenance_raw) - _MAINTENANCE_FIELDS) + if unknown_maintenance: + raise PlanError(f"maintenance configuration contains unsupported settings: {', '.join(unknown_maintenance)}") + remote_url = maintenance_raw.get("remote_manifest_url") + remote_sha256 = maintenance_raw.get("remote_manifest_sha256") + if (remote_url is None) != (remote_sha256 is None): + raise PlanError("remote manifest URL and SHA-256 must be configured together") + if remote_url is not None and ( + not isinstance(remote_url, str) or not remote_url.startswith("https://") or "\n" in remote_url): + raise PlanError("remote_manifest_url must be an HTTPS URL") + if remote_sha256 is not None and ( + not isinstance(remote_sha256, str) or re.fullmatch(r"[0-9a-f]{64}", remote_sha256) is None): + raise PlanError("remote_manifest_sha256 must be a lowercase SHA-256 digest") + local_manifest_value = maintenance_raw.get("local_manifest_path", "project.manifest.json") + if not isinstance(local_manifest_value, str) or not local_manifest_value.strip(): + raise PlanError("local_manifest_path must be a non-empty relative path") + local_manifest_path = Path(local_manifest_value) + if local_manifest_path.is_absolute() or local_manifest_path.drive or ".." in local_manifest_path.parts: + raise PlanError("local_manifest_path must remain inside the project") + version_pattern = re.compile(r"^\d+\.\d+$") + minimum_python = maintenance_raw.get("minimum_python", "3.11") + recommended_python = maintenance_raw.get("recommended_python", "3.11") + if not isinstance(minimum_python, str) or version_pattern.fullmatch(minimum_python) is None: + raise PlanError("minimum_python must use major.minor form") + if not isinstance(recommended_python, str) or version_pattern.fullmatch(recommended_python) is None: + raise PlanError("recommended_python must use major.minor form") + if tuple(map(int, recommended_python.split("."))) < tuple(map(int, minimum_python.split("."))): + raise PlanError("recommended_python must not be older than minimum_python") + sysinternals_enabled = maintenance_raw.get("sysinternals_enabled", True) + sysinternals_download_url = maintenance_raw.get( + "sysinternals_download_url", + DEFAULT_SYSINTERNALS_DOWNLOAD_URL, + ) + if not isinstance(sysinternals_enabled, bool): + raise PlanError("maintenance sysinternals_enabled must be boolean") + if not isinstance(sysinternals_download_url, str) or not sysinternals_download_url.startswith("https://"): + raise PlanError("maintenance sysinternals_download_url must be an HTTPS URL") + + logging_raw = raw.get("logging", {}) + if not isinstance(logging_raw, dict): + raise PlanError("logging configuration must be an object") + unknown_logging = sorted(set(logging_raw) - _LOGGING_FIELDS) + if unknown_logging: + raise PlanError(f"logging configuration contains unsupported settings: {', '.join(unknown_logging)}") + logging_level = logging_raw.get("level", "INFO") + if not isinstance(logging_level, str) or logging_level.upper() not in _LOG_LEVELS: + raise PlanError("logging level must be DEBUG, INFO, WARNING, ERROR, CRITICAL, INTERNAL, or EXCEPTION") + console_enabled = logging_raw.get("console_enabled", True) + color_enabled = logging_raw.get("color_enabled", True) + file_enabled = logging_raw.get("file_enabled", True) + delete_previous = logging_raw.get("delete_previous", False) + if not all(isinstance(value, bool) for value in (console_enabled, color_enabled, file_enabled, delete_previous)): + raise PlanError("logging enable, color, file, and deletion settings must be boolean") + log_maximum_bytes = logging_raw.get("maximum_bytes", 4 * 1024 * 1024) + if not isinstance(log_maximum_bytes, int) or isinstance(log_maximum_bytes, + bool) or not 1024 <= log_maximum_bytes <= 64 * 1024 * 1024: + raise PlanError("logging maximum_bytes must be an integer from 1024 to 67108864") + retention_days = logging_raw.get("retention_days", 30) + if not isinstance(retention_days, int) or isinstance(retention_days, bool) or not 0 <= retention_days <= 3650: + raise PlanError("logging retention_days must be an integer from 0 to 3650") + + collector_settings = raw.get("collectors", {}) + if not isinstance(collector_settings, dict) or not all( + isinstance(key, str) and isinstance(value, dict) for key, value in collector_settings.items() + ): + raise PlanError("collectors configuration must map collector IDs to objects") + _validate_collector_settings(collector_settings) + return AppConfig( + schema_version=schema_version, + runtime=RuntimeSettings( + output_root=output_root, + default_max_workers=default_workers, + maximum_workers=maximum_workers, + package_completed_runs=package_completed_runs, + maximum_run_output_bytes=maximum_run_output_bytes, + blocked_capabilities=blocked_capabilities, + temporary_directory=temporary_directory, + ), + interaction=InteractionSettings( + history_enabled=history_enabled, + similarity_threshold=float(similarity_threshold), + model_name=model_name, + model_debug=model_debug, + ), + maintenance=MaintenanceSettings( + remote_manifest_url=remote_url, + remote_manifest_sha256=remote_sha256, + local_manifest_path=local_manifest_path, + minimum_python=minimum_python, + recommended_python=recommended_python, + sysinternals_enabled=sysinternals_enabled, + sysinternals_download_url=sysinternals_download_url, + ), + logging=LoggingSettings( + level=logging_level.upper(), + console_enabled=console_enabled, + color_enabled=color_enabled, + file_enabled=file_enabled, + maximum_bytes=log_maximum_bytes, + delete_previous=delete_previous, + retention_days=retention_days, + ), + collector_settings=collector_settings, + migrated_from_schema=None, + ) diff --git a/logicytics/module/contracts.py b/logicytics/module/contracts.py new file mode 100644 index 00000000..15b51563 --- /dev/null +++ b/logicytics/module/contracts.py @@ -0,0 +1,695 @@ +"""Stable v4 contracts shared by the engine, core collectors, and plugins.""" + +from __future__ import annotations + +import re +from abc import ABC, abstractmethod +from collections.abc import Mapping +from dataclasses import asdict, dataclass, field +from datetime import datetime +from enum import StrEnum +from math import isfinite +from pathlib import Path +from typing import Any + +CONTRACT_VERSION = "4.0" +_CUSTOM_SPECIALTY = re.compile(r"^[a-z][a-z0-9_]{1,63}$") +_COLLECTOR_ID = re.compile(r"^(?:core|plugin)\.[a-z][a-z0-9_]*(?:\.[a-z][a-z0-9_]*)?$") +_SEMANTIC_VERSION = re.compile(r"^\d+\.\d+\.\d+(?:[-+][0-9A-Za-z.-]+)?$") +_CONTRACT_VERSION = re.compile(r"^\d+\.\d+$") +_LABEL = re.compile(r"^[a-z][a-z0-9_]{0,63}$") +_RUN_ID = re.compile(r"^run-[0-9a-f]{32}$") +_ARTIFACT_ID = re.compile(r"^artifact\.[0-9a-f]{32}$") +_ARTIFACT_SHA256 = re.compile(r"^[0-9a-f]{64}$") +_MEDIA_TYPE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9!#$&^_.+-]*/[A-Za-z0-9][A-Za-z0-9!#$&^_.+-]*$") + + +class CollectorKind(StrEnum): + """The source tree that owns a collector.""" + + CORE = "core" + PLUGIN = "plugin" + + +class CollectorStatus(StrEnum): + """Terminal states reported for an individual collector.""" + + SUCCEEDED = "succeeded" + PARTIAL = "partial" + SKIPPED = "skipped" + CANCELLED = "cancelled" + FAILED = "failed" + + +class RunStatus(StrEnum): + """Terminal and in-progress states for a complete run.""" + + PLANNED = "planned" + RUNNING = "running" + SUCCEEDED = "succeeded" + PARTIAL = "partial" + FAILED = "failed" + CANCELLED = "cancelled" + + +class OutputPolicy(StrEnum): + """How a completed run publishes its evidence.""" + + PACKAGE = "package" + MANIFEST_ONLY = "manifest_only" + + +class PostRunAction(StrEnum): + """An explicitly requested host action after durable run publication.""" + + NONE = "none" + REBOOT = "reboot" + SHUTDOWN = "shutdown" + + +class EvidenceKind(StrEnum): + """The package section that owns a registered evidence artifact.""" + + RAW = "raw" + DERIVED = "derived" + + +class Specialty(StrEnum): + """The closed set of supported collector specialties.""" + + SYSTEM = "system" + HARDWARE = "hardware" + PROCESS = "process" + MEMORY = "memory" + FILESYSTEM = "filesystem" + NETWORK = "network" + PACKET = "packet" + WIRELESS = "wireless" + BLUETOOTH = "bluetooth" + USB = "usb" + BROWSER = "browser" + REGISTRY = "registry" + EVENT_LOG = "event_log" + STORAGE = "storage" + ENCRYPTION = "encryption" + MEDIA = "media" + SSH = "ssh" + DIAGNOSTICS = "diagnostics" + REPORTING = "reporting" + INTEGRATION = "integration" + + +class Capability(StrEnum): + """Platform access a collector must declare before requesting it.""" + + FILESYSTEM_READ = "filesystem_read" + FILESYSTEM_WRITE = "filesystem_write" + REGISTRY_READ = "registry_read" + SUBPROCESS = "subprocess" + NETWORK = "network" + PACKET_CAPTURE = "packet_capture" + BROWSER_DATA = "browser_data" + SENSITIVE_FILES = "sensitive_files" + PRIVATE_KEYS = "private_keys" + ELEVATED_PRIVILEGES = "elevated_privileges" + + +class ResourceClass(StrEnum): + """Scheduling resources whose conflicting collectors cannot safely overlap.""" + + GENERAL = "general" + DISK_HEAVY = "disk_heavy" + NETWORK_HEAVY = "network_heavy" + REGISTRY_SENSITIVE = "registry_sensitive" + INTERACTIVE = "interactive" + + +class PrivilegeLevel(StrEnum): + """Host privilege required before a collector may launch.""" + + STANDARD = "standard" + ELEVATED = "elevated" + + +class NetworkAccess(StrEnum): + """Declared network reach of a collector's implementation.""" + + NONE = "none" + LOCAL = "local" + REMOTE = "remote" + + +class EstimatedCost(StrEnum): + """Coarse scheduling and consent cost declared before execution.""" + + LOW = "low" + MEDIUM = "medium" + HIGH = "high" + + +@dataclass(frozen=True, slots=True) +class CollectorMetadata: + """Declarative identity, limits, and permissions for one collector.""" + + id: str + name: str + version: str + specialty: Specialty | str + description: str + author: str + privilege_level: PrivilegeLevel = PrivilegeLevel.STANDARD + network_access: NetworkAccess = NetworkAccess.NONE + estimated_cost: EstimatedCost = EstimatedCost.LOW + secondary_categories: tuple[str, ...] = () + output_media_types: tuple[str, ...] = ("application/octet-stream",) + supported_platforms: tuple[str, ...] = ("win32",) + capabilities: tuple[Capability, ...] = () + sensitive_data_categories: tuple[str, ...] = () + dependencies: tuple[str, ...] = () + default_profiles: tuple[str, ...] = ("standard",) + timeout_seconds: int = 60 + maximum_memory_bytes: int = 512 * 1024 * 1024 + maximum_output_bytes: int = 100 * 1024 * 1024 + maximum_artifact_bytes: int | None = None + maximum_artifact_files: int = 500 + maximum_retries: int = 0 + retry_delay_seconds: float = 0.0 + minimum_contract_version: str = CONTRACT_VERSION + parallel_safe: bool = True + resource_class: ResourceClass = ResourceClass.GENERAL + + def __post_init__(self) -> None: + """Reject malformed metadata before it can enter planning or runtime.""" + for name in ("id", "name", "version", "description", "author", "minimum_contract_version"): + value = getattr(self, name) + if not isinstance(value, str) or not value.strip() or "\n" in value or "\r" in value: + raise ValueError(f"metadata {name} must be a non-empty single-line string") + if not _COLLECTOR_ID.fullmatch(self.id): + raise ValueError("metadata id has an invalid schema") + if not _SEMANTIC_VERSION.fullmatch(self.version): + raise ValueError("metadata version must use semantic versioning") + if not _CONTRACT_VERSION.fullmatch(self.minimum_contract_version): + raise ValueError("metadata minimum_contract_version has an invalid schema") + if not isinstance(self.specialty, Specialty) and ( + not isinstance(self.specialty, str) or not _CUSTOM_SPECIALTY.fullmatch(self.specialty) + ): + raise ValueError("metadata specialty has an invalid schema") + self._validate_labels("supported_platforms", self.supported_platforms, require_value=True) + self._validate_labels("sensitive_data_categories", self.sensitive_data_categories) + self._validate_labels("secondary_categories", self.secondary_categories) + self._validate_labels("default_profiles", self.default_profiles, require_value=True) + primary_specialty = self.specialty.value if isinstance(self.specialty, Specialty) else self.specialty + if primary_specialty in self.secondary_categories: + raise ValueError("metadata secondary_categories must not repeat the primary specialty") + if ( + not isinstance(self.output_media_types, tuple) + or not self.output_media_types + or not all( + isinstance(media_type, str) and _MEDIA_TYPE.fullmatch(media_type) for media_type in self.output_media_types) + ): + raise ValueError("metadata output_media_types must be a non-empty tuple of MIME types") + if len(set(self.output_media_types)) != len(self.output_media_types): + raise ValueError("metadata output_media_types must not contain duplicates") + standard_categories = { + "encryption_configuration", + "hardware_inventory", + "security_configuration", + "system_configuration", + } + if "minimal" in self.default_profiles and self.sensitive_data_categories: + raise ValueError("sensitive collectors must not belong to the minimal profile") + if "standard" in self.default_profiles and not set(self.sensitive_data_categories).issubset( + standard_categories): + raise ValueError("standard collectors may contain only routine local configuration categories") + if not isinstance(self.capabilities, tuple) or not all( + isinstance(capability, Capability) for capability in self.capabilities): + raise ValueError("metadata capabilities must be a tuple of Capability values") + if not isinstance(self.privilege_level, PrivilegeLevel): + raise ValueError("metadata privilege_level must be a PrivilegeLevel value") + if not isinstance(self.network_access, NetworkAccess): + raise ValueError("metadata network_access must be a NetworkAccess value") + if not isinstance(self.estimated_cost, EstimatedCost): + raise ValueError("metadata estimated_cost must be an EstimatedCost value") + requires_elevation = Capability.ELEVATED_PRIVILEGES in self.capabilities + if requires_elevation != (self.privilege_level is PrivilegeLevel.ELEVATED): + raise ValueError("metadata privilege_level must match the elevated_privileges capability") + if Capability.NETWORK in self.capabilities and self.network_access is NetworkAccess.NONE: + raise ValueError("metadata network_access must declare local or remote access") + if not isinstance(self.dependencies, tuple) or not all( + isinstance(dependency, str) and _COLLECTOR_ID.fullmatch(dependency) for dependency in self.dependencies + ): + raise ValueError("metadata dependencies must be collector IDs") + if len(set(self.dependencies)) != len(self.dependencies) or self.id in self.dependencies: + raise ValueError("metadata dependencies must be unique and cannot include the collector itself") + if self.maximum_artifact_bytes is None: + object.__setattr__(self, "maximum_artifact_bytes", self.maximum_output_bytes) + for name in ( + "timeout_seconds", + "maximum_memory_bytes", + "maximum_output_bytes", + "maximum_artifact_bytes", + "maximum_artifact_files", + ): + value = getattr(self, name) + if not isinstance(value, int) or isinstance(value, bool) or value < 1: + raise ValueError(f"metadata {name} must be a positive integer") + if self.maximum_artifact_bytes > self.maximum_output_bytes: + raise ValueError("metadata maximum_artifact_bytes must not exceed maximum_output_bytes") + if not isinstance(self.maximum_retries, int) or isinstance(self.maximum_retries, bool) or not ( + 0 <= self.maximum_retries <= 3): + raise ValueError("metadata maximum_retries must be an integer from 0 to 3") + if ( + not isinstance(self.retry_delay_seconds, (int, float)) + or isinstance( + self.retry_delay_seconds, + bool, + ) + or not isfinite(self.retry_delay_seconds) + or not 0 <= self.retry_delay_seconds <= 30 + ): + raise ValueError("metadata retry_delay_seconds must be a number from 0 to 30") + if not isinstance(self.parallel_safe, bool): + raise ValueError("metadata parallel_safe must be boolean") + if not isinstance(self.resource_class, ResourceClass): + raise ValueError("metadata resource_class must be a ResourceClass value") + + @staticmethod + def _validate_labels(name: str, values: tuple[str, ...], *, require_value: bool = False) -> None: + """Require unique lower-snake-case labels for selector-like metadata fields.""" + if ( + not isinstance(values, tuple) + or (require_value and not values) + or not all(isinstance(value, str) and _LABEL.fullmatch(value) for value in values) + ): + raise ValueError(f"metadata {name} must be a tuple of lowercase labels") + if len(set(values)) != len(values): + raise ValueError(f"metadata {name} must not contain duplicates") + + def to_dict(self) -> dict[str, Any]: + """Return JSON-safe metadata.""" + data = asdict(self) + data["specialty"] = self.specialty.value if isinstance(self.specialty, Specialty) else self.specialty + data["capabilities"] = [capability.value for capability in self.capabilities] + data["resource_class"] = self.resource_class.value + data["privilege_level"] = self.privilege_level.value + data["network_access"] = self.network_access.value + data["estimated_cost"] = self.estimated_cost.value + return data + + @classmethod + def from_dict(cls, data: Mapping[str, Any], *, allow_custom_specialty: bool = False) -> CollectorMetadata: + """Build metadata returned by an isolated validation worker.""" + values = dict(data) + specialty = values["specialty"] + try: + values["specialty"] = Specialty(specialty) + except ValueError: + if not allow_custom_specialty or not isinstance(specialty, str) or not _CUSTOM_SPECIALTY.fullmatch( + specialty): + raise ValueError("collector specialty is unsupported") + values["specialty"] = specialty + values["capabilities"] = tuple(Capability(capability) for capability in values.get("capabilities", ())) + values["resource_class"] = ResourceClass(values.get("resource_class", ResourceClass.GENERAL)) + values["privilege_level"] = PrivilegeLevel(values.get("privilege_level", PrivilegeLevel.STANDARD)) + values["network_access"] = NetworkAccess(values.get("network_access", NetworkAccess.NONE)) + values["estimated_cost"] = EstimatedCost(values.get("estimated_cost", EstimatedCost.LOW)) + for field_name in ( + "supported_platforms", + "sensitive_data_categories", + "secondary_categories", + "output_media_types", + "dependencies", + "default_profiles", + ): + values[field_name] = tuple(values.get(field_name, ())) + return cls(**values) + + +@dataclass(frozen=True, slots=True) +class Artifact: + """An evidence artifact registered by a collector.""" + + id: str + relative_path: str + sha256: str + size_bytes: int + media_type: str + collector_id: str + source_category: str + collected_at: str + transformations: tuple[str, ...] + evidence_kind: EvidenceKind + name: str + status: str = "registered" + + def __post_init__(self) -> None: + """Reject malformed evidence records before worker or package publication.""" + if not isinstance(self.id, str) or not _ARTIFACT_ID.fullmatch(self.id): + raise ValueError("artifact id must be a stable artifact identifier") + if not isinstance(self.relative_path, str) or not self.relative_path.strip(): + raise ValueError("artifact relative_path must be a non-empty string") + if not isinstance(self.sha256, str) or not _ARTIFACT_SHA256.fullmatch(self.sha256): + raise ValueError("artifact sha256 must be a lowercase SHA-256 digest") + if not isinstance(self.size_bytes, int) or isinstance(self.size_bytes, bool) or self.size_bytes < 0: + raise ValueError("artifact size_bytes must be a non-negative integer") + if not isinstance(self.media_type, str) or not _MEDIA_TYPE.fullmatch(self.media_type): + raise ValueError("artifact media_type must be a valid MIME type") + if not isinstance(self.collector_id, str) or not _COLLECTOR_ID.fullmatch(self.collector_id): + raise ValueError("artifact collector_id must identify its producing collector") + if not isinstance(self.source_category, str) or not _LABEL.fullmatch(self.source_category): + raise ValueError("artifact source_category must be a lowercase category label") + if not isinstance(self.name, str) or not self.name.strip() or any(value in self.name for value in "\r\n/\\"): + raise ValueError("artifact name must be a safe human-readable filename") + if self.status != "registered": + raise ValueError("artifact status must be registered") + try: + collected_at = datetime.fromisoformat(self.collected_at) + except (TypeError, ValueError) as error: + raise ValueError("artifact collected_at must be an ISO-8601 timestamp") from error + if collected_at.tzinfo is None: + raise ValueError("artifact collected_at must include a timezone") + if not isinstance(self.transformations, tuple) or any( + not isinstance(step, str) or not step.strip() for step in self.transformations + ): + raise ValueError("artifact transformations must be a tuple of non-empty strings") + if not isinstance(self.evidence_kind, EvidenceKind): + raise ValueError("artifact evidence_kind must be an EvidenceKind value") + + @classmethod + def from_dict(cls, data: Mapping[str, Any]) -> Artifact: + """Reconstruct a strict artifact after a JSON or worker boundary.""" + values = dict(data) + transformations = values.get("transformations") + if not isinstance(transformations, (list, tuple)): + raise ValueError("artifact transformations must be an array or tuple") + values["transformations"] = tuple(transformations) + try: + values["evidence_kind"] = EvidenceKind(values["evidence_kind"]) + except (KeyError, TypeError, ValueError) as error: + raise ValueError("artifact evidence_kind must be raw or derived") from error + return cls(**values) + + def to_dict(self) -> dict[str, Any]: + """Return the normalized artifact as a JSON-safe mapping.""" + return asdict(self) + + +@dataclass(frozen=True, slots=True) +class ValidationResult: + """The outcome of a collector's side-effect-free validation phase.""" + + valid: bool + reasons: tuple[str, ...] = () + warnings: tuple[str, ...] = () + + +@dataclass(frozen=True, slots=True) +class CollectionEstimate: + """A collector's optional, side-effect-free resource estimate.""" + + estimated_seconds: float + estimated_output_bytes: int + notes: tuple[str, ...] = () + + def __post_init__(self) -> None: + """Reject estimates that cannot describe a possible collection run.""" + if self.estimated_seconds < 0: + raise ValueError("estimated_seconds must not be negative") + if self.estimated_output_bytes < 0: + raise ValueError("estimated_output_bytes must not be negative") + + +@dataclass(frozen=True, slots=True) +class CollectorResult: + """The normalized result returned by an isolated collector worker.""" + + status: CollectorStatus + summary: str + artifacts: tuple[Artifact, ...] = () + errors: tuple[str, ...] = () + metrics: Mapping[str, int | float | str] = field(default_factory=dict) + + def __post_init__(self) -> None: + """Reject malformed terminal results before they can cross a worker boundary.""" + if not isinstance(self.status, CollectorStatus): + raise ValueError("collector result status must be a CollectorStatus value") + if not isinstance(self.summary, str) or not self.summary.strip(): + raise ValueError("collector result summary must be a non-empty string") + if not isinstance(self.artifacts, tuple) or not all(isinstance(item, Artifact) for item in self.artifacts): + raise ValueError("collector result artifacts must be a tuple of Artifact values") + artifact_ids = tuple(item.id for item in self.artifacts) + artifact_paths = tuple(item.relative_path for item in self.artifacts) + if len(set(artifact_ids)) != len(artifact_ids) or len(set(artifact_paths)) != len(artifact_paths): + raise ValueError("collector result artifacts must have unique IDs and paths") + if not isinstance(self.errors, tuple) or not all( + isinstance(error, str) and error.strip() for error in self.errors): + raise ValueError("collector result errors must be a tuple of non-empty strings") + if not isinstance(self.metrics, Mapping) or not all( + isinstance(name, str) + and name.strip() + and isinstance(value, (int, float, str)) + and not isinstance(value, bool) + and (not isinstance(value, float) or isfinite(value)) + for name, value in self.metrics.items() + ): + raise ValueError("collector result metrics must contain finite scalar values") + + @classmethod + def succeeded(cls, summary: str, artifacts: tuple[Artifact, ...] = ()) -> CollectorResult: + """Return a successful result containing any registered evidence.""" + return cls(CollectorStatus.SUCCEEDED, summary, artifacts) + + @classmethod + def partial( + cls, + summary: str, + artifacts: tuple[Artifact, ...] = (), + *, + errors: tuple[str, ...] = (), + ) -> CollectorResult: + """Return an explicitly incomplete result while preserving registered evidence.""" + return cls(CollectorStatus.PARTIAL, summary, artifacts, errors=errors) + + @classmethod + def skipped(cls, summary: str, *, errors: tuple[str, ...] = ()) -> CollectorResult: + """Return an explicit prerequisite or policy skip.""" + return cls(CollectorStatus.SKIPPED, summary, errors=errors) + + @classmethod + def cancelled( + cls, + summary: str, + artifacts: tuple[Artifact, ...] = (), + ) -> CollectorResult: + """Return explicit cancellation while retaining already registered evidence.""" + return cls(CollectorStatus.CANCELLED, summary, artifacts) + + @classmethod + def failed( + cls, + summary: str, + *, + errors: tuple[str, ...] = (), + artifacts: tuple[Artifact, ...] = (), + ) -> CollectorResult: + """Return explicit failure while retaining already registered evidence.""" + return cls(CollectorStatus.FAILED, summary, artifacts, errors=errors) + + +@dataclass(frozen=True, slots=True) +class RunRequest: + """An immutable user request resolved before any collector starts.""" + + profile: str = "standard" + include: tuple[str, ...] = () + exclude: tuple[str, ...] = () + selection_only: bool = False + enable_plugins: bool = False + max_workers: int = 4 + acknowledge_authorization: bool = False + blocked_capabilities: tuple[Capability, ...] = () + """Capabilities globally disabled for this request, unless absent from metadata.""" + # Retained for manifest/API compatibility; declared metadata is now allowed by default. + approved_capabilities: tuple[Capability, ...] = () + performance_check: bool = False + rerun_from: str | None = None + output_policy: OutputPolicy = OutputPolicy.PACKAGE + post_run_action: PostRunAction = PostRunAction.NONE + + def __post_init__(self) -> None: + """Reject malformed selections and execution policy before planning.""" + if not isinstance(self.profile, str) or not _LABEL.fullmatch(self.profile): + raise ValueError("request profile must be a lowercase profile label") + for name in ("include", "exclude"): + selections = getattr(self, name) + if not isinstance(selections, tuple) or not all( + isinstance(collector_id, str) and _COLLECTOR_ID.fullmatch(collector_id) for collector_id in + selections + ): + raise ValueError(f"request {name} must be a tuple of collector IDs") + if len(set(selections)) != len(selections): + raise ValueError(f"request {name} must not contain duplicate collector IDs") + if set(self.include).intersection(self.exclude): + raise ValueError("request include and exclude selections must not overlap") + if self.rerun_from is not None and ( + not isinstance(self.rerun_from, str) or not _RUN_ID.fullmatch(self.rerun_from)): + raise ValueError("request rerun_from must be a valid original run ID") + if self.rerun_from is not None and not self.include: + raise ValueError("request rerun_from requires explicit included collector IDs") + for name in ( + "selection_only", + "enable_plugins", + "acknowledge_authorization", + "performance_check", + ): + if not isinstance(getattr(self, name), bool): + raise ValueError(f"request {name} must be boolean") + if self.selection_only and not self.include: + raise ValueError("request selection_only requires explicit included collector IDs") + if not isinstance(self.max_workers, int) or isinstance(self.max_workers, + bool) or not 1 <= self.max_workers <= 64: + raise ValueError("request max_workers must be an integer from 1 to 64") + if self.performance_check and self.max_workers != 1: + raise ValueError("request performance_check requires max_workers=1") + if not isinstance(self.blocked_capabilities, tuple) or not all( + isinstance(capability, Capability) for capability in self.blocked_capabilities + ): + raise ValueError("request blocked_capabilities must be a tuple of Capability values") + if len(set(self.blocked_capabilities)) != len(self.blocked_capabilities): + raise ValueError("request blocked_capabilities must not contain duplicates") + if not isinstance(self.approved_capabilities, tuple) or not all( + isinstance(capability, Capability) for capability in self.approved_capabilities + ): + raise ValueError("request approved_capabilities must be a tuple of Capability values") + if len(set(self.approved_capabilities)) != len(self.approved_capabilities): + raise ValueError("request approved_capabilities must not contain duplicates") + if not isinstance(self.output_policy, OutputPolicy): + raise ValueError("request output_policy must be an OutputPolicy value") + if not isinstance(self.post_run_action, PostRunAction): + raise ValueError("request post_run_action must be a PostRunAction value") + if self.post_run_action is not PostRunAction.NONE and self.output_policy is not OutputPolicy.PACKAGE: + raise ValueError("post-run actions require packaged output") + + +class EventLogger(ABC): + """A structured event sink scoped to one run or collector worker.""" + + @abstractmethod + def event(self, level: str, message: str, *, console: bool = True, **fields: float | str) -> None: + """Record one machine-readable event without exposing raw evidence.""" + + +class CollectorContext: + """The limited, per-worker interface exposed to a collector.""" + + def __init__( + self, + run_id: str, + collector_id: str, + workspace: Path, + temporary_directory: Path, + artifacts: ArtifactWriter, + logger: EventLogger, + settings: Mapping[str, object], + cancellation_file: Path, + ) -> None: + """Initialize the limited worker context and its cancellation boundary.""" + self.run_id = run_id + self.collector_id = collector_id + self.workspace = workspace + self.temporary_directory = temporary_directory + self.artifacts = artifacts + self.logger = logger + self.settings = settings + self._cancellation_file = cancellation_file + + @property + def is_cancelled(self) -> bool: + """Whether the supervisor has requested cancellation.""" + return self._cancellation_file.exists() + + def setting_int(self, name: str, default: int) -> int: + """Return one integer collector setting or its default.""" + value = self.settings.get(name, default) + if isinstance(value, int) and not isinstance(value, bool): + return value + return default + + def setting_float(self, name: str, default: float) -> float: + """Return one numeric collector setting or its default.""" + value = self.settings.get(name, default) + if isinstance(value, (int, float)) and not isinstance(value, bool): + return float(value) + return default + + def setting_str(self, name: str, default: str) -> str: + """Return one textual collector setting or its default.""" + value = self.settings.get(name, default) + return value if isinstance(value, str) else default + + def report_progress(self, event: str, **metrics: float | str) -> None: + """Write a structured progress event local to this collector workspace.""" + self.logger.event("info", event, **metrics) + + +class ArtifactWriter(ABC): + """The only supported route from a collector workspace into run artifacts.""" + + @abstractmethod + def register_file( + self, + source: Path, + *, + media_type: str = "application/octet-stream", + evidence_kind: EvidenceKind = EvidenceKind.DERIVED, + transformations: tuple[str, ...] = (), + ) -> Artifact: + """Register a file created within the collector workspace.""" + + +class Collector(ABC): + """Base interface that all v4 collector classes must implement.""" + + @classmethod + @abstractmethod + def metadata(cls) -> CollectorMetadata: + """Return immutable collector metadata without performing collection.""" + + @abstractmethod + def validate(self, context: CollectorContext) -> ValidationResult: + """Validate prerequisites without creating evidence artifacts.""" + + @staticmethod + def prepare(_context: CollectorContext) -> ValidationResult: + """Prepare collector-local resources after validation and authorization.""" + return ValidationResult(True) + + @abstractmethod + def collect(self, context: CollectorContext) -> CollectorResult: + """Collect evidence and return a normalized result.""" + + @staticmethod + def finalize(_context: CollectorContext, result: CollectorResult) -> CollectorResult: + """Finalize collector-local evidence and return the publishable result.""" + return result + + @staticmethod + def estimate(_context: CollectorContext) -> CollectionEstimate: + """Optionally estimate time and output without collecting evidence.""" + return CollectionEstimate(estimated_seconds=0, estimated_output_bytes=0) + + @classmethod + def dependencies(cls) -> tuple[str, ...]: + """Optionally declare collector IDs that must complete before this collector.""" + return () + + def cleanup(self, context: CollectorContext) -> None: + """Release collector-local resources. The engine owns filesystem cleanup.""" + + +class CoreCollector(Collector, ABC): + """Marker base class for collectors shipped with Logicytics.""" + + +class PluginCollector(Collector, ABC): + """Marker base class for user-owned collectors discovered in plugins/.""" diff --git a/logicytics/module/discovery.py b/logicytics/module/discovery.py new file mode 100644 index 00000000..62665b0b --- /dev/null +++ b/logicytics/module/discovery.py @@ -0,0 +1,823 @@ +"""Strict, side-effect-free discovery and preflight for collector modules.""" + +from __future__ import annotations + +import ast +import hashlib +import json +import os +import re +import sys +import tempfile +from collections.abc import Callable +from dataclasses import asdict, dataclass, field +from pathlib import Path +from subprocess import TimeoutExpired + +from logicytics.module.contracts import ( + CONTRACT_VERSION, + Capability, + CollectorKind, + CollectorMetadata, + Specialty, +) +from logicytics.module.platform_adapters import process_adapter + +_FILENAME = re.compile(r"^[a-z][a-z0-9_]*\.py$") +_VAGUE_NAMES = {"main.py", "misc.py", "stuff.py", "utils.py"} +_APPLICATION_IMPORTS = { + "CollectorSnapshot", + "RunSnapshot", + "api", + "artifacts", + "cli", + "configuration", + "discovery", + "environment", + "load_configuration", + "manifest", + "packaging", + "plan_run", + "planner", + "query_run", + "read_artifact", + "runtime", + "run_collection", + "validation_worker", +} +_CACHE_SCHEMA_VERSION = 2 +_COLLECTOR_SERVICE_MODULES = { + "logicytics.contracts", + "logicytics.module.contracts", + "logicytics.global.ctypes_collector", + "logicytics.platform_adapters", + "logicytics.module.platform_adapters", +} + +PreflightProgress = Callable[[str, int, int, str], None] + + +@dataclass(frozen=True, slots=True) +class ValidationDiagnostic: + """One stable, source-addressable collector validation failure.""" + + path: str + line: int + rule: str + message: str + + +@dataclass(slots=True) +class CollectorCandidate: + """A file that may be a runnable collector.""" + + path: Path + kind: CollectorKind + expected_class: str + static_errors: list[str] = field(default_factory=list) + metadata: CollectorMetadata | None = None + runtime_error: str | None = None + execution_type: str = "collector" + + @property + def valid(self) -> bool: + """Whether static and runtime validation both passed.""" + return not self.static_errors and self.metadata is not None and self.runtime_error is None + + @property + def selection_id(self) -> str: + """Return the ID used to explicitly select this path even when metadata is invalid.""" + if self.metadata is not None: + return self.metadata.id + owner = self.path.parent.name if self.path.name == "main.py" else self.path.stem + return f"{self.kind.value}.{owner}" + + @property + def diagnostics(self) -> tuple[ValidationDiagnostic, ...]: + """Normalize free-form validator details into exact source diagnostics.""" + messages = [*self.static_errors] + if self.runtime_error: + messages.append(self.runtime_error) + return tuple( + ValidationDiagnostic( + path=str(self.path), + line=_diagnostic_line(self.path, message), + rule=_diagnostic_rule(message, runtime=index >= len(self.static_errors)), + message=message, + ) + for index, message in enumerate(messages) + ) + + +@dataclass(frozen=True, slots=True) +class PreflightReport: + """Complete validation outcome for all discovered collector candidates.""" + + candidates: tuple[CollectorCandidate, ...] + + @property + def valid(self) -> tuple[CollectorCandidate, ...]: + """Return candidates that passed both static and runtime validation.""" + return tuple(candidate for candidate in self.candidates if candidate.valid) + + @property + def invalid(self) -> tuple[CollectorCandidate, ...]: + """Return candidates that must be blocked or quarantined.""" + return tuple(candidate for candidate in self.candidates if not candidate.valid) + + def to_dict( + self, + *, + selected_plugins: tuple[str, ...] = (), + enable_plugins: bool = False, + ) -> dict[str, list[dict[str, object]]]: + """Classify invalid plugins as quarantined unless the request selects them.""" + valid = [ + {"id": candidate.metadata.id, "kind": candidate.kind.value, "path": str(candidate.path)} + for candidate in self.valid + if candidate.metadata is not None + ] + invalid: list[dict[str, object]] = [] + quarantined: list[dict[str, object]] = [] + selected = set(selected_plugins) + for candidate in self.invalid: + item = { + "id": candidate.selection_id, + "kind": candidate.kind.value, + "path": str(candidate.path), + "diagnostics": [asdict(diagnostic) for diagnostic in candidate.diagnostics], + } + if ( + candidate.kind is CollectorKind.CORE + or (enable_plugins and candidate.kind is CollectorKind.PLUGIN) + or candidate.selection_id in selected + ): + invalid.append(item) + else: + quarantined.append(item) + return {"valid": valid, "quarantined": quarantined, "invalid": invalid} + + +def _diagnostic_rule(message: str, *, runtime: bool) -> str: + """Return a stable machine-readable rule for one validator message.""" + if runtime: + return "runtime.contract" + normalized = message.casefold() + rules = ( + ("filename", "static.filename"), + ("docstring", "static.docstring"), + ("print", "static.console_output"), + ("import-time", "static.import_time_side_effect"), + ("application import", "static.engine_boundary"), + ("top-level", "static.top_level_statement"), + ("inherit", "static.inheritance"), + ("collector class", "static.class_name"), + ("return", "static.return_annotation"), + ("parameter", "static.method_signature"), + ("classmethod", "static.method_signature"), + ("method", "static.method_contract"), + ) + return next((rule for marker, rule in rules if marker in normalized), "static.module_contract") + + +def _diagnostic_line(path: Path, message: str) -> int: + """Locate the exact source line implicated by a normalized validation message.""" + explicit = re.search(r"\(line (\d+)\)", message) + if explicit: + return int(explicit.group(1)) + traceback_lines = re.findall(r'File "([^"]+)", line (\d+)', message) + resolved = path.resolve() + for source, line in reversed(traceback_lines): + try: + if Path(source).resolve() == resolved: + return int(line) + except OSError: + continue + try: + tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) + except (OSError, UnicodeDecodeError, SyntaxError) as error: + return max(1, int(getattr(error, "lineno", 1) or 1)) + method_match = re.match(r"(metadata|validate|collect|cleanup|estimate|dependencies)\b", message) + if method_match: + for node in ast.walk(tree): + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)) and node.name == method_match.group(1): + return node.lineno + if "print" in message: + for node in ast.walk(tree): + if isinstance(node, ast.Call) and _top_level_call_name(node) == "print": + return node.lineno + if "class" in message or "inherit" in message: + collector_class = next((node for node in tree.body if isinstance(node, ast.ClassDef)), None) + if collector_class is not None: + return collector_class.lineno + return 1 + + +def _pascal_case(filename: str) -> str: + """Convert a snake-case collector filename to its required class name.""" + return "".join(part.capitalize() for part in Path(filename).stem.split("_")) + "Collector" + + +def _is_within(path: Path, root: Path) -> bool: + """Return whether a candidate path resolves inside its discovery root.""" + try: + path.resolve().relative_to(root.resolve()) + except ValueError: + return False + return True + + +def _iter_candidates(project_root: Path, kind: CollectorKind) -> list[Path]: + """Enumerate eligible collector files for one trusted or opt-in source tree.""" + root = project_root / ("core" if kind is CollectorKind.CORE else "plugins") + if not root.exists(): + return [] + paths: list[Path] = [] + for path in sorted(item for item in root.rglob("*") if item.is_file()): + relative = path.relative_to(root) + if any( + part.startswith(("_", ".")) or part in {"tests", "examples", "venv", ".venv"} + for part in relative.parts + ): + continue + if not _is_within(path, root): + continue + if kind is CollectorKind.CORE and len(relative.parts) != 2: + continue + if kind is CollectorKind.PLUGIN and not (len(relative.parts) == 1 or relative.name == "main.py"): + continue + paths.append(path) + return paths + + +def _top_level_call_name(node: ast.Call) -> str | None: + """Return a stable dotted name for a simple AST call expression.""" + if isinstance(node.func, ast.Name): + return node.func.id + if isinstance(node.func, ast.Attribute) and isinstance(node.func.value, ast.Name): + return f"{node.func.value.id}.{node.func.attr}" + return None + + +class _ImportTimeCallFinder(ast.NodeVisitor): + """Find calls whose expressions execute while Python imports a module.""" + + def __init__(self) -> None: + """Initialize an empty call collector for import-time AST expressions.""" + self.calls: list[ast.Call] = [] + + def visit_Call(self, node: ast.Call) -> None: + """Record a call and avoid duplicate nested call diagnostics.""" + self.calls.append(node) + + def visit_FunctionDef(self, node: ast.FunctionDef) -> None: + """Inspect decorators, defaults, and annotations but not a function body.""" + self._visit_callable_header(node) + + def visit_AsyncFunctionDef(self, node: ast.AsyncFunctionDef) -> None: + """Inspect async callable headers without treating deferred bodies as imports.""" + self._visit_callable_header(node) + + def visit_Lambda(self, node: ast.Lambda) -> None: + """Inspect lambda defaults without treating its deferred body as import-time work.""" + self._visit_arguments(node.args) + + def visit_ClassDef(self, node: ast.ClassDef) -> None: + """Inspect class headers and class-body expressions evaluated at definition time.""" + for decorator in node.decorator_list: + self.visit(decorator) + for base in node.bases: + self.visit(base) + for keyword in node.keywords: + self.visit(keyword.value) + for item in node.body: + if isinstance(item, ast.Expr) and isinstance(item.value, ast.Constant): + continue + self.visit(item) + + def _visit_callable_header(self, node: ast.FunctionDef | ast.AsyncFunctionDef) -> None: + """Visit decorators, annotations, and defaults evaluated at definition time.""" + for decorator in node.decorator_list: + self.visit(decorator) + self._visit_arguments(node.args) + if node.returns is not None: + self.visit(node.returns) + + def _visit_arguments(self, arguments: ast.arguments) -> None: + """Visit argument defaults and annotations that execute during module import.""" + for default in (*arguments.defaults, *arguments.kw_defaults): + if default is not None: + self.visit(default) + for argument in (*arguments.posonlyargs, *arguments.args, *arguments.kwonlyargs): + if argument.annotation is not None: + self.visit(argument.annotation) + if arguments.vararg is not None and arguments.vararg.annotation is not None: + self.visit(arguments.vararg.annotation) + if arguments.kwarg is not None and arguments.kwarg.annotation is not None: + self.visit(arguments.kwarg.annotation) + + +def _validate_import_time_expressions(tree: ast.Module, candidate: CollectorCandidate) -> None: + """Reject calls that can collect, mutate, or exit before worker isolation exists.""" + finder = _ImportTimeCallFinder() + finder.visit(tree) + for call in finder.calls: + name = _top_level_call_name(call) or ast.unparse(call.func) + candidate.static_errors.append(f"forbidden import-time call: {name} (line {call.lineno})") + + +def _validate_engine_boundary_imports(tree: ast.Module, candidate: CollectorCandidate) -> None: + """Keep collector modules on contracts only, never application or orchestration services.""" + application_aliases: set[str] = set() + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + if alias.name.startswith("logicytics.") and alias.name not in _COLLECTOR_SERVICE_MODULES: + candidate.static_errors.append(f"forbidden application import: {alias.name} (line {node.lineno})") + elif alias.name == "logicytics": + application_aliases.add(alias.asname or "logicytics") + elif isinstance(node, ast.ImportFrom): + module = node.module or "" + if module.startswith("logicytics.") and module not in _COLLECTOR_SERVICE_MODULES: + candidate.static_errors.append(f"forbidden application import: {module} (line {node.lineno})") + elif module == "logicytics": + forbidden = sorted(alias.name for alias in node.names if alias.name in _APPLICATION_IMPORTS) + if forbidden: + candidate.static_errors.append( + f"forbidden application import: {', '.join(forbidden)} (line {node.lineno})") + for node in ast.walk(tree): + if ( + isinstance(node, ast.Attribute) + and isinstance(node.value, ast.Name) + and node.value.id in application_aliases + and node.attr in _APPLICATION_IMPORTS + ): + candidate.static_errors.append( + f"forbidden application import: {node.value.id}.{node.attr} (line {node.lineno})") + + +def _validate_static(path: Path, kind: CollectorKind) -> CollectorCandidate: + """Apply filename, AST shape, import-boundary, and side-effect rules to a candidate.""" + expected_class = _pascal_case(path.name) + candidate = CollectorCandidate(path=path, kind=kind, expected_class=expected_class) + if path.name in _VAGUE_NAMES and not (kind is CollectorKind.PLUGIN and path.name == "main.py"): + candidate.static_errors.append("collector filename is reserved or too vague") + if not _FILENAME.fullmatch(path.name): + candidate.static_errors.append("filename must be lowercase snake_case.py") + try: + tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) + except (OSError, UnicodeDecodeError, SyntaxError) as error: + candidate.static_errors.append(f"cannot parse UTF-8 Python module: {error}") + return candidate + if ast.get_docstring(tree) is None: + candidate.static_errors.append("module requires a docstring") + if any(isinstance(node, ast.Call) and _top_level_call_name(node) == "print" for node in ast.walk(tree)): + candidate.static_errors.append("collectors must not print; use structured progress or logging") + + public_classes = [item for item in tree.body if isinstance(item, ast.ClassDef) and not item.name.startswith("_")] + if len(public_classes) != 1: + candidate.static_errors.append("module must define exactly one public collector class") + elif public_classes[0].name != expected_class: + candidate.static_errors.append(f"collector class must be named {expected_class}") + elif not any( + (isinstance(base, ast.Name) and base.id == ( + "CoreCollector" if kind is CollectorKind.CORE else "PluginCollector")) + or (isinstance(base, ast.Attribute) and base.attr == ( + "CoreCollector" if kind is CollectorKind.CORE else "PluginCollector")) + for base in public_classes[0].bases + ): + candidate.static_errors.append("collector class must inherit from its required base class") + elif public_classes: + _validate_class_shape(public_classes[0], candidate) + + _validate_import_time_expressions(tree, candidate) + _validate_engine_boundary_imports(tree, candidate) + for item in tree.body: + if isinstance(item, ast.Expr) and isinstance(item.value, ast.Constant): + continue + if isinstance(item, (ast.Assign, ast.AnnAssign)): + continue + if isinstance(item, (ast.Import, ast.ImportFrom, ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)): + continue + candidate.static_errors.append(f"forbidden top-level statement: {type(item).__name__}") + return candidate + + + +def _validate_class_shape( + class_node: ast.ClassDef, + candidate: CollectorCandidate, +) -> None: + """Enforce the small, documented public API allowed on a collector class.""" + if ast.get_docstring(class_node) is None: + candidate.static_errors.append("collector class requires a docstring") + + methods: dict[str, ast.FunctionDef | ast.AsyncFunctionDef] = {} + + for item in class_node.body: + if isinstance(item, (ast.FunctionDef, ast.AsyncFunctionDef)): + if not item.name.startswith("_"): + methods[item.name] = item + + allowed = { + "metadata", + "validate", + "prepare", + "collect", + "finalize", + "cleanup", + "estimate", + "dependencies", + } + + metadata_calls: list[ast.Call] = [] + + metadata_method = methods.get("metadata") + if metadata_method is not None: + for node in ast.walk(metadata_method): + if not isinstance(node, ast.Call): + continue + + is_metadata_constructor = (isinstance(node.func, ast.Name) and node.func.id == "CollectorMetadata") or ( + isinstance(node.func, ast.Attribute) and node.func.attr == "CollectorMetadata" + ) + + if is_metadata_constructor: + metadata_calls.append(node) + + if metadata_method is not None and len(metadata_calls) != 1: + candidate.static_errors.append("metadata must construct one CollectorMetadata object directly") + + if candidate.kind is CollectorKind.PLUGIN and metadata_method is not None: + required_plugin_fields = { + "capabilities", + "privilege_level", + "sensitive_data_categories", + "network_access", + "estimated_cost", + "timeout_seconds", + "maximum_output_bytes", + "output_media_types", + "minimum_contract_version", + } + + if len(metadata_calls) == 1: + metadata_call = metadata_calls[0] + + declared_fields = {keyword.arg for keyword in metadata_call.keywords if keyword.arg is not None} + + missing_fields = sorted(required_plugin_fields - declared_fields) + + if missing_fields: + candidate.static_errors.append(f"plugin metadata must explicitly declare: {', '.join(missing_fields)}") + + if candidate.kind is CollectorKind.CORE and len(metadata_calls) == 1: + if not any(keyword.arg == "capabilities" for keyword in metadata_calls[0].keywords): + candidate.static_errors.append( + "CAPABILITY_METADATA_MISSING: core metadata must explicitly declare capabilities") + + collect_method = methods.get("collect") + + if collect_method is not None and len(metadata_calls) == 1: + metadata_call = metadata_calls[0] + + declared_keyword = next( + (keyword for keyword in metadata_call.keywords if keyword.arg == "output_media_types"), + None, + ) + + if declared_keyword is None: + candidate.static_errors.append("metadata must explicitly declare output_media_types") + else: + try: + evaluated_media_types = ast.literal_eval(declared_keyword.value) + declared_media_types = set(evaluated_media_types) + except (TypeError, ValueError): + candidate.static_errors.append("metadata output_media_types must be a literal tuple") + else: + registered_media_types: set[str] = set() + + lifecycle_methods: list[ast.FunctionDef | ast.AsyncFunctionDef] = [collect_method] + + finalize_method = methods.get("finalize") + if finalize_method is not None: + lifecycle_methods.append(finalize_method) + + for method in lifecycle_methods: + for call in ast.walk(method): + if not isinstance(call, ast.Call): + continue + + if not (isinstance(call.func, ast.Attribute) and call.func.attr == "register_file"): + continue + + media_keyword = next( + (keyword for keyword in call.keywords if keyword.arg == "media_type"), + None, + ) + + if media_keyword is None: + registered_media_types.add("application/octet-stream") + elif isinstance(media_keyword.value, ast.Constant) and isinstance(media_keyword.value.value, + str): + registered_media_types.add(media_keyword.value.value) + else: + candidate.static_errors.append( + f"register_file media_type must be a literal string (line {call.lineno})") + + if not registered_media_types.issubset(declared_media_types): + candidate.static_errors.append( + "metadata output_media_types must include every registered artifact type") + + unknown = sorted(set(methods) - allowed) + + if unknown: + candidate.static_errors.append(f"unsupported public collector methods: {', '.join(unknown)}") + + required = { + "metadata", + "validate", + "collect", + "cleanup", + } + + expected_returns = { + "metadata": "CollectorMetadata", + "validate": "ValidationResult", + "prepare": "ValidationResult", + "collect": "CollectorResult", + "finalize": "CollectorResult", + "estimate": "CollectionEstimate", + "dependencies": "tuple[str,...]", + "cleanup": "None", + } + + missing = sorted(required - set(methods)) + + if missing: + candidate.static_errors.append(f"missing required collector methods: {', '.join(missing)}") + + for name, method in methods.items(): + if ast.get_docstring(method) is None: + candidate.static_errors.append(f"{name} requires a docstring") + + if method.returns is None: + candidate.static_errors.append(f"{name} requires a return type annotation") + elif name in expected_returns: + actual_return = ast.unparse(method.returns).replace(" ", "") + expected_return = expected_returns[name] + + if actual_return != expected_return: + candidate.static_errors.append(f"{name} must return {expected_return}") + + parameters = method.args.args + + expected_parameter_count = 1 if name in {"metadata", "dependencies"} else 3 if name == "finalize" else 2 + + if len(parameters) != expected_parameter_count: + candidate.static_errors.append(f"{name} has an invalid parameter count") + continue + + expected_first = "cls" if name in {"metadata", "dependencies"} else "self" + + if parameters[0].arg != expected_first: + candidate.static_errors.append(f"{name} must begin with {expected_first}") + + if name in {"metadata", "dependencies"}: + is_classmethod = any(isinstance(decorator, ast.Name) and decorator.id == "classmethod" for decorator in + method.decorator_list) + + if not is_classmethod: + candidate.static_errors.append(f"{name} must be a classmethod") + + continue + + context_parameter = parameters[1] + + if context_parameter.annotation is None: + context_annotation = None + else: + context_annotation = ast.unparse(context_parameter.annotation).rsplit(".", 1)[-1] + + if context_parameter.arg != "context" or context_annotation != "CollectorContext": + candidate.static_errors.append(f"{name} must accept an annotated context parameter") + + if name == "finalize": + result_parameter = parameters[2] + + if result_parameter.annotation is None: + result_annotation = None + else: + result_annotation = ast.unparse(result_parameter.annotation).rsplit(".", 1)[-1] + + if result_parameter.arg != "result" or result_annotation != "CollectorResult": + candidate.static_errors.append("finalize must accept an annotated result parameter") + + +def discover(project_root: Path) -> tuple[CollectorCandidate, ...]: + """Discover candidates without importing their modules.""" + candidates = [_validate_static(path, CollectorKind.CORE) for path in + _iter_candidates(project_root, CollectorKind.CORE)] + candidates.extend( + _validate_static(path, CollectorKind.PLUGIN) for path in _iter_candidates(project_root, CollectorKind.PLUGIN)) + return tuple(candidates) + + +def _accept_runtime_metadata(candidate: CollectorCandidate, payload: object) -> None: + """Validate cached or freshly probed metadata through the same strict path.""" + try: + if not isinstance(payload, dict): + raise ValueError("metadata payload must be an object") + metadata = CollectorMetadata.from_dict( + payload, + allow_custom_specialty=candidate.kind is not CollectorKind.CORE, + ) + except (KeyError, TypeError, ValueError) as error: + candidate.runtime_error = f"invalid validation response: {error}" + return + + specialty = metadata.specialty.value if isinstance(metadata.specialty, Specialty) else metadata.specialty + + if metadata.minimum_contract_version != CONTRACT_VERSION: + candidate.runtime_error = "collector contract version is unsupported" + elif metadata.timeout_seconds < 1 or metadata.maximum_output_bytes < 1: + candidate.runtime_error = "collector must declare positive timeout and output limits" + elif not metadata.id.startswith(f"{candidate.kind.value}."): + candidate.runtime_error = "collector ID must start with its owner kind" + elif candidate.kind is CollectorKind.CORE and specialty != candidate.path.parent.name: + candidate.runtime_error = "core collector specialty must match its parent folder" + elif candidate.kind is CollectorKind.CORE and metadata.id != ( + f"core.{candidate.path.parent.name}.{candidate.path.stem}"): + candidate.runtime_error = "core collector ID must match core//.py" + elif candidate.kind is CollectorKind.PLUGIN and metadata.id != ( + f"plugin.{candidate.path.parent.name if candidate.path.name == 'main.py' else candidate.path.stem}" + ): + candidate.runtime_error = "plugin collector ID must match its plugin folder or filename" + else: + candidate.metadata = metadata + + +def _runtime_probe(project_root: Path, candidate: CollectorCandidate, temporary_root: Path | None = None) -> None: + """Probe metadata in a short-lived restricted worker after static validation.""" + engine_root = Path(__file__).resolve().parents[2] + pythonpath = os.pathsep.join((str(engine_root), str(project_root))) + command = [ + sys.executable, + "-m", + "logicytics.module.validation_worker", + str(candidate.path), + candidate.kind.value, + candidate.expected_class, + ] + if temporary_root is not None: + temporary_root.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix="logicytics-preflight-", dir=temporary_root) as temporary: + probe_directory = Path(temporary) + environment = { + "PYTHONPATH": pythonpath, + "LOGICYTICS_VALIDATION": "1", + "PATH": os.environ.get("PATH", ""), + "TEMP": str(probe_directory), + "TMP": str(probe_directory), + "PYTHONUTF8": "1", + } + for name in ("SYSTEMROOT", "WINDIR", "COMSPEC"): + if value := os.environ.get(name): + environment[name] = value + try: + completed = process_adapter.run( + command, + cwd=probe_directory, + capture_output=True, + text=True, + timeout=10, + check=False, + env=environment, + ) + except (OSError, TimeoutExpired) as error: + candidate.runtime_error = f"validation worker failed: {error}" + return + if completed.returncode != 0: + candidate.runtime_error = completed.stderr.strip() or "validation worker rejected collector" + return + try: + payload = json.loads(completed.stdout) + metadata_payload = payload["metadata"] + except (KeyError, TypeError, json.JSONDecodeError) as error: + candidate.runtime_error = f"invalid validation response: {error}" + return + _accept_runtime_metadata(candidate, metadata_payload) + + +def _cache_path(project_root: Path, cache_directory: Path | None = None) -> Path: + """Keep disposable validation state outside source and evidence directories.""" + project_key = hashlib.sha256(str(project_root.resolve()).encode("utf-8")).hexdigest() + root = cache_directory if cache_directory is not None else Path( + tempfile.gettempdir()) / "logicytics-preflight-cache" + return root / f"{project_key}.json" + + +def _source_hash(path: Path) -> str: + """Hash exact collector bytes so any source edit invalidates its probe result.""" + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _load_cache(project_root: Path, configuration_hash: str, cache_directory: Path | None = None) -> dict[str, object]: + """Load only a cache created for this interpreter, contract, and configuration.""" + path = _cache_path(project_root, cache_directory) + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError): + return {} + expected = { + "schema_version": _CACHE_SCHEMA_VERSION, + "interpreter": sys.version, + "contract_version": CONTRACT_VERSION, + "configuration_hash": configuration_hash, + } + if not isinstance(payload, dict) or any(payload.get(key) != value for key, value in expected.items()): + return {} + entries = payload.get("entries") + return entries if isinstance(entries, dict) else {} + + +def _write_cache( + project_root: Path, + configuration_hash: str, + candidates: list[CollectorCandidate], + cache_directory: Path | None = None, +) -> None: + """Atomically persist successful probes; invalid candidates are always reprobed.""" + entries: dict[str, object] = {} + for candidate in candidates: + if candidate.metadata is None or candidate.runtime_error or candidate.static_errors: + continue + relative_path = candidate.path.resolve().relative_to(project_root.resolve()).as_posix() + entries[relative_path] = { + "kind": candidate.kind.value, + "source_hash": _source_hash(candidate.path), + "metadata": candidate.metadata.to_dict(), + } + payload = { + "schema_version": _CACHE_SCHEMA_VERSION, + "interpreter": sys.version, + "contract_version": CONTRACT_VERSION, + "configuration_hash": configuration_hash, + "entries": entries, + } + path = _cache_path(project_root, cache_directory) + path.parent.mkdir(parents=True, exist_ok=True) + temporary = path.with_suffix(".tmp") + temporary.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8") + temporary.replace(path) + + +def preflight( + project_root: Path, + *, + configuration_hash: str = "unconfigured", + progress: PreflightProgress | None = None, + invalidate_cache: bool = False, + cache_directory: Path | None = None, + temporary_directory: Path | None = None, +) -> PreflightReport: + """Perform static checks then a short-lived isolated metadata probe.""" + if invalidate_cache: + cache_path = _cache_path(project_root, cache_directory) + try: + cache_path.unlink() + except FileNotFoundError: + pass + except OSError as error: + raise OSError(f"unable to invalidate preflight cache: {error}") from error + candidates = list(discover(project_root)) + cached = _load_cache(project_root, configuration_hash, cache_directory) + total = len(candidates) + for index, candidate in enumerate(candidates, start=1): + if progress is not None: + progress("checking", index - 1, total, candidate.selection_id) + if candidate.static_errors: + if progress is not None: + progress("checked", index, total, candidate.selection_id) + continue + relative_path = candidate.path.resolve().relative_to(project_root.resolve()).as_posix() + entry = cached.get(relative_path) + if ( + isinstance(entry, dict) + and entry.get("kind") == candidate.kind.value + and entry.get("source_hash") == _source_hash(candidate.path) + ): + _accept_runtime_metadata(candidate, entry.get("metadata")) + else: + _runtime_probe(project_root, candidate, temporary_directory) + if progress is not None: + progress("checked", index, total, candidate.selection_id) + seen_ids: set[str] = set() + for candidate in candidates: + if candidate.metadata is None: + continue + if candidate.metadata.id in seen_ids: + candidate.runtime_error = f"duplicate collector ID: {candidate.metadata.id}" + seen_ids.add(candidate.metadata.id) + _write_cache(project_root, configuration_hash, candidates, cache_directory) + return PreflightReport(tuple(candidates)) diff --git a/logicytics/module/environment.py b/logicytics/module/environment.py new file mode 100644 index 00000000..036c77b9 --- /dev/null +++ b/logicytics/module/environment.py @@ -0,0 +1,54 @@ +"""Read-only Windows environment checks used before v4 planning and execution.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass + +from logicytics.module.platform_adapters import ( + process_adapter, + registry_adapter, + which, + windows_api_adapter, +) + + +@dataclass(frozen=True, slots=True) +class EnvironmentReport: + """Local privilege, UAC, and PowerShell-policy state.""" + + is_administrator: bool | None + uac_enabled: bool | None + powershell_execution_policy: str | None + + def to_dict(self) -> dict[str, bool | str | None]: + """Return a JSON-safe environment report.""" + return asdict(self) + + +def inspect_environment() -> EnvironmentReport: + """Read local environment state without changing policy or elevation.""" + is_administrator = windows_api_adapter.is_administrator() + try: + with registry_adapter.OpenKey( + registry_adapter.HKEY_LOCAL_MACHINE, + r"SOFTWARE\Microsoft\Windows\CurrentVersion\Policies\System", + ) as key: + uac_enabled = bool(registry_adapter.QueryValueEx(key, "EnableLUA")[0]) + except OSError: + uac_enabled = None + policy = None + powershell = which("powershell") + if powershell is not None: + try: + completed = process_adapter.run( + [powershell, "-NoProfile", "-NonInteractive", "-Command", "Get-ExecutionPolicy"], + capture_output=True, + check=False, + text=True, + timeout=15, + ) + if completed.returncode == 0: + policy = completed.stdout.strip() or None + except OSError: + pass + return EnvironmentReport(is_administrator, uac_enabled, policy) diff --git a/logicytics/module/errors.py b/logicytics/module/errors.py new file mode 100644 index 00000000..21f3ee54 --- /dev/null +++ b/logicytics/module/errors.py @@ -0,0 +1,32 @@ +"""Domain exceptions raised by the v4 engine.""" + +from __future__ import annotations + + +class LogicyticsError(Exception): + """Base error for expected application-level failures.""" + + +class PreflightError(LogicyticsError): + """Raised when selected collectors fail strict preflight validation.""" + + +class PlanError(LogicyticsError): + """Raised when a requested run cannot be converted into a valid plan.""" + + +class ArtifactError(LogicyticsError): + """Raised when an artifact violates workspace or output rules.""" + + +class CapabilityPolicyError(PermissionError): + """Raised when a collector's declared capability policy is violated.""" + + def __init__(self, code: str, capability: str, operation: str, detail: str = "") -> None: + """Create a stable, machine-searchable capability diagnostic.""" + self.code = code + self.capability = capability + self.operation = operation + action = "requested blocked" if code == "CAPABILITY_BLOCKED" else "used undeclared" + suffix = f"; {detail}" if detail else "" + super().__init__(f"{code}: collector {action} {capability} capability; operation={operation}{suffix}") diff --git a/logicytics/module/interaction.py b/logicytics/module/interaction.py new file mode 100644 index 00000000..b9ef479f --- /dev/null +++ b/logicytics/module/interaction.py @@ -0,0 +1,220 @@ +"""Local semantic flag matching, opt-in history, and usage reporting.""" + +from __future__ import annotations + +import gzip +import json +import platform +import re +from collections import Counter +from collections.abc import Iterable, Mapping +from dataclasses import asdict, dataclass +from datetime import UTC, datetime +from difflib import SequenceMatcher +from html import escape +from pathlib import Path + +FLAG_DESCRIPTIONS: Mapping[str, str] = { + "default": "standard collection sequentially", + "threaded": "standard collection with bounded parallel workers", + "minimal": "quick basic essential collection", + "depth": "deep exhaustive slow collection", + "performance-check": "sequential collector duration performance analysis", + "usage": "interaction statistics and flag usage graph", + "debug": "diagnostic environment configuration and integrity checks", + "update": "check or explicitly update the Git checkout", + "dev": "developer contribution manifest and version checks", + "mode": "select a user facing typed execution mode", + "modes": "show execution mode profiles scheduling and compatibility aliases", +} +_TOKEN = re.compile(r"[a-z0-9]+") + + +@dataclass(frozen=True, slots=True) +class FlagMatch: + """One deterministic flag suggestion and its evidence.""" + + input: str + matched_flag: str | None + accuracy: float + source: str + model_name: str + + +def _normalized(value: str) -> str: + """Normalize free-form input into lowercase, whitespace-separated tokens.""" + return " ".join(_TOKEN.findall(value.casefold())) + + +def _score(query: str, flag: str, description: str) -> float: + """Score a query using sequence similarity and token overlap with a flag description.""" + normalized_query = _normalized(query) + candidate = _normalized(f"{flag} {description}") + sequence = SequenceMatcher(None, normalized_query, candidate).ratio() + query_tokens = set(normalized_query.split()) + candidate_tokens = set(candidate.split()) + overlap = len(query_tokens.intersection(candidate_tokens)) / max(1, len(query_tokens)) + flag_score = SequenceMatcher(None, normalized_query, _normalized(flag)).ratio() + return round(max(sequence, overlap, flag_score), 6) + + +def match_flag( + user_input: str, + *, + threshold: float, + model_name: str, + history: Iterable[Mapping[str, object]] = (), +) -> FlagMatch: + """Match names and descriptions, then consult prior accepted inputs when weak.""" + if not isinstance(user_input, str) or not user_input.strip(): + raise ValueError("semantic flag input must be non-empty") + ranked = sorted( + ((_score(user_input, flag, description), flag) for flag, description in FLAG_DESCRIPTIONS.items()), + key=lambda item: (-item[0], item[1]), + ) + direct_score, direct_flag = ranked[0] + if direct_score >= threshold: + return FlagMatch(user_input, direct_flag, direct_score, "direct", model_name) + historical: list[tuple[float, str]] = [] + for item in history: + old_input = item.get("input") + old_flag = item.get("matched_flag") + if isinstance(old_input, str) and isinstance(old_flag, str) and old_flag in FLAG_DESCRIPTIONS: + historical.append( + ( + SequenceMatcher(None, _normalized(user_input), _normalized(old_input)).ratio(), + old_flag, + ) + ) + if historical: + history_score, history_flag = sorted(historical, key=lambda item: (-item[0], item[1]))[0] + if history_score > direct_score: + return FlagMatch(user_input, history_flag, round(history_score, 6), "history", model_name) + return FlagMatch(user_input, None, direct_score, "below_threshold", model_name) + + +def load_history(path: Path) -> list[dict[str, object]]: + """Read bounded compressed local history; malformed history fails closed.""" + if not path.exists(): + return [] + try: + with gzip.open(path, "rt", encoding="utf-8") as stream: + payload = json.load(stream) + except (OSError, UnicodeDecodeError, json.JSONDecodeError): + return [] + if not isinstance(payload, list): + return [] + return [dict(item) for item in payload[-10_000:] if isinstance(item, dict)] + + +def record_match(path: Path, match: FlagMatch) -> None: + """Atomically append one compressed, local-only interaction record.""" + history = load_history(path) + history.append( + { + **asdict(match), + "timestamp": datetime.now(UTC).isoformat(), + "device_name": platform.node() or "unknown", + } + ) + path.parent.mkdir(parents=True, exist_ok=True) + temporary = path.with_suffix(path.suffix + ".tmp") + with gzip.open(temporary, "wt", encoding="utf-8") as stream: + json.dump(history[-10_000:], stream, sort_keys=True, separators=(",", ":")) + temporary.replace(path) + + +def record_command(path: Path, command: str, *, mode: str | None = None) -> None: + """Record one local command interaction when history tracking is enabled.""" + if not command: + raise ValueError("command history requires a command name") + matched_flag = mode or command + input_value = command if mode is None else f"{command} --profile {mode}" + record_match( + path, + FlagMatch( + input=input_value, + matched_flag=matched_flag, + accuracy=1.0, + source="command", + model_name="command-history", + ), + ) + + +def usage_statistics(history: Iterable[Mapping[str, object]]) -> dict[str, object]: + """Aggregate total, accuracy, common values, and per-flag frequencies.""" + records = list(history) + + accuracies: list[float] = [] + flags: Counter[str] = Counter() + devices: Counter[str] = Counter() + inputs: Counter[str] = Counter() + + for item in records: + accuracy = item.get("accuracy") + matched_flag = item.get("matched_flag") + device_name = item.get("device_name") + input_value = item.get("input") + + if isinstance(accuracy, (int, float)): + accuracies.append(float(accuracy)) + + if isinstance(matched_flag, str): + flags[matched_flag] += 1 + + if isinstance(device_name, str): + devices[device_name] += 1 + + if isinstance(input_value, str): + inputs[input_value] += 1 + + return { + "total_interactions": len(records), + "average_accuracy": (round(sum(accuracies) / len(accuracies), 6) if accuracies else 0.0), + "common_device": devices.most_common(1)[0][0] if devices else None, + "common_input": inputs.most_common(1)[0][0] if inputs else None, + "per_flag_frequency": dict(sorted(flags.items())), + } + + +def write_usage_graph(path: Path, statistics: Mapping[str, object]) -> Path: + """Write a portable SVG graph for the currently recorded commands and modes.""" + raw_counts = statistics.get("per_flag_frequency", {}) + source_counts = raw_counts if isinstance(raw_counts, Mapping) else {} + counts = { + label: count + for label, count in source_counts.items() + if isinstance(label, str) + and isinstance(count, int) + and not isinstance(count, bool) + and count > 0 + } + labels = sorted(counts) + width, row_height = 760, 30 + height = 70 + row_height * max(len(labels), 1) + maximum = max(counts.values(), default=1) + rows: list[str] = [] + for index, label in enumerate(labels): + count = counts[label] + y = 48 + index * row_height + bar_width = int(500 * count / maximum) + rows.extend( + ( + f'{escape(label)}', + f'', + f'{count}', + ) + ) + if not rows: + rows.append('No tracked interactions yet.') + svg = ( + f'' + 'Logicytics command and mode usage' + + "".join(rows) + + "\n" + ) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(svg, encoding="utf-8") + return path diff --git a/logicytics/module/logging.py b/logicytics/module/logging.py new file mode 100644 index 00000000..f3718438 --- /dev/null +++ b/logicytics/module/logging.py @@ -0,0 +1,954 @@ +"""Run-scoped structured logging without global mutable logging configuration.""" + +from __future__ import annotations + +import argparse +import inspect +import json +import math +import re +import sys +import textwrap +import traceback +from collections.abc import Callable, Iterable, Mapping +from datetime import UTC, datetime +from pathlib import Path +from threading import RLock +from time import perf_counter, time +from typing import ParamSpec, TextIO, TypeVar + +from logicytics.module.contracts import EventLogger +from logicytics.module.configuration import LoggingSettings +from logicytics.module.presentation import ( + console_width as presentation_console_width, +) +from logicytics.module.presentation import ( + render_alert as render_presentation_alert, +) +from logicytics.module.presentation import ( + render_section as render_presentation_section, +) +from logicytics.module.presentation import ( + render_step_heading as render_presentation_step_heading, +) +from logicytics.module.redaction import redact_mapping, redact_text + +Parameters = ParamSpec("Parameters") +Result = TypeVar("Result") +_LEVEL_ORDER = { + "DEBUG": 10, + "INTERNAL": 15, + "INFO": 20, + "WARNING": 30, + "ERROR": 40, + "EXCEPTION": 45, + "CRITICAL": 50, +} +_LEVEL_PRESENTATION = { + "DEBUG": ("\u00b7", "\033[90m", "\033[90m"), + "INTERNAL": ("\u00b7", "\033[90m", "\033[90m"), + "INFO": ("\u25cf", "\033[96m", "\033[97m"), + "WARNING": ("!", "\033[93m", "\033[93m"), + "ERROR": ("\u00d7", "\033[91m", "\033[91m"), + "EXCEPTION": ("\u00d7", "\033[91m", "\033[91m"), + "CRITICAL": ("\u00d7", "\033[31m", "\033[31m"), +} +_DETAIL_MARKER_COLOR = "\033[95m" +_RESET = "\033[0m" +_BOLD = "\033[1m" +_FILE_LOG_LINE_WIDTH = 140 +_TIME_WIDTH = 23 +_SEVERITY_WIDTH = 9 +_SOURCE_WIDTH = 28 +_RECORD_START = re.compile(rb"(?m)^\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}(?:\.\d{3})? \|") +_EVENT_NAME = re.compile(r"^[a-z0-9]+(?:_[a-z0-9]+)+$") +_LOGGER_LOCK = RLock() +_APPLICATION_LOGGERS: dict[Path, ApplicationLogger] = {} +_EVENT_LOGGERS: dict[tuple[Path, str, str | None], FileEventLogger] = {} +_ANSI_ESCAPE = re.compile(r"\x1B(?:[@-Z\\-_]|\[[0-?]*[ -/]*[@-~])") + + +def _caller_location() -> tuple[str, int]: + """Return the first caller outside this logging module and its source line.""" + frame = inspect.currentframe() + try: + frame = frame.f_back if frame is not None else None + while frame is not None: + module = str(frame.f_globals.get("__name__", "")) + if module and module != __name__: + return module, frame.f_lineno + frame = frame.f_back + finally: + del frame + return __name__, 0 + + +def _script_initials(script: str) -> str: + """Use the first letter of each underscore-delimited script word.""" + return "".join(part[0] for part in script.split("_") if part) + + +def collector_log_source(collector_id: str | None) -> str | None: + """Render a stable human-facing source name for one collector identifier.""" + if not collector_id: + return None + parts = collector_id.split(".") + if len(parts) < 2: + return collector_id + if parts[0] == "core" and len(parts) >= 3: + return f"core.{parts[1]}.{_script_initials(parts[-1])}" + if parts[0] == "plugin": + return f"plugins.{'.'.join(parts[1:])}" + return collector_id + + +def _source_with_line(source: str, line: int, *, debug: bool) -> str: + """Add a source line in debug logs, except for library implementation sources.""" + if debug and line > 0 and not source.startswith("library."): + return f"{source}:{line}" + return source + + +def _normalize_level(level: str) -> str: + """Normalize one severity spelling and reject values no sink can interpret.""" + if not isinstance(level, str): + raise ValueError(f"unsupported log level: {level!r}") + normalized = level.strip().upper() + if normalized not in _LEVEL_ORDER: + raise ValueError(f"unsupported log level: {level}") + return normalized + + +class ApplicationLogger(EventLogger): + """Thread-safe human-readable file and colored-console event sink.""" + + def __init__( + self, + path: Path, + settings: LoggingSettings, + *, + console: TextIO | None = None, + ) -> None: + """Initialize a bounded application logger with explicit file and console sinks.""" + self.path = path.resolve() + self.settings = settings + self.console = sys.stderr if console is None else console + self._lock = RLock() + self._last_console_step: str | None = None + self.path.parent.mkdir(parents=True, exist_ok=True) + self._prepare_file() + + self._last_console_line = "" + self._progress_indent = "" + self._progress_checked = 0 + self._progress_total = 0 + self._progress_current = "" + + @staticmethod + def _visible_text(text: str) -> str: + """Return text with ANSI escape sequences removed.""" + return _ANSI_ESCAPE.sub("", text) + + def _leading_indent(self, text: str) -> str: + """Return the leading whitespace from visible terminal text.""" + visible = self._visible_text(text) + + return visible[: len(visible) - len(visible.lstrip())] + + def _remember_console_line(self, text: str) -> None: + """Remember the last visible non-empty console line.""" + visible = self._visible_text(text).rstrip("\n") + + if "\n" in visible: + visible = visible.split("\n")[-1] + + if visible: + self._last_console_line = visible + + @staticmethod + def _shorten_component(component: str) -> str: + """ + Shorten one dotted-name component to initials. + + Examples: + packet_capture -> pc + process_memory_map -> pmm + network -> n + collector -> c + """ + parts = [part for part in component.split("_") if part] + + if not parts: + return component + + if len(parts) == 1: + return parts[0][0] + + return "".join(part[0] for part in parts) + + def _shorten_current(self, current: str, max_width: int) -> str: + """ + Render dotted collector names in compact code form. + + The final component is always abbreviated from underscore-separated + words, while namespaces remain readable until width requires further + shortening. + + Examples: + collector.network.packet_capture + -> collector.network.pc + -> collector.n.pc + -> c.n.pc + + collector.system.process_memory_map + -> collector.system.pmm + -> collector.s.pmm + -> c.s.pmm + """ + if not current: + return "" + + components = current.split(".") + + if len(components) <= 1: + return current + + shortened = components.copy() + + # Always use compact code form for the final component. + shortened[-1] = self._shorten_component(shortened[-1]) + + result = ".".join(shortened) + + if len(result) <= max_width: + return result + + # Shorten namespace components from right to left while preserving + # the leading namespace for as long as possible. + for index in range(len(shortened) - 2, 0, -1): + shortened[index] = self._shorten_component(shortened[index]) + result = ".".join(shortened) + + if len(result) <= max_width: + return result + + # Finally abbreviate the first component. + shortened[0] = self._shorten_component(shortened[0]) + result = ".".join(shortened) + + if len(result) <= max_width: + return result + + if max_width <= 0: + return "" + + if max_width <= 3: + return result[-max_width:] + + return f"…{result[-(max_width - 1):]}" + + def _prepare_file(self) -> None: + """Apply explicit deletion, retention, and bounded truncation policies.""" + if self.settings.delete_previous and self.path.exists(): + self.path.unlink() + cutoff = time() - self.settings.retention_days * 86400 + for candidate in self.path.parent.glob("Logicytics*.log"): + if candidate != self.path and candidate.is_file() and candidate.stat().st_mtime < cutoff: + candidate.unlink() + self._truncate_file() + + def _truncate_file(self) -> None: + """Retain the newest complete rows when the configured byte limit is exceeded.""" + if self.path.is_file() and self.path.stat().st_size > self.settings.maximum_bytes: + with self.path.open("rb") as stream: + stream.seek(-self.settings.maximum_bytes, 2) + stream.readline() + retained = stream.read() + record = _RECORD_START.search(retained) + retained = retained[record.start():] if record is not None else b"" + self.path.write_bytes(retained) + + @staticmethod + def _rows(level: str, source: str, message: str) -> tuple[str, ...]: + """Format AIBrain-style fixed columns and aligned wrapped rows.""" + timestamp = datetime.now().astimezone().strftime("%Y-%m-%d %H:%M:%S.%f")[:-3] + source_column = source + prefix = f"{timestamp:<{_TIME_WIDTH}} | {level:<{_SEVERITY_WIDTH}} | {source_column:<{_SOURCE_WIDTH}} | " + continuation = f"{'':<{_TIME_WIDTH}} | {'':<{_SEVERITY_WIDTH}} | {'':<{_SOURCE_WIDTH}} | " + available = max(_FILE_LOG_LINE_WIDTH - len(prefix), 1) + wrapped = [ + segment + for line in message.splitlines() or [""] + for segment in ( + textwrap.wrap( + line, + width=available, + break_long_words=True, + break_on_hyphens=False, + ) + or [""] + ) + ] + first = prefix + wrapped[0] + return tuple([first, *(continuation + row for row in wrapped[1:])]) + + @staticmethod + def _console_width() -> int: + """Return the AIBrain console width with its safety margin and minimum.""" + return presentation_console_width() + + def _supports_unicode(self) -> bool: + """Return whether this console can encode AIBrain's presentation glyphs.""" + encoding = getattr(self.console, "encoding", None) or "utf-8" + try: + "\u00b7\u25cf\u00d7\u256d\u2500\u256e\u2502\u251c\u2524\u2570\u256f".encode(encoding) + except (LookupError, UnicodeEncodeError): + return False + return True + + @classmethod + def _console_rows(cls, marker: str, message: str) -> tuple[str, ...]: + """Word-wrap compact status text with aligned continuation indentation.""" + prefix = f" {marker} " + continuation = " " * len(prefix) + width = cls._console_width() + rows: list[str] = [] + for index, raw_line in enumerate(message.expandtabs(4).splitlines() or [""]): + indentation = raw_line[: len(raw_line) - len(raw_line.lstrip())] + remaining = raw_line.lstrip().rstrip() + current_prefix = (prefix if index == 0 else continuation) + indentation + while len(remaining) > max(width - len(current_prefix), 1): + available = max(width - len(current_prefix), 1) + split_at = remaining.rfind(" ", 0, available + 1) + if split_at <= 0: + split_at = available + rows.append(current_prefix + remaining[:split_at].rstrip()) + remaining = remaining[split_at:].lstrip() + current_prefix = continuation + indentation + rows.append(current_prefix + remaining) + return tuple(rows) + + @staticmethod + def _color_console_text(row: str, text_color: str) -> str: + """Color a leading detail marker purple while retaining the event color.""" + indentation_length = len(row) - len(row.lstrip()) + detail = row[indentation_length:] + if detail.startswith("> "): + return f"{text_color}{_BOLD}{row[:indentation_length]}{_DETAIL_MARKER_COLOR}{_BOLD}>{_RESET}{text_color}{_BOLD}{detail[1:]}" + return f"{text_color}{_BOLD}{row}" + + @staticmethod + def _console_message(message: str) -> str: + """Make machine-oriented lifecycle event names readable on the console.""" + if _EVENT_NAME.fullmatch(message): + return message.replace("_", " ").capitalize() + return message + + @classmethod + def _message_lines(cls, message: str) -> tuple[str, ...]: + """Render ordinary or JSON-looking messages as readable presentation lines.""" + safe_message = redact_text(message) + stripped = safe_message.strip() + if stripped.startswith(("{", "[")) and stripped.endswith(("}", "]")): + try: + parsed = json.loads(stripped) + except json.JSONDecodeError: + parsed = None + if isinstance(parsed, Mapping): + rows = ["Structured details"] + rows.extend(cls._console_fields(parsed)) + return tuple(rows) + if isinstance(parsed, list): + rows = ["Structured details"] + rows.extend(f"Item {index}: {cls._console_value(item)}" for index, item in enumerate(parsed, start=1)) + return tuple(rows) + return (cls._console_message(safe_message),) + + @classmethod + def _console_fields( + cls, + fields: Mapping[str, object], + *, + compact_configuration_hash: bool = False, + ) -> tuple[str, ...]: + """Render structured fields as readable labels instead of JSON fragments.""" + rows: list[str] = [] + for key, value in fields.items(): + label = key.replace("_", " ").capitalize().replace("Github", "GitHub") + rendered_value = cls._console_value(value) + if compact_configuration_hash and key in {"configuration_hash", "fingerprint"}: + rendered_value = rendered_value[:7] + rendered = rendered_value.splitlines() or ["none"] + rows.append(f"> {label}: {rendered[0]}") + rows.extend(f" {line}" for line in rendered[1:]) + return tuple(rows) + + @staticmethod + def _lifecycle_step(message: str) -> str | None: + """Map lifecycle events to the console phase that owns their details.""" + if not _EVENT_NAME.fullmatch(message): + return None + if message.startswith("run_packag") or message.startswith("package_"): + return "Packaging" + prefix = message.partition("_")[0] + return { + "command": "Command", + "preflight": "Preflight", + "plan": "Planning", + "run": "Collection", + "collector": "Collector execution", + "package": "Packaging", + "update": "Update", + "development": "Development", + }.get(prefix) + + def _render_step(self, message: str) -> None: + """Insert a visual break when lifecycle output moves to a new phase.""" + step = self._lifecycle_step(message) + if step is None or step == self._last_console_step: + return + render_presentation_step_heading( + self.console, + step, + width=self._console_width, + ) + self._last_console_step = step + + @classmethod + def _console_value(cls, value: object) -> str: + """Normalize common structured values for compact, human-readable console output.""" + if value is None: + return "none" + if isinstance(value, bool): + return "yes" if value else "no" + if isinstance(value, Mapping): + if not value: + return "none" + return "; ".join( + f"{str(key).replace('_', ' ')}: {cls._console_value(item)}" + for key, item in sorted(value.items(), key=lambda item: str(item[0])) + ) + if isinstance(value, (set, frozenset)): + if not value: + return "none" + rendered_items = sorted(cls._console_value(item) for item in value) + return ", ".join(rendered_items) + if isinstance(value, (list, tuple)): + if not value: + return "none" + return ", ".join(cls._console_value(item) for item in value) + if isinstance(value, (bytes, bytearray, memoryview)): + return f"<{len(value)} bytes>" + if isinstance(value, float): + if not math.isfinite(value): + return "non-finite float" + return f"{value:.6f}".rstrip("0").rstrip(".") + if isinstance(value, str): + stripped = value.strip() + if stripped.startswith(("{", "[")) and stripped.endswith(("}", "]")): + try: + parsed = json.loads(stripped) + except json.JSONDecodeError: + parsed = None + if parsed is not None: + return cls._console_value(parsed) + return str(value) + + def event(self, level: str, message: str, *, console: bool = True, **fields: float | str) -> None: + """Dispatch one typed, redacted event to configured console and file sinks.""" + normalized = _normalize_level(level) + minimum_level = _normalize_level(self.settings.level) + if _LEVEL_ORDER[normalized] < _LEVEL_ORDER[minimum_level]: + return + safe_message = redact_text(message) + safe_fields = redact_mapping(fields) + caller_module, caller_line = _caller_location() + source = str(safe_fields.pop("source", caller_module)) + if source in {"cli", "logicytics.cli", "runtime", "logicytics.runtime"}: + source = caller_module + source = _source_with_line(source, caller_line, debug=minimum_level == "DEBUG") + file_lines = list(self._message_lines(safe_message)) + file_lines.extend(self._console_fields(safe_fields)) + rendered_message = "\n".join(file_lines) + console_lines = list(self._message_lines(safe_message)) + console_lines.extend( + self._console_fields( + safe_fields, + compact_configuration_hash=minimum_level != "DEBUG", + ) + ) + console_message = "\n".join(console_lines) + rows = self._rows(normalized, source, rendered_message) + with self._lock: + if self.settings.file_enabled: + with self.path.open("a", encoding="utf-8") as stream: + stream.write("\n".join(rows) + "\n") + self._truncate_file() + if self.settings.console_enabled and console: + self._render_step(message) + marker, marker_color, text_color = _LEVEL_PRESENTATION[normalized] + + if not self._supports_unicode(): + marker = {"●": "*", "×": "X", "·": "."}.get(marker, marker) + + console_rows = self._console_rows(marker, console_message) + + if self.settings.color_enabled and self.console.isatty(): + marker_prefix = f" {marker} " + first_row = self._color_console_text( + console_rows[0][len(marker_prefix):], + text_color, + ) + colored_rows = ( + f"{marker_color}{_BOLD}" + f"{marker_prefix}{_RESET}" + f"{first_row}" + ) + + if len(console_rows) > 1: + colored_rows += "\n" + "\n".join( + self._color_console_text(row, text_color) + for row in console_rows[1:] + ) + + self.console.write(f"{colored_rows}{_RESET}\n") + else: + self.console.write("\n".join(console_rows) + "\n") + + self.console.flush() + self._remember_console_line(console_rows[-1]) + + def incomplete_progress( + self, + label: str, + checked: int | None = None, + total: int | None = None, + status: str = "Incomplete", + ) -> None: + """Replace the active progress bar with an incomplete x/y status.""" + if not self.settings.console_enabled or not self.console.isatty(): + return + + bounded_total = max( + total if total is not None else self._progress_total, + 1, + ) + bounded_checked = min( + max( + checked if checked is not None else self._progress_checked, + 0, + ), + bounded_total, + ) + + rendered = ( + f"{self._progress_indent}" + f"{label} [{status}] " + f"{bounded_checked}/{bounded_total}" + ) + + with self._lock: + if self.settings.color_enabled: + rendered = f"\033[91m\033[1m{rendered}\033[0m" + + self.console.write(f"\r\033[2K{rendered}\n") + self.console.flush() + + self._remember_console_line(rendered) + + self._progress_indent = "" + self._progress_checked = 0 + self._progress_total = 0 + self._progress_current = "" + + def progress( + self, + label: str, + checked: int, + total: int, + current: str = "", + ) -> None: + """Render one full-width adaptive in-place progress bar.""" + if not self.settings.console_enabled or not self.console.isatty(): + return + + bounded_total = max(total, 1) + bounded_checked = min(max(checked, 0), bounded_total) + + if checked == 0 or not self._progress_indent: + self._progress_indent = self._leading_indent( + self._last_console_line + ) + + self._progress_checked = bounded_checked + self._progress_total = bounded_total + self._progress_current = current + + console_width = self._console_width() + count = f"{bounded_checked}/{bounded_total}" + + if current: + display_current = self._shorten_current( + current, + max_width=console_width, + ) + else: + display_current = "" + + suffix = f" {display_current}" if display_current else "" + + # Layout: + # + #