diff --git a/README.md b/README.md index ce4307d..7140468 100644 --- a/README.md +++ b/README.md @@ -53,7 +53,7 @@ wm draft.md -o draft.cleaned.md --json --audit The install pulls no dependencies — the core is standard library only. Extras are opt-in: `watermark-remover[visible]` for image inpainting, `[quality]` for scoring, `[ai]` for the torch-backed adapters, `[provenance]` -for C2PA, `[tui]` for the terminal UI, or `[all]`. Five commands are installed: +for C2PA, or `[all]`. Five commands are installed: `wm`, `wm-tui`, `wm-serve`, `wm-audit-dir`, `wm-audit-site`. ```bash @@ -61,10 +61,6 @@ for C2PA, `[tui]` for the terminal UI, or `[all]`. Five commands are installed: wm draft.md -o draft.cleaned.md \ --rewrite humanize --rewrite-backend openai-compatible \ --rewrite-base-url http://127.0.0.1:8000 --rewrite-model my-local-model - -# Or drive the whole loop interactively -pip install "watermark-remover[tui]" -wm-tui ./drafts --recursive ``` ### From a clone @@ -116,62 +112,66 @@ python3 "$SCRIPTS/inspect_file.py" ./inputs --recursive --glob "*.md" --json ### Terminal UI ```bash -pip install "watermark-remover[tui]" # adds textual; the core stays dependency-free -wm-tui # opens the current directory +wm-tui # the current directory wm-tui ./drafts --recursive --glob "*.md" ``` -`wm-tui` is a front end over the same seam the CLI uses: it fills a -`CleanRequest`, runs it through `clean_request.plan_work` and -`clean_file.run_clean_item`, and inherits every refusal the CLI makes. It never -speaks HTTP itself and never displays or persists an API key. +`wm-tui` is a single-screen terminal UI in the style of opencode and pi. It is +built on [OpenTUI](https://github.com/anomalyco/opentui), so it needs +[Bun](https://bun.sh) 1.4 or later on the `PATH`. The Python install stays +dependency-free: on first launch `wm-tui` runs `bun install` for its own +frontend, and it prints the install hint if Bun is missing. macOS, Linux and +Windows are supported. + +The screen has four parts: + +- **The file list** on the left, which also shows progress and results. +- **The selected file** on the right: what was found, what was removed, and a + before/after diff. +- **One prompt** at the bottom: + - a path adds files; + - a `--flag` adds a `wm` option (any flag `wm` accepts, checked by the + CLI's own parser; type a bare flag again to turn it off); + - a `/command` runs a command. +- **The footer**, which always shows the exact `wm …` command a clean would + run. + +| Key | Does | +| --- | --- | +| `ctrl+e` | inspect, which writes nothing | +| `ctrl+r` | clean; each file gets a `NAME.cleaned.EXT` beside it and originals stay untouched | +| `tab` | next preset: Hidden marks, Hidden marks aggressive, Deep clean (LLM rewrite), Images | +| `ctrl+p` | every command: model, history, doctor, setup, help | +| `esc` | stop after the current file | -It opens on **Start**, which is the whole job in three steps: add a file or a -folder, choose a preset, press Clean. Nothing has to be configured first, and -nothing about a preset is hidden — every option it sets is a visible control on -the Plan tab and appears in the equivalent `wm …` command. +**First run** opens a three-step setup: -| Preset | What it turns on | Result class | -| --- | --- | --- | -| Hidden marks | zero-width carriers, bidi controls, AI metadata — identical to a bare `wm FILE` | Verifiable | -| Hidden marks, aggressive | adds NFKC normalisation and homoglyph folding | Verifiable | -| Deep clean (LLM rewrite) | adds a local-model paraphrase; needs a Layer B endpoint | Best-effort | -| Images: metadata + degrade | strips C2PA/AI metadata, then perturbs the frequency domain | Best-effort | - -No preset can set `--in-place`, `--strip-semantic-format` or `--dry-run`: those -overwrite the input, change what the text means, or replace the run with a -description, and each is a deliberate choice with its own confirmation. - -The rest of Start is setup. The **Layer B endpoint** block sets the backend, -base URL and model and probes them; **Save setup** writes them to -`~/.config/watermark-remover/tui.json` (`$XDG_CONFIG_HOME` or `%APPDATA%` when -set, or `WATERMARKS_TUI_SETTINGS` to point somewhere else) so the next run -starts configured. The API key is never in that file — it is read from -`WATERMARKS_REWRITE_API_KEY` at run time and has no field to be written to. -**Installed capabilities** lists every optional extra and hands you the exact -`pip install` line for the missing ones. - -The other panes: **Files** (select, rescan, glob and extension filters), -**Inspect** (Layer A carriers, metadata, stylometry, soft binding), **Plan** -(every option, plus the equivalent `wm …` command), **Run** (sequential batch, -per-file result table, live Layer B token stream, before/after diff), -**History** (every command this session generated, copyable and re-loadable). - -The command-line arguments are `path`, `--recursive`, `--glob`, and -`--extensions` — the initial file selection; more paths can be added from -Start once it is running. Everything else is configured in the UI, and the -exact `wm` command it corresponds to is shown and copyable so a run can be -reproduced outside it. - -Results are labeled by *layer*, never by outcome: Layer A and Layer M are -Verifiable, Layer B and Layer V are Best-effort, soft binding is -Detection-only. A run that stops to confirm — remote egress, `--in-place`, -`--strip-semantic-format`, or an expensive batch — does so before the first -write, not after. - -Clipboard copy uses OSC 52, which some terminals (including macOS Terminal.app) -ignore without acknowledging. Every copy button is therefore paired with a -read-only, selectable text box holding the same string. +1. What the result classes mean. +2. A scan for a local model server. The scan only contacts `127.0.0.1`: it + checks Ollama, LM Studio, llama.cpp and vLLM for their model lists, and + never sends a document. +3. The default preset. + +`esc` skips the setup, and it does not come back unless you run `/setup` or +`wm-tui --setup`. `--no-setup` never shows it. + +The choices are saved to `~/.config/watermark-remover/tui.json`. The API key is +never in that file: it is read from `WATERMARKS_REWRITE_API_KEY` at run time +and never displayed. + +Results are labelled by layer, never by outcome: Verifiable, Best-effort or +Detection-only. A clean that needs confirmation stops and asks before the +first write. That covers a remote endpoint, `--in-place`, +`--strip-semantic-format`, and a long LLM batch. + +The UI is two processes: + +- A Bun frontend (`skills/remove-ai-marks/tui/`) that only renders. +- A Python bridge (`scripts/tui_bridge.py`) that owns every decision. It builds + the request through `wm`'s own parser, `plan_work` and `run_clean_item`, so + it hits every refusal the CLI makes. + +`tui/PROTOCOL.md` documents the protocol between them. --- diff --git a/pyproject.toml b/pyproject.toml index 657d9bc..36dc8dc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -57,19 +57,11 @@ ai = [ provenance = [ "c2pa-python", ] -# Interactive terminal UI (wm-tui). Textual brings rich; both are pure Python. -# Deliberately not in the core: `pip install watermark-remover` still pulls -# nothing. stdlib curses was rejected — it is absent on Windows, which this -# project supports, so "no dependency" would have cost windows-curses anyway. -tui = [ - "textual>=0.80", -] all = [ "watermark-remover[visible]", "watermark-remover[quality]", "watermark-remover[ai]", "watermark-remover[provenance]", - "watermark-remover[tui]", ] [project.scripts] @@ -89,7 +81,18 @@ packages = ["watermark_remover", "watermark_remover.scripts"] [tool.setuptools.package-data] # Ship the skill documentation and bootstrap assets so a wheel install is # self-sufficient for the CLI surfaces that read them at runtime. -"watermark_remover" = ["SKILL.md", "references/*.md"] +# The wm-tui frontend ships as source; the launcher runs `bun install` on +# first start (into a cache copy when site-packages is read-only). +"watermark_remover" = [ + "SKILL.md", + "references/*.md", + "tui/package.json", + "tui/bun.lock", + "tui/bunfig.toml", + "tui/tsconfig.json", + "tui/src/**/*", + "tui/PROTOCOL.md", +] "watermark_remover.scripts" = ["*.txt", "setup_*.sh", "setup_*.ps1"] [tool.ruff] diff --git a/requirements-test.txt b/requirements-test.txt index d2d210e..075297e 100644 --- a/requirements-test.txt +++ b/requirements-test.txt @@ -4,5 +4,3 @@ pypdf>=6.16.2,<7 ruff>=0.16.6,<1 # OpenAPI contract validation for wm-serve (CI validates /openapi.json). openapi-spec-validator==0.9.0 -# Interactive TUI (wm-tui) — installed so CI actually exercises the tui tests. -textual>=0.80 diff --git a/skills/clean-user-facing-text/scripts/common.py b/skills/clean-user-facing-text/scripts/common.py index 5455406..c9bd581 100644 --- a/skills/clean-user-facing-text/scripts/common.py +++ b/skills/clean-user-facing-text/scripts/common.py @@ -44,10 +44,8 @@ def _configure_stdio() -> None: ): reconfigure = getattr(stream, "reconfigure", None) if reconfigure is not None: - try: + with suppress(OSError, ValueError): reconfigure(encoding="utf-8", errors=errors) - except (OSError, ValueError): - pass _configure_stdio() diff --git a/skills/clean-user-facing-text/scripts/inspect_text.py b/skills/clean-user-facing-text/scripts/inspect_text.py index bf8a71e..0138a4b 100755 --- a/skills/clean-user-facing-text/scripts/inspect_text.py +++ b/skills/clean-user-facing-text/scripts/inspect_text.py @@ -10,8 +10,8 @@ # Allow running as script from any cwd sys.path.insert(0, str(Path(__file__).resolve().parent)) -from common import emit_json, read_text_input # noqa: E402 -from text_unicode import human_report, inspect_text # noqa: E402 +from common import emit_json, read_text_input +from text_unicode import human_report, inspect_text def main() -> int: diff --git a/skills/remove-ai-marks/SKILL.md b/skills/remove-ai-marks/SKILL.md index f37336c..7ded230 100644 --- a/skills/remove-ai-marks/SKILL.md +++ b/skills/remove-ai-marks/SKILL.md @@ -183,16 +183,15 @@ Always include: ## Interactive surface -`wm-tui` (install `watermark-remover[tui]`) drives this same workflow with the -before/after evidence on screen: Layer A carrier counts, stylometry, and the -Layer B token stream side by side, plus the equivalent `wm …` command for every -run so the result is reproducible outside the UI. Its Start pane is the whole -job in three steps — add files, choose a preset, clean — and each preset names -its result class at the point of choice rather than only in the results. It fills a `CleanRequest` and -goes through `clean_request.plan_work` / `clean_file.run_clean_item`, so every -refusal documented here applies there unchanged. It never displays or persists -an API key, and it labels results by layer — Verifiable, Best-effort, -Detection-only — never by outcome. +`wm-tui` (it needs Bun) drives this same workflow from one screen: +- a file list that doubles as progress and results; +- the selected file's findings and before/after diff; +- a single prompt that takes paths, any `wm` flag, or `/commands`. + +Its Python bridge builds every request through `clean_file`'s own parser and +runs it through `plan_work` and `run_clean_item`, so every refusal documented +here applies unchanged. It never displays or persists an API key. It labels +results by layer (Verifiable, Best-effort, Detection-only), never by outcome. ## Hard limits diff --git a/skills/remove-ai-marks/scripts/clean_asset.py b/skills/remove-ai-marks/scripts/clean_asset.py index 5e2f8ea..40d788f 100644 --- a/skills/remove-ai-marks/scripts/clean_asset.py +++ b/skills/remove-ai-marks/scripts/clean_asset.py @@ -36,6 +36,7 @@ ) from perturb_text import MODES as PERTURB_MODES from perturb_text import perturb_text +from pipeline_actions import retarget_report from rewrite_text import RewritePlan, TokenSink, rewrite from text_unicode import clean_text @@ -570,10 +571,7 @@ def _clean_image_asset(path: Path, dest: Path, plan: CleanPlan) -> CleanResult: staged_mask_text = str(staged_mask) final_mask_text = str(final_mask_output) visible_report["mask"] = final_mask_text - visible_report["actions"] = [ - action.replace(staged_mask_text, final_mask_text) - for action in visible_report["actions"] - ] + retarget_report(visible_report, staged_mask_text, final_mask_text) report["input"] = str(path) if visible_report is not None: diff --git a/skills/remove-ai-marks/scripts/clean_file.py b/skills/remove-ai-marks/scripts/clean_file.py index ce2467c..f804999 100644 --- a/skills/remove-ai-marks/scripts/clean_file.py +++ b/skills/remove-ai-marks/scripts/clean_file.py @@ -21,8 +21,7 @@ from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) -from asset_kind import SUPPORTED_EXTENSIONS -from batch_inputs import select_inputs +import pipeline_actions as act from clean_asset import ( DEGRADE_CLI_CHOICES, MORPHO_CLI_CHOICES, @@ -39,6 +38,7 @@ describe_dropped_text_transforms, dropped_text_transforms, plan_work, + select_request_inputs, ) from common import ( atomic_write_text, @@ -48,6 +48,7 @@ from morphomod import VISIBLE_CLEAN_BACKENDS from operation import ExitCode, OperationStatus, status_to_exit_code from perturb_text import MODES as PERTURB_MODES +from pipeline_actions import report_actions from rewrite_text import LIVE_REWRITE_BACKENDS, REASONING_EFFORTS, TokenSink, remote_warning # Preserved under the old private name: external callers are not expected, but @@ -246,32 +247,13 @@ def main() -> int: except ValueError as error: eprint(f"invalid options: {error}") return ExitCode.USAGE_ERROR.value - allowed = request.allowed_extensions(SUPPORTED_EXTENSIONS) - excluded_roots = ( - (request.output,) - if request.output - and not request.in_place - and any(source.is_dir() for source in request.paths) - else () - ) try: - selection = select_inputs( - request.paths, - recursive=request.recursive, - pattern=request.glob, - extensions=allowed, - excluded_roots=excluded_roots, - ) + selection = select_request_inputs(request) except ValueError as error: eprint(f"invalid input selection: {error}") return ExitCode.USAGE_ERROR.value items = selection.items batch = selection.batch - if batch and (request.visible_mask or request.visible_box): - eprint( - "error: --visible-mask/--visible-box are single-file options; use --detect-command for batch" - ) - return ExitCode.USAGE_ERROR.value if request.in_place and request.output: eprint("warning: -o ignored with --in-place") try: @@ -350,14 +332,14 @@ def dry_run_payload( else: localization = "external-detector" actions = [ - f"localize visible mark via {localization}", - f"fill holes + dilate radius={visible.dilation_radius}", - f"inpaint with {visible.backend} backend", - "strip requested metadata", + act.plan_localize(localization), + act.plan_refine_mask(visible.dilation_radius), + act.plan_inpaint(visible.backend), + act.plan_strip_metadata(), ] if plan.degrade is not None: - actions.append(f"apply {plan.degrade.strategy} degradation") - actions.append(f"publish mask to {visible.mask_output} and image to {destination}") + actions.append(act.plan_degrade(plan.degrade.strategy)) + actions.append(act.plan_publish(str(visible.mask_output), str(destination))) return { "kind": "image", "status": "dry-run", @@ -366,7 +348,7 @@ def dry_run_payload( "mask": str(visible.mask_output), "backend": visible.backend, "timeout": visible.timeout, - "actions": actions, + **report_actions(actions), "exit_code": ExitCode.SUCCESS.value, } @@ -405,7 +387,7 @@ def _error_payload(path: Path, output: Path, error: Exception) -> dict: "kind": "unknown", "input": str(path), "output": str(output), - "actions": [f"error: {error}"], + **report_actions([act.failed(error)]), "error": str(error), "exit_code": status_to_exit_code(OperationStatus.FAILED), } @@ -478,10 +460,5 @@ def run_clean_item( return payload -#: Preserved private aliases for in-tree callers predating the renames. -_run_clean_item = run_clean_item -_dry_run_payload = dry_run_payload - - if __name__ == "__main__": raise SystemExit(main()) diff --git a/skills/remove-ai-marks/scripts/clean_request.py b/skills/remove-ai-marks/scripts/clean_request.py index b20a59a..d8daf6a 100644 --- a/skills/remove-ai-marks/scripts/clean_request.py +++ b/skills/remove-ai-marks/scripts/clean_request.py @@ -32,8 +32,8 @@ from dataclasses import dataclass, field, replace from pathlib import Path -from asset_kind import AssetKind, classify_asset -from batch_inputs import InputItem, safe_output_path +from asset_kind import SUPPORTED_EXTENSIONS, AssetKind, classify_asset +from batch_inputs import InputItem, InputSelection, safe_output_path, select_inputs from clean_asset import ( CleanPlan, ImageDegradePlan, @@ -42,10 +42,10 @@ from common import ( MAX_INPUT_BYTES, ROUTER_ADVICE, + alias_key, backup_path, cleaned_path, guard_binary, - paths_alias, validate_output_path, ) from morphomod import DEFAULT_DILATION_RADIUS, VisiblePlan @@ -566,6 +566,34 @@ def resolve_kind(path: Path, request: CleanRequest) -> AssetKind: return kind +def select_request_inputs(request: CleanRequest) -> InputSelection: + """The files a clean of *request* touches, and whether it is a batch. + + A batch output directory that sits under an input is excluded, so a + rerun never cleans its own outputs. Raises ``ValueError`` with the + selector's message, or when a single-file option meets a batch. + """ + excluded_roots = ( + (request.output,) + if request.output + and not request.in_place + and any(source.is_dir() for source in request.paths) + else () + ) + selection = select_inputs( + request.paths, + recursive=request.recursive, + pattern=request.glob, + extensions=request.allowed_extensions(SUPPORTED_EXTENSIONS), + excluded_roots=excluded_roots, + ) + if selection.batch and (request.visible_mask or request.visible_box): + raise ValueError( + "--visible-mask/--visible-box are single-file options; use --detect-command for batch" + ) + return selection + + def plan_work( items: Sequence[InputItem], request: CleanRequest, @@ -577,8 +605,8 @@ def plan_work( for ancillary in ancillary_inputs: if not ancillary.is_file() or ancillary.is_symlink(): raise ValueError(f"not a regular mask file: {ancillary}") - all_inputs = [*inputs, *ancillary_inputs] - destinations: list[Path] = [] + input_keys = {alias_key(path) for path in (*inputs, *ancillary_inputs)} + destination_keys: set[tuple[object, ...]] = set() work: list[tuple[InputItem, Path | None, CleanPlan]] = [] for item in items: @@ -597,13 +625,12 @@ def plan_work( output = request.output validate_output_path(item.path, output) - for source in all_inputs: - if paths_alias(output, source): - raise ValueError(f"output aliases an input: {output}") - for existing in destinations: - if paths_alias(output, existing): - raise ValueError(f"batch output collision: {output}") - destinations.append(output) + key = alias_key(output) + if key in input_keys: + raise ValueError(f"output aliases an input: {output}") + if key in destination_keys: + raise ValueError(f"batch output collision: {output}") + destination_keys.add(key) dest = output if item.path.stat().st_size > MAX_INPUT_BYTES: @@ -620,11 +647,12 @@ def plan_work( raise ValueError("visible plan is missing a mask output path") if mask_output.is_symlink(): raise ValueError(f"mask output is a symlink: {mask_output}") - if any(paths_alias(mask_output, source) for source in all_inputs): + mask_key = alias_key(mask_output) + if mask_key in input_keys: raise ValueError(f"mask output aliases an input: {mask_output}") - if any(paths_alias(mask_output, existing) for existing in destinations): + if mask_key in destination_keys: raise ValueError(f"mask/output collision: {mask_output}") - destinations.append(mask_output) + destination_keys.add(mask_key) work.append((item, output, plan)) return work diff --git a/skills/remove-ai-marks/scripts/common.py b/skills/remove-ai-marks/scripts/common.py index 98f88b2..7fc683a 100755 --- a/skills/remove-ai-marks/scripts/common.py +++ b/skills/remove-ai-marks/scripts/common.py @@ -271,6 +271,22 @@ def paths_alias(left: Path, right: Path) -> bool: return left.resolve(strict=False) == right.resolve(strict=False) +def alias_key(path: Path) -> tuple[object, ...]: + """A hashable identity with the same equality as :func:`paths_alias`. + + An existing file is its ``(st_dev, st_ino)``, so hard links match; a path + that does not exist yet is its resolved form. An existing and a missing + path never alias, which is also what ``paths_alias`` answers for them. + Checking N paths against each other through a set of keys is linear, + where pairwise ``paths_alias`` calls are quadratic. + """ + try: + info = path.stat() + except FileNotFoundError: + return ("path", path.resolve(strict=False)) + return ("inode", info.st_dev, info.st_ino) + + def validate_output_path(source: Path, dest: Path) -> None: """Reject implicit in-place writes and destination symlinks.""" if dest.is_symlink(): @@ -430,8 +446,8 @@ def classify_finding_confidence(finding: str) -> str: parsed field such as digitalSourceType / trainedAlgorithmicMedia). - probable: an AI/vendor marker found inside a recognized metadata structure, but not a fully parsed provenance claim. - - informational: context-only notes (CMS generators, presence of an XMP - packet or customXml parts, unsupported/partial inspection). + - informational: problems with the scan itself (unsupported or malformed + input) and markers that alone say nothing about AI provenance. - likely_false_positive: raw whole-file byte scans that can collide with compressed image/stream data. @@ -459,18 +475,12 @@ def classify_finding_confidence(finding: str) -> str: ): return "confirmed" - if t.startswith("info:") or any( + if any( s in t for s in ( - "cms generator", - "customxml parts", - "xmp packet present", "unsupported", - "not fully inspected", - "format not", "svg present", "not a valid", - "truncated chunk", "bad segment length", "svg decode note", ) diff --git a/skills/remove-ai-marks/scripts/configuration.py b/skills/remove-ai-marks/scripts/configuration.py index 0849a30..792654a 100644 --- a/skills/remove-ai-marks/scripts/configuration.py +++ b/skills/remove-ai-marks/scripts/configuration.py @@ -51,6 +51,8 @@ from pathlib import Path from typing import Any +from rewrite_text import DEFAULT_BASE_URL + class ConfigSource(Enum): """Where a setting value came from.""" @@ -188,7 +190,7 @@ def _read_env_file(path: Path) -> dict[str, str]: ), ( "rewrite_base_url", - "http://127.0.0.1:11434", + DEFAULT_BASE_URL, "Layer B backend base URL", ConfigSource.DEFAULT, ), diff --git a/skills/remove-ai-marks/scripts/container_meta.py b/skills/remove-ai-marks/scripts/container_meta.py index c3b00f8..a230075 100644 --- a/skills/remove-ai-marks/scripts/container_meta.py +++ b/skills/remove-ai-marks/scripts/container_meta.py @@ -16,10 +16,13 @@ from typing import Any import external_command +import pipeline_actions as act from common import atomic_write_bytes, atomic_write_text, classify_finding_confidence, which from image_meta import ( AI_META_HINTS, C2PA_MARKERS, + MetadataScan, + _contains_any, detect_format, inspect_avif, inspect_bmp, @@ -39,9 +42,16 @@ strip_tiff, strip_webp, ) +from pipeline_actions import Action, any_change, report_actions run_command = external_command.run_command +#: One container scanner's verdict: ``(has_c2pa, has_ai_metadata, findings, +#: notes, details)``. As with ``image_meta.MetadataScan``, findings are the +#: AI/provenance signals (and scan problems); notes are context that is +#: neither, and are never counted as marks. +ContainerScan = tuple[bool, bool, list[str], list[str], dict[str, Any]] + # Frontmatter / meta keys that often carry AI provenance AI_FRONTMATTER_KEYS = frozenset( { @@ -168,21 +178,12 @@ def detect_container_format(path: Path, data: bytes | None = None) -> str: def _blob_hits(blob: bytes) -> tuple[bool, bool, list[str]]: - lower = blob.lower() - findings: list[str] = [] - has_c2pa = False - has_ai = False - for n in C2PA_MARKERS: - if n.lower() in lower: - has_c2pa = True - findings.append(f"marker:{n.decode('ascii', errors='replace')}") - for n in AI_META_HINTS: - if n.lower() in lower: - has_ai = True - label = n.decode("ascii", errors="replace") - if label not in {f.split(":", 1)[-1] for f in findings}: - findings.append(f"ai:{label}") - return has_c2pa, has_ai or has_c2pa, findings[:30] + c2pa_hits = _contains_any(blob, C2PA_MARKERS) + seen = {h.lower() for h in c2pa_hits} + ai_hits = [h for h in _contains_any(blob, AI_META_HINTS) if h.lower() not in seen] + findings = [f"marker:{h}" for h in c2pa_hits] + [f"ai:{h}" for h in ai_hits] + has_c2pa = bool(c2pa_hits) + return has_c2pa, has_c2pa or bool(ai_hits), findings[:30] # --------------------------------------------------------------------------- @@ -198,23 +199,22 @@ def _blob_hits(blob: bytes) -> tuple[bool, bool, list[str]]: ) -def _media_strip_succeeded(sub_actions: list[str], cleaned: bytes, raw: bytes) -> bool: +def _media_strip_succeeded(sub_actions: list[Action], cleaned: bytes, raw: bytes) -> bool: """True when a media stripper changed bytes while reporting a removal. - Most raster strippers "drop" chunks/segments; heif_meta neutralizes in - place ("neutralized"/"zeroed") to preserve offsets, so accept both - vocabularies. The no-op case always returns the input bytes unchanged. + A step's ``effect`` says whether it changed the output, whatever its + wording: raster strippers drop chunks, heif_meta zeroes in place, and an + SVG may only have cleaned a nested data URI. The no-op case always + returns the input bytes unchanged. """ - if cleaned == raw: - return False - verbs = ("drop", "neutraliz", "zero") - return any(verb in action.lower() for action in sub_actions for verb in verbs) + return cleaned != raw and any_change(sub_actions) -def _inspect_embedded_data_uris(text: str) -> tuple[bool, bool, list[str]]: +def _inspect_embedded_data_uris(text: str) -> MetadataScan: has_c2pa = False has_ai = False findings: list[str] = [] + notes: list[str] = [] import base64 import urllib.parse @@ -241,18 +241,19 @@ def _inspect_embedded_data_uris(text: str) -> tuple[bool, bool, list[str]]: continue fmt = detect_format(data) + sub_notes: list[str] = [] if fmt == "png": - sub_c2pa, sub_ai, sub_findings = inspect_png(data) + sub_c2pa, sub_ai, sub_findings, sub_notes = inspect_png(data) elif fmt == "jpeg": - sub_c2pa, sub_ai, sub_findings = inspect_jpeg(data) + sub_c2pa, sub_ai, sub_findings, sub_notes = inspect_jpeg(data) elif fmt == "webp": - sub_c2pa, sub_ai, sub_findings = inspect_webp(data) + sub_c2pa, sub_ai, sub_findings, sub_notes = inspect_webp(data) elif fmt == "avif": - sub_c2pa, sub_ai, sub_findings = inspect_avif(data) + sub_c2pa, sub_ai, sub_findings, sub_notes = inspect_avif(data) elif fmt == "heif": - sub_c2pa, sub_ai, sub_findings = inspect_heic(data) + sub_c2pa, sub_ai, sub_findings, sub_notes = inspect_heic(data) elif "svg" in mime or data.lstrip().startswith(b"<"): - sub_c2pa, sub_ai, sub_findings, _ = inspect_svg(data) + sub_c2pa, sub_ai, sub_findings, sub_notes, _ = inspect_svg(data) else: sub_c2pa, sub_ai, sub_findings = _blob_hits(data) @@ -260,19 +261,19 @@ def _inspect_embedded_data_uris(text: str) -> tuple[bool, bool, list[str]]: has_c2pa = True if sub_ai or sub_c2pa: has_ai = True - for f in sub_findings: - findings.append(f"embedded data:image/{mime}: {f}") + findings.extend(f"embedded data:image/{mime}: {f}" for f in sub_findings) + notes.extend(f"embedded data:image/{mime}: {n}" for n in sub_notes) - return has_c2pa, has_ai, findings + return has_c2pa, has_ai, findings, notes def _clean_embedded_data_uris( text: str, *, strip_all_metadata: bool = True -) -> tuple[str, list[str]]: +) -> tuple[str, list[Action]]: import base64 import urllib.parse - actions: list[str] = [] + actions: list[Action] = [] def _replace_uri(m: re.Match[str]) -> str: full_match = m.group(0) @@ -297,7 +298,7 @@ def _replace_uri(m: re.Match[str]) -> str: return full_match fmt = detect_format(data) - sub_actions: list[str] = [] + sub_actions: list[Action] = [] cleaned_bytes = data try: @@ -319,7 +320,7 @@ def _replace_uri(m: re.Match[str]) -> str: if not _media_strip_succeeded(sub_actions, cleaned_bytes, data): return full_match - actions.append(f"cleaned embedded data:image/{mime} ({', '.join(sub_actions[:2])})") + actions.append(act.clean_data_uri(mime, sub_actions)) if is_b64: new_b64 = base64.b64encode(cleaned_bytes).decode("ascii") @@ -340,7 +341,7 @@ def _replace_uri(m: re.Match[str]) -> str: re.I, ) _META_ATTR_RE = re.compile( - r"""(name|property|content|generator)s*=s*["']([^"']*)["']""", + r"""(name|property|content|generator)\s*=\s*["']([^"']*)["']""", re.I, ) @@ -364,6 +365,7 @@ def _is_cms_generator_meta(tag: str) -> bool: # Markdown frontmatter # --------------------------------------------------------------------------- +_C2PA_KEY_RE = re.compile(r"c2pa|content.?credential", re.I) _FM_RE = re.compile(r"\A---\r?\n(.*?)\r?\n---\r?\n?", re.DOTALL) @@ -381,7 +383,7 @@ def _parse_simple_yaml_keys(block: str) -> list[tuple[str, str, int]]: return rows -def inspect_markdown(text: str) -> tuple[bool, bool, list[str], dict]: +def inspect_markdown(text: str) -> ContainerScan: findings: list[str] = [] has_ai = False has_fm = False @@ -390,30 +392,34 @@ def inspect_markdown(text: str) -> tuple[bool, bool, list[str], dict]: if m: has_fm = True block = m.group(1) - for key, _line, _i in _parse_simple_yaml_keys(block): + for key, line, _i in _parse_simple_yaml_keys(block): keys.append(key) + val = line.split(":", 1)[1] if key.lower() in AI_FRONTMATTER_KEYS or AI_META_NAME_RE.search(key): has_ai = True findings.append(f"frontmatter key: {key}") - # also check value - val = _line.split(":", 1)[1] if ":" in _line else "" - if AI_META_NAME_RE.search(val): + elif AI_META_NAME_RE.search(val): has_ai = True findings.append(f"frontmatter value hit on {key}") - uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text) + uri_c2pa, uri_ai, uri_findings, uri_notes = _inspect_embedded_data_uris(text) if uri_c2pa: has_ai = True if uri_ai: has_ai = True findings.extend(uri_findings) - c2pa = uri_c2pa or any("c2pa" in f.lower() or "content" in f.lower() for f in findings) - return c2pa, has_ai, findings, {"has_frontmatter": has_fm, "keys": keys} + c2pa = uri_c2pa or any(_C2PA_KEY_RE.search(f) for f in findings) + return c2pa, has_ai, findings, uri_notes, {"has_frontmatter": has_fm, "keys": keys} -def clean_markdown(text: str) -> tuple[str, list[str]]: - actions: list[str] = [] +_NOTHING_IN_MARKDOWN = act.nothing_removed( + "markdown", "no AI frontmatter keys or embedded data URIs removed" +) + + +def clean_markdown(text: str) -> tuple[str, list[Action]]: + actions: list[Action] = [] m = _FM_RE.match(text) if not m: # no frontmatter: still scrub embedded data URIs in the body @@ -421,7 +427,7 @@ def clean_markdown(text: str) -> tuple[str, list[str]]: if uri_actions: actions.extend(uri_actions) if not actions: - actions.append("no AI frontmatter keys or embedded data URIs removed") + actions.append(_NOTHING_IN_MARKDOWN) return out, actions block = m.group(1) body = text[m.end() :] @@ -448,30 +454,30 @@ def clean_markdown(text: str) -> tuple[str, list[str]]: key = km.group(1) val = line.split(":", 1)[1] if ":" in line else "" if key.lower() in AI_FRONTMATTER_KEYS or AI_META_NAME_RE.search(key): - actions.append(f"drop frontmatter key: {key}") + actions.append(act.drop_frontmatter_key(key)) dropping_parent = True continue if AI_META_NAME_RE.search(val): - actions.append(f"drop frontmatter key (value hit): {key}") + actions.append(act.drop_frontmatter_key(key, value_hit=True)) dropping_parent = True continue kept.append(line) if not actions: - actions.append("no AI frontmatter keys removed") + actions.append(act.nothing_removed("markdown", "no AI frontmatter keys removed")) # strip trailing empty nested orphans already handled new_block = "\n".join(kept).strip("\n") if new_block: out = f"---\n{new_block}\n---\n{body}" else: out = body.lstrip("\n") - actions.append("removed empty frontmatter block") + actions.append(act.drop_empty_frontmatter()) out, uri_actions = _clean_embedded_data_uris(out) if uri_actions: actions.extend(uri_actions) if not actions: - actions.append("no AI frontmatter keys or embedded data URIs removed") + actions.append(_NOTHING_IN_MARKDOWN) return out, actions @@ -489,15 +495,16 @@ def clean_markdown(text: str) -> tuple[str, list[str]]: ) -def inspect_html(text: str) -> tuple[bool, bool, list[str], dict]: +def inspect_html(text: str) -> ContainerScan: findings: list[str] = [] + notes: list[str] = [] has_ai = False has_c2pa = False for tag in _META_TAG_RE.findall(text): if re.search(r"c2pa|content.?credential", tag, re.I): has_c2pa = True if _is_cms_generator_meta(tag): - findings.append(f"info: cms generator: {tag[:120]}") + notes.append(f"info: cms generator: {tag[:120]}") continue if AI_META_NAME_RE.search(tag) or any( h.decode("ascii", "ignore").lower() in tag.lower() for h in AI_META_HINTS[:12] @@ -518,18 +525,19 @@ def inspect_html(text: str) -> tuple[bool, bool, list[str], dict]: has_ai = True findings.append(f"attr: {m.group(0)[:80]}") - uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text) + uri_c2pa, uri_ai, uri_findings, uri_notes = _inspect_embedded_data_uris(text) if uri_c2pa: has_c2pa = True if uri_ai: has_ai = True findings.extend(uri_findings) + notes.extend(uri_notes) - return has_c2pa, has_ai, findings, {} + return has_c2pa, has_ai, findings, notes, {} -def clean_html(text: str) -> tuple[str, list[str]]: - actions: list[str] = [] +def clean_html(text: str) -> tuple[str, list[Action]]: + actions: list[Action] = [] def _meta_sub(m: re.Match[str]) -> str: tag = m.group(0) @@ -538,7 +546,7 @@ def _meta_sub(m: re.Match[str]) -> str: if AI_META_NAME_RE.search(tag) or re.search( r"generator|claude|anthropic|openai|gemini|synthid|c2pa|aigc", tag, re.I ): - actions.append(f"drop meta: {tag[:80]}") + actions.append(act.drop_html_meta(tag)) return "" return tag @@ -549,19 +557,19 @@ def _jsonld_sub(m: re.Match[str]) -> str: if AI_META_NAME_RE.search(blob) or re.search( r"DigitalSourceType|trainedAlgorithmicMedia|SoftwareAgent", blob, re.I ): - actions.append("drop json-ld provenance-like script") + actions.append(act.drop_json_ld()) return "" return blob out = _JSONLD_RE.sub(_jsonld_sub, out) out2, n = re.subn(r"\sdata-ai[\w-]*\s*=\s*[\"'][^\"']*[\"']", "", out, flags=re.I) if n: - actions.append(f"drop data-ai* attributes x{n}") + actions.append(act.drop_data_ai_attributes(n)) out = out2 out, uri_actions = _clean_embedded_data_uris(out) actions.extend(uri_actions) if not actions: - actions.append("no HTML AI meta removed") + actions.append(act.nothing_removed("html", "no HTML AI meta removed")) return out, actions @@ -570,8 +578,9 @@ def _jsonld_sub(m: re.Match[str]) -> str: # --------------------------------------------------------------------------- -def inspect_svg(data: bytes) -> tuple[bool, bool, list[str], dict]: +def inspect_svg(data: bytes) -> ContainerScan: findings: list[str] = [] + notes: list[str] = [] has_c2pa, has_ai, hits = _blob_hits(data) findings.extend(hits) try: @@ -585,19 +594,20 @@ def inspect_svg(data: bytes) -> tuple[bool, bool, list[str], dict]: if re.search(r"c2pa|jumbf", text, re.I): has_c2pa = True - uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text) + uri_c2pa, uri_ai, uri_findings, uri_notes = _inspect_embedded_data_uris(text) if uri_c2pa: has_c2pa = True if uri_ai: has_ai = True findings.extend(uri_findings) + notes.extend(uri_notes) except Exception as e: findings.append(f"svg decode note: {e}") - return has_c2pa, has_ai or has_c2pa, findings, {} + return has_c2pa, has_ai or has_c2pa, findings, notes, {} -def clean_svg(data: bytes) -> tuple[bytes, list[str]]: - actions: list[str] = [] +def clean_svg(data: bytes) -> tuple[bytes, list[Action]]: + actions: list[Action] = [] text = data.decode("utf-8", errors="surrogateescape") # Drop metadata blocks new, n = re.subn( @@ -607,7 +617,7 @@ def clean_svg(data: bytes) -> tuple[bytes, list[str]]: flags=re.I | re.DOTALL, ) if n: - actions.append(f"drop x{n}") + actions.append(act.drop_svg_metadata(n)) text = new # Drop adobe xmp packets new, n = re.subn( @@ -617,14 +627,14 @@ def clean_svg(data: bytes) -> tuple[bytes, list[str]]: flags=re.I | re.DOTALL, ) if n: - actions.append(f"drop xmpmeta x{n}") + actions.append(act.drop_svg_xmp(n)) text = new # Drop comments that look like provenance def _cmt(m: re.Match[str]) -> str: body = m.group(0) if AI_META_NAME_RE.search(body): - actions.append("drop SVG comment with AI markers") + actions.append(act.drop_svg_comment()) return "" return body @@ -644,10 +654,10 @@ def _cmt(m: re.Match[str]) -> str: flags=re.I, ) if n: - actions.append(f"drop generator-like attrs x{n}") + actions.append(act.drop_svg_generator_attributes(n)) text = new if not actions: - actions.append("no SVG metadata removed") + actions.append(act.nothing_removed("svg", "no SVG metadata removed")) return text.encode("utf-8", errors="surrogateescape"), actions @@ -791,8 +801,9 @@ def _is_docx_meta_part(name: str) -> bool: return name.startswith(("docProps/", "customXml/")) -def _inspect_ooxml_zip(data: bytes, fmt: str) -> tuple[bool, bool, list[str], dict]: +def _inspect_ooxml_zip(data: bytes, fmt: str) -> ContainerScan: findings: list[str] = [] + notes: list[str] = [] has_c2pa = False has_ai = False parts: list[str] = [] @@ -824,31 +835,32 @@ def _inspect_ooxml_zip(data: bytes, fmt: str) -> tuple[bool, bool, list[str], di ): raw = _read_zip_member(zf, info, budget) img_fmt = detect_format(raw) - sub_c2pa, sub_ai, sub_findings = False, False, [] + scan: MetadataScan = (False, False, [], []) if img_fmt == "png": - sub_c2pa, sub_ai, sub_findings = inspect_png(raw) + scan = inspect_png(raw) elif img_fmt == "jpeg": - sub_c2pa, sub_ai, sub_findings = inspect_jpeg(raw) + scan = inspect_jpeg(raw) elif img_fmt == "webp": - sub_c2pa, sub_ai, sub_findings = inspect_webp(raw) + scan = inspect_webp(raw) elif img_fmt == "avif": - sub_c2pa, sub_ai, sub_findings = inspect_avif(raw) + scan = inspect_avif(raw) elif img_fmt == "heif": - sub_c2pa, sub_ai, sub_findings = inspect_heic(raw) + scan = inspect_heic(raw) elif img_fmt == "gif": - sub_c2pa, sub_ai, sub_findings = inspect_gif(raw) + scan = inspect_gif(raw) elif img_fmt == "tiff": - sub_c2pa, sub_ai, sub_findings = inspect_tiff(raw) + scan = inspect_tiff(raw) elif img_fmt == "bmp": - sub_c2pa, sub_ai, sub_findings = inspect_bmp(raw) + scan = inspect_bmp(raw) elif name.lower().endswith(".svg") or raw.lstrip().startswith(b"<"): - sub_c2pa, sub_ai, sub_findings, _ = inspect_svg(raw) + scan = inspect_svg(raw)[:4] + sub_c2pa, sub_ai, sub_findings, sub_notes = scan if sub_c2pa: has_c2pa = True if sub_ai or sub_c2pa: has_ai = True - for sf in sub_findings: - findings.append(f"{name}: {sf}") + findings.extend(f"{name}: {sf}" for sf in sub_findings) + notes.extend(f"{name}: {sn}" for sn in sub_notes) continue # Only metadata/provenance parts carry AI markers. The visible @@ -867,21 +879,21 @@ def _inspect_ooxml_zip(data: bytes, fmt: str) -> tuple[bool, bool, list[str], di # always flag customXml presence lightly custom = [n for n in parts if n.startswith("customXml/")] if custom: - findings.append(f"customXml parts: {len(custom)}") + notes.append(f"customXml parts: {len(custom)}") except _ZIP_PARSE_ERRORS: - return False, False, [f"not a valid {fmt.upper()} zip"], {} - return has_c2pa, has_ai or has_c2pa, findings, {"parts": len(parts)} + return False, False, [f"not a valid {fmt.upper()} zip"], [], {} + return has_c2pa, has_ai or has_c2pa, findings, notes, {"parts": len(parts)} -def inspect_docx(data: bytes) -> tuple[bool, bool, list[str], dict]: +def inspect_docx(data: bytes) -> ContainerScan: return _inspect_ooxml_zip(data, "docx") -def inspect_xlsx(data: bytes) -> tuple[bool, bool, list[str], dict]: +def inspect_xlsx(data: bytes) -> ContainerScan: return _inspect_ooxml_zip(data, "xlsx") -def inspect_pptx(data: bytes) -> tuple[bool, bool, list[str], dict]: +def inspect_pptx(data: bytes) -> ContainerScan: return _inspect_ooxml_zip(data, "pptx") @@ -1007,8 +1019,8 @@ def _drop(m: re.Match[str]) -> str: def _scrub_ooxml_zip( data: bytes, fmt: str, *, also_layer_a_text: bool = True -) -> tuple[bytes, list[str]]: - actions: list[str] = [] +) -> tuple[bytes, list[Action]]: + actions: list[Action] = [] budget = [0] layer_removed = 0 layer_replaced = 0 @@ -1027,7 +1039,7 @@ def _scrub_ooxml_zip( re.I, ): img_fmt = detect_format(raw) - sub_actions: list[str] = [] + sub_actions: list[Action] = [] cleaned_bytes = raw try: if img_fmt == "png": @@ -1051,20 +1063,20 @@ def _scrub_ooxml_zip( except Exception: # noqa: S110 pass if _media_strip_succeeded(sub_actions, cleaned_bytes, raw): - actions.append(f"clean embedded media in {name} ({', '.join(sub_actions[:2])})") + actions.append(act.clean_embedded_media(name, sub_actions)) raw = cleaned_bytes kept.append((info, raw)) continue # 2. Drop customXml trees if name.startswith("customXml/"): - actions.append(f"drop part {name}") + actions.append(act.drop_part(name)) continue # 3. docProps/ provenance if name in DOCX_META_PARTS or name.startswith("docProps/"): if name.endswith("custom.xml"): - actions.append(f"drop part {name}") + actions.append(act.drop_part(name)) continue text = raw.decode("utf-8", errors="replace") new = text @@ -1073,7 +1085,7 @@ def _scrub_ooxml_zip( def _empty(m: re.Match[str], _label=label, _name=name) -> str: if m.group(2): - actions.append(f"scrub {_name} field {_label}") + actions.append(act.scrub_field(_name, _label)) return m.group(1) + m.group(3) new = re.sub(pat, _empty, new, flags=re.I | re.DOTALL) @@ -1087,7 +1099,7 @@ def _empty(m: re.Match[str], _label=label, _name=name) -> str: text, ) if n: - actions.append(f"drop Content_Types customXml overrides x{n}") + actions.append(act.drop_content_type_overrides("customXml", n)) raw = new.encode("utf-8") new, n = re.subn( r"]*PartName=\"/docProps/custom\.xml\"[^>]*/>", @@ -1095,7 +1107,7 @@ def _empty(m: re.Match[str], _label=label, _name=name) -> str: raw.decode("utf-8", errors="replace"), ) if n: - actions.append(f"drop Content_Types custom.xml override x{n}") + actions.append(act.drop_content_type_overrides("custom.xml", n)) raw = new.encode("utf-8") # 5. Layer A text runs @@ -1131,7 +1143,7 @@ def _empty(m: re.Match[str], _label=label, _name=name) -> str: if info.filename.endswith(".rels"): part_raw, n = _prune_dangling_relationships(info.filename, raw, kept_names) if n: - actions.append(f"prune dangling relationships x{n} in {info.filename}") + actions.append(act.prune_relationships(info.filename, n)) final.append((info, part_raw)) out_buf = io.BytesIO() @@ -1139,21 +1151,21 @@ def _empty(m: re.Match[str], _label=label, _name=name) -> str: for info, raw in final: zout.writestr(info, raw) if layer_removed or layer_replaced: - actions.append(f"layer A text: removed={layer_removed} replaced={layer_replaced}") + actions.append(act.layer_a_text(layer_removed, layer_replaced)) if not actions: - actions.append(f"no {fmt.upper()} metadata parts removed") + actions.append(act.nothing_removed(fmt, f"no {fmt.upper()} metadata parts removed")) return out_buf.getvalue(), actions -def clean_docx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: +def clean_docx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[Action]]: return _scrub_ooxml_zip(data, "docx", also_layer_a_text=also_layer_a_text) -def clean_xlsx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: +def clean_xlsx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[Action]]: return _scrub_ooxml_zip(data, "xlsx", also_layer_a_text=also_layer_a_text) -def clean_pptx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: +def clean_pptx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[Action]]: return _scrub_ooxml_zip(data, "pptx", also_layer_a_text=also_layer_a_text) @@ -1180,7 +1192,7 @@ def _drop(m: re.Match[str]) -> str: return new.encode("utf-8"), removed[0] -def inspect_odt(data: bytes) -> tuple[bool, bool, list[str], dict]: +def inspect_odt(data: bytes) -> ContainerScan: findings: list[str] = [] has_c2pa = False has_ai = False @@ -1206,11 +1218,11 @@ def inspect_odt(data: bytes) -> tuple[bool, bool, list[str], dict]: has_ai = True findings.append("meta.xml generator-like fields") except _ZIP_PARSE_ERRORS: - return False, False, ["not a valid ODT zip"], {} - return has_c2pa, has_ai or has_c2pa, findings, {} + return False, False, ["not a valid ODT zip"], [], {} + return has_c2pa, has_ai or has_c2pa, findings, [], {} -def clean_odt(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: +def clean_odt(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[Action]]: actions: list[str] = [] budget = [0] layer_removed = 0 @@ -1232,13 +1244,13 @@ def clean_odt(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, li flags=re.I | re.DOTALL, ) if n: - actions.append("drop meta:generator") + actions.append(act.drop_generator_meta()) text = new # scrub creator-like if AI def _creator(m: re.Match[str]) -> str: if AI_META_NAME_RE.search(m.group(0)): - actions.append("scrub creator-like meta") + actions.append(act.scrub_creator()) return "" return m.group(0) @@ -1257,7 +1269,7 @@ def _creator(m: re.Match[str]) -> str: "mimetype", "META-INF/manifest.xml", ): - actions.append(f"drop part {name} (AI/C2PA markers)") + actions.append(act.drop_part(name, markers=True)) dropped.add(name) continue # Layer A over the visible paragraph text of the body part. @@ -1280,7 +1292,7 @@ def _creator(m: re.Match[str]) -> str: if info.filename == "META-INF/manifest.xml": pruned, n = _prune_odt_manifest_entries(part_raw, dropped) if n: - actions.append(f"drop manifest entries x{n}") + actions.append(act.prune_odf_manifest(n)) out_raw = pruned rewritten.append((info, out_raw)) kept = rewritten @@ -1290,9 +1302,9 @@ def _creator(m: re.Match[str]) -> str: for info, raw in kept: zout.writestr(info, raw) if layer_removed or layer_replaced: - actions.append(f"layer A text: removed={layer_removed} replaced={layer_replaced}") + actions.append(act.layer_a_text(layer_removed, layer_replaced)) if not actions: - actions.append("no ODT metadata removed") + actions.append(act.nothing_removed("odt", "no ODT metadata removed")) return out_buf.getvalue(), actions @@ -1344,8 +1356,9 @@ def _epub_encrypted_parts(data: bytes) -> set[str]: return names -def inspect_epub(data: bytes) -> tuple[bool, bool, list[str], dict]: +def inspect_epub(data: bytes) -> ContainerScan: findings: list[str] = [] + notes: list[str] = [] has_c2pa = False has_ai = False budget = [0] @@ -1369,13 +1382,13 @@ def inspect_epub(data: bytes) -> tuple[bool, bool, list[str], dict]: raw = _read_zip_member(zf, info, budget) if name.lower().endswith((".xhtml", ".html", ".htm")): text = raw.decode("utf-8", errors="surrogateescape") - c2, ai, sub, _ = inspect_html(text) + c2, ai, sub, sub_notes, _ = inspect_html(text) if c2: has_c2pa = True if ai: has_ai = True - for f in sub: - findings.append(f"{name}: {f}") + findings.extend(f"{name}: {f}" for f in sub) + notes.extend(f"{name}: {n}" for n in sub_notes) continue if name.lower().endswith(".opf"): text = raw.decode("utf-8", errors="surrogateescape") @@ -1394,8 +1407,8 @@ def inspect_epub(data: bytes) -> tuple[bool, bool, list[str], dict]: has_ai = has_ai or ai findings.append(f"{name}: {', '.join(hits[:6])}") except zipfile.BadZipFile: - return False, False, ["not a valid EPUB zip"], {} - return has_c2pa, has_ai or has_c2pa, findings, {"parts": len(names)} + return False, False, ["not a valid EPUB zip"], [], {} + return has_c2pa, has_ai or has_c2pa, findings, notes, {"parts": len(names)} def _prune_opf_manifest(raw: bytes, opf_name: str, dropped: set[str]) -> tuple[bytes, int]: @@ -1438,14 +1451,14 @@ def _drop_itemref(m: re.Match[str]) -> str: return new.encode("utf-8"), removed[0] -def _scrub_epub_opf(text: str) -> tuple[str, list[str]]: +def _scrub_epub_opf(text: str) -> tuple[str, list[Action]]: """Scrub AI-ish metadata from the EPUB package document (OPF).""" - actions: list[str] = [] + actions: list[Action] = [] def _meta(m: re.Match[str]) -> str: tag = m.group(0) if AI_META_NAME_RE.search(tag): - actions.append("drop OPF meta tag") + actions.append(act.drop_opf_meta()) return "" return tag @@ -1454,7 +1467,7 @@ def _meta(m: re.Match[str]) -> str: def _dc(m: re.Match[str]) -> str: if AI_META_NAME_RE.search(m.group(0)): - actions.append(f"scrub {m.group(1)} (AI vendor name)") + actions.append(act.scrub_opf_field(m.group(1))) return f"<{m.group(1)}/>" return m.group(0) @@ -1466,15 +1479,15 @@ def _dc(m: re.Match[str]) -> str: ) if not actions: - actions.append("no OPF metadata removed") + actions.append(act.nothing_removed("opf", "no OPF metadata removed")) return new, actions -def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: +def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[Action]]: """Rewrite the EPUB: scrub OPF metadata, XHTML meta/JSON-LD, and Layer A.""" from text_unicode import clean_text # local import to avoid cycles - actions: list[str] = [] + actions: list[Action] = [] budget = [0] layer_removed = 0 layer_replaced = 0 @@ -1496,7 +1509,7 @@ def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, l # 1. Embedded raster / vector media: strip metadata if _EPUB_MEDIA_RE.search(low): img_fmt = detect_format(raw) - sub_actions: list[str] = [] + sub_actions: list[Action] = [] cleaned = raw try: if img_fmt == "png": @@ -1520,7 +1533,7 @@ def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, l except Exception: # noqa: S110 pass if _media_strip_succeeded(sub_actions, cleaned, raw): - actions.append(f"clean embedded media in {name} ({', '.join(sub_actions[:2])})") + actions.append(act.clean_embedded_media(name, sub_actions)) raw = cleaned kept.append((info, raw)) continue @@ -1528,9 +1541,9 @@ def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, l # 2. XHTML content: strip AI meta/JSON-LD, then Layer A if low.endswith((".xhtml", ".html", ".htm")): text = raw.decode("utf-8", errors="surrogateescape") - text, sub_actions = clean_html(text) - if sub_actions and sub_actions != ["no HTML AI meta removed"]: - actions.append(f"{name}: {', '.join(sub_actions[:2])}") + text, html_actions = clean_html(text) + if any_change(html_actions): + actions.append(act.clean_part(name, html_actions)) if also_layer_a_text: text2, stats = clean_text(text) if stats["removed_count"] or stats["replaced_count"]: @@ -1544,9 +1557,9 @@ def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, l # 3. Package document (OPF): scrub AI-ish metadata if low.endswith(".opf"): text = raw.decode("utf-8", errors="surrogateescape") - new_text, sub_actions = _scrub_epub_opf(text) - if sub_actions and sub_actions != ["no OPF metadata removed"]: - actions.extend(f"{name}: {a}" for a in sub_actions) + new_text, opf_actions = _scrub_epub_opf(text) + if any_change(opf_actions): + actions.extend(act.clean_part(name, [action]) for action in opf_actions) raw = new_text.encode("utf-8", errors="surrogateescape") kept.append((info, raw)) continue @@ -1554,7 +1567,7 @@ def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, l # 4. Other parts: drop non-content parts carrying AI/C2PA markers c2, ai, _hits = _blob_hits(raw) if (c2 or ai) and not _epub_content_part(name): - actions.append(f"drop part {name} (AI/C2PA markers)") + actions.append(act.drop_part(name, markers=True)) dropped.add(name) continue @@ -1572,7 +1585,7 @@ def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, l if info.filename.lower().endswith(".opf"): pruned, n = _prune_opf_manifest(part_raw, info.filename, dropped) if n: - actions.append(f"prune OPF manifest entries x{n}") + actions.append(act.prune_opf_manifest(n)) out_raw = pruned rewritten.append((info, out_raw)) kept = rewritten @@ -1583,9 +1596,9 @@ def clean_epub(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, l zout.writestr(info, raw) if layer_removed or layer_replaced: - actions.append(f"layer A text: removed={layer_removed} replaced={layer_replaced}") + actions.append(act.layer_a_text(layer_removed, layer_replaced)) if not actions: - actions.append("no EPUB metadata removed") + actions.append(act.nothing_removed("epub", "no EPUB metadata removed")) return out_buf.getvalue(), actions @@ -1612,13 +1625,14 @@ def _pdf_structured_blob(data: bytes) -> bytes: return no_streams + b"\n" + xmp -def inspect_pdf(path: Path, data: bytes) -> tuple[bool, bool, list[str], dict]: +def inspect_pdf(path: Path, data: bytes) -> ContainerScan: findings: list[str] = [] + notes: list[str] = [] has_c2pa, has_ai, hits = _blob_hits(_pdf_structured_blob(data)) findings.extend(f"pdf-structured:{h}" for h in hits) xmp_blob = b"\n".join(_XMP_PACKET_RE.findall(data)) if xmp_blob: - findings.append("XMP packet present") + notes.append("XMP packet present") has_ai = has_ai or bool( re.search( rb"digitalSourceType|trainedAlgorithmicMedia|SoftwareAgent|c2pa", @@ -1631,18 +1645,18 @@ def inspect_pdf(path: Path, data: bytes) -> tuple[bool, bool, list[str], dict]: if ct.get("has_manifest"): has_c2pa = True findings.append("c2patool reports C2PA-related manifest") - return has_c2pa, has_ai or has_c2pa, findings, {"tools": tools} + return has_c2pa, has_ai or has_c2pa, findings, notes, {"tools": tools} def clean_pdf_pypdf( path: Path, dest: Path, *, skip_exiftool: bool = False -) -> tuple[list[str], dict]: +) -> tuple[list[Action], dict]: """Clean PDF metadata. exiftool > full-document pypdf clone > unchanged copy. *skip_exiftool* is used by clean_pdf when exiftool already ran and failed, so the fallback does not invoke the same failing command a second time. """ - actions: list[str] = [] + actions: list[Action] = [] data = path.read_bytes() dest.parent.mkdir(parents=True, exist_ok=True) @@ -1657,15 +1671,36 @@ def clean_pdf_pypdf( timeout=60, output_limit=2 * 1024 * 1024, ) - actions.append(f"exiftool -all= (rc={result.returncode})") + actions.append(act.exiftool_run(result.returncode)) if result.returncode == 0 and not (result.stdout_truncated or result.stderr_truncated): return actions, {"mode": "exiftool", "degraded": False} if result.stdout_truncated or result.stderr_truncated: - actions.append("exiftool output exceeded safety limit; trying pypdf") + actions.append( + act.tool_failed( + "exiftool", + "exiftool output exceeded safety limit; trying pypdf", + detail="output exceeded safety limit", + fallback="pypdf", + ) + ) else: - actions.append(f"exiftool degraded (rc={result.returncode}); trying pypdf") + actions.append( + act.tool_failed( + "exiftool", + f"exiftool degraded (rc={result.returncode}); trying pypdf", + returncode=result.returncode, + fallback="pypdf", + ) + ) except Exception as error: - actions.append(f"exiftool failed: {error}; trying pypdf") + actions.append( + act.tool_failed( + "exiftool", + f"exiftool failed: {error}; trying pypdf", + detail=str(error), + fallback="pypdf", + ) + ) # Strategy 2: clone the complete document graph, then remove only metadata. # Copying pages alone loses outlines, forms, attachments, labels, and viewer state. @@ -1678,10 +1713,10 @@ def clean_pdf_pypdf( reader = PdfReader(str(path)) if reader.is_encrypted: if reader.decrypt("") == 0: - actions.append("encrypted PDF (password required); copied as-is") + actions.append(act.pdf_encrypted()) dest.write_bytes(data) return actions, {"mode": "copy-encrypted", "degraded": True} - actions.append("decrypted with empty password") + actions.append(act.pdf_decrypted()) writer = PdfWriter() writer.clone_document_from_reader(reader) page_metadata = 0 @@ -1690,11 +1725,11 @@ def clean_pdf_pypdf( del page["/Metadata"] page_metadata += 1 if page_metadata: - actions.append(f"pypdf: drop per-page /Metadata x{page_metadata}") + actions.append(act.drop_pdf_page_metadata(page_metadata)) if reader.metadata: - actions.append("pypdf: drop document info dictionary") + actions.append(act.drop_pdf_docinfo()) if reader.xmp_metadata is not None or "/Metadata" in writer.root_object: - actions.append("pypdf: drop catalog XMP packet") + actions.append(act.drop_pdf_catalog_xmp()) writer.metadata = None writer.xmp_metadata = None if "/Metadata" in writer.root_object: @@ -1703,26 +1738,31 @@ def clean_pdf_pypdf( writer.write(buf) # Publish only after a complete in-memory rewrite. atomic_write_bytes(dest, buf.getvalue()) - actions.append("pypdf: cloned full document graph; removed docinfo/XMP") + actions.append(act.pdf_rewritten_pypdf()) return actions, {"mode": "pypdf", "degraded": False} except Exception as e: - actions.append(f"pypdf failed: {e}; copied unchanged") + actions.append( + act.tool_failed("pypdf", f"pypdf failed: {e}; copied unchanged", detail=str(e)) + ) else: - actions.append("pypdf not installed; copied unchanged") + actions.append(act.tool_missing("pypdf", "pypdf not installed; copied unchanged")) # Never delete bytes from a PDF without rebuilding xref/object offsets. atomic_write_bytes(dest, data) - actions.append("no structural PDF cleaner succeeded; copied unchanged") + actions.append(act.pdf_copied_unchanged()) return actions, {"mode": "copy", "degraded": True} -def _pdf_structural_rewrite(dest: Path, actions: list[str]) -> bool: +def _pdf_structural_rewrite(dest: Path, actions: list[Action]) -> bool: """Rebuild a PDF so unreferenced objects are dropped (qpdf --linearize).""" qpdf = which("qpdf") if not qpdf: actions.append( - "warning: exiftool PDF edits are incremental — the original metadata " - "bytes remain recoverable; install qpdf for a structural rewrite" + act.pdf_rewrite_failed( + "warning: exiftool PDF edits are incremental — the original metadata " + "bytes remain recoverable; install qpdf for a structural rewrite", + detail="qpdf not installed", + ) ) return False @@ -1735,24 +1775,31 @@ def _pdf_structural_rewrite(dest: Path, actions: list[str]) -> bool: ) except Exception as e: tmp.unlink(missing_ok=True) - actions.append(f"qpdf rewrite failed: {e}; metadata bytes may remain recoverable") + actions.append( + act.pdf_rewrite_failed( + f"qpdf rewrite failed: {e}; metadata bytes may remain recoverable", detail=str(e) + ) + ) return False if r.returncode in (0, 3) and tmp.is_file() and tmp.stat().st_size > 0: tmp.replace(dest) - actions.append(f"qpdf --linearize structural rewrite (rc={r.returncode})") + actions.append(act.pdf_rewritten_qpdf(r.returncode)) return True tmp.unlink(missing_ok=True) actions.append( - f"qpdf rewrite skipped (rc={r.returncode}); metadata bytes may remain recoverable" + act.pdf_rewrite_failed( + f"qpdf rewrite skipped (rc={r.returncode}); metadata bytes may remain recoverable", + returncode=r.returncode, + ) ) return False -def clean_pdf(path: Path, dest: Path) -> tuple[list[str], dict]: +def clean_pdf(path: Path, dest: Path) -> tuple[list[Action], dict]: """Best-effort PDF clean. Prefers exiftool + qpdf; falls back to pypdf.""" - actions: list[str] = [] + actions: list[Action] = [] data = path.read_bytes() dest.parent.mkdir(parents=True, exist_ok=True) @@ -1766,28 +1813,40 @@ def clean_pdf(path: Path, dest: Path) -> tuple[list[str], dict]: timeout=60, output_limit=2 * 1024 * 1024, ) - actions.append(f"exiftool -all= (rc={r.returncode})") + actions.append(act.exiftool_run(r.returncode)) truncated = bool( getattr(r, "stdout_truncated", False) or getattr(r, "stderr_truncated", False) ) exiftool_ok = r.returncode == 0 and not truncated if r.returncode != 0: - actions.append(f"exiftool degraded (rc={r.returncode})") + actions.append( + act.tool_failed( + "exiftool", + f"exiftool degraded (rc={r.returncode})", + returncode=r.returncode, + ) + ) elif truncated: - actions.append("exiftool output exceeded safety limit") + actions.append( + act.tool_failed( + "exiftool", + "exiftool output exceeded safety limit", + detail="output exceeded safety limit", + ) + ) except Exception as e: - actions.append(f"exiftool failed: {e}") + actions.append(act.tool_failed("exiftool", f"exiftool failed: {e}", detail=str(e))) if not exiftool_ok: # exiftool ran but did not strip; dest still holds the original # bytes. Hand off to the pypdf path rather than publishing # unstripped output under mode "exiftool" with no degraded flag. - actions.append("trying pypdf fallback") + actions.append(act.try_fallback("pypdf")) fallback_actions, fallback_meta = clean_pdf_pypdf(path, dest, skip_exiftool=True) return actions + fallback_actions, fallback_meta rewritten = _pdf_structural_rewrite(dest, actions) c2patool = which("c2patool") if c2patool: - actions.append("c2patool available for inspect; strip via exiftool/re-export") + actions.append(act.c2patool_hint()) return actions, {"mode": "exiftool", "structural_rewrite": rewritten} return clean_pdf_pypdf(path, dest) @@ -1806,24 +1865,25 @@ def inspect_container(path: Path) -> ContainerInspectReport: layer_a_total = 0 layer_a_hits: list[dict] = [] + notes: list[str] = [] if fmt == "svg": - has_c2pa, has_ai, findings, details = inspect_svg(data) + has_c2pa, has_ai, findings, notes, details = inspect_svg(data) elif fmt == "pdf": - has_c2pa, has_ai, findings, details = inspect_pdf(path, data) + has_c2pa, has_ai, findings, notes, details = inspect_pdf(path, data) tools = details.pop("tools", {}) elif fmt == "docx": - has_c2pa, has_ai, findings, details = inspect_docx(data) + has_c2pa, has_ai, findings, notes, details = inspect_docx(data) elif fmt == "xlsx": - has_c2pa, has_ai, findings, details = inspect_xlsx(data) + has_c2pa, has_ai, findings, notes, details = inspect_xlsx(data) elif fmt == "pptx": - has_c2pa, has_ai, findings, details = inspect_pptx(data) + has_c2pa, has_ai, findings, notes, details = inspect_pptx(data) elif fmt == "odt": - has_c2pa, has_ai, findings, details = inspect_odt(data) + has_c2pa, has_ai, findings, notes, details = inspect_odt(data) elif fmt == "epub": - has_c2pa, has_ai, findings, details = inspect_epub(data) + has_c2pa, has_ai, findings, notes, details = inspect_epub(data) elif fmt == "html": body = data.decode("utf-8", errors="surrogateescape") - has_c2pa, has_ai, findings, details = inspect_html(body) + has_c2pa, has_ai, findings, notes, details = inspect_html(body) from text_unicode import inspect_text # local import to avoid cycles ta = inspect_text(body).to_dict() @@ -1833,7 +1893,7 @@ def inspect_container(path: Path) -> ContainerInspectReport: findings.append(f"layer-a: {h['codepoint']} {h['label']} x{h['count']} ({h['kind']})") elif fmt == "markdown": body = data.decode("utf-8", errors="surrogateescape") - has_c2pa, has_ai, findings, details = inspect_markdown(body) + has_c2pa, has_ai, findings, notes, details = inspect_markdown(body) from text_unicode import inspect_text # local import to avoid cycles ta = inspect_text(body).to_dict() @@ -1870,7 +1930,6 @@ def inspect_container(path: Path) -> ContainerInspectReport: except zipfile.BadZipFile: pass - notes: list[str] = [] if fmt == "pdf": notes.append( "PDF inspection is best-effort; exiftool/c2patool give more reliable metadata detection" @@ -1918,7 +1977,7 @@ def clean_container( data = path.read_bytes() fmt = fmt or detect_container_format(path, data) - actions: list[str] = [] + actions: list[Action] = [] dest.parent.mkdir(parents=True, exist_ok=True) meta: dict[str, Any] = {"format": fmt} @@ -1949,9 +2008,7 @@ def clean_container( if also_layer_a_text: text2, stats = clean_text(text) if stats["removed_count"] or stats["replaced_count"]: - actions.append( - f"layer A text: removed={stats['removed_count']} replaced={stats['replaced_count']}" - ) + actions.append(act.layer_a_text(stats["removed_count"], stats["replaced_count"])) text = text2 atomic_write_text(dest, text) elif fmt == "markdown": @@ -1960,9 +2017,7 @@ def clean_container( if also_layer_a_text: text2, stats = clean_text(text) if stats["removed_count"] or stats["replaced_count"]: - actions.append( - f"layer A text: removed={stats['removed_count']} replaced={stats['replaced_count']}" - ) + actions.append(act.layer_a_text(stats["removed_count"], stats["replaced_count"])) text = text2 atomic_write_text(dest, text) else: @@ -1973,7 +2028,7 @@ def clean_container( "input": str(path), "output": str(dest), "format": fmt, - "actions": actions, + **report_actions(actions), "bytes_in": len(data), "bytes_out": dest.stat().st_size, "still_has_c2pa": after.has_c2pa, diff --git a/skills/remove-ai-marks/scripts/external_command.py b/skills/remove-ai-marks/scripts/external_command.py index 801b93e..fd142a3 100644 --- a/skills/remove-ai-marks/scripts/external_command.py +++ b/skills/remove-ai-marks/scripts/external_command.py @@ -76,6 +76,24 @@ def _validate_command( return command +def _sweep_group(process_group_id: int) -> None: + """SIGKILL a process group again until no member is left. + + One ``killpg`` can miss a child that a member was forking while the signal + was delivered; that child inherits the group, so repeating the signal + reaches it. Stops once the group is empty or the grace period ends. + """ + deadline = time.monotonic() + _TERMINATION_GRACE + while True: + try: + os.killpg(process_group_id, signal.SIGKILL) + except (ProcessLookupError, PermissionError): + return + if time.monotonic() >= deadline: + return + time.sleep(0.01) + + def run_command( argv: Sequence[str], *, @@ -117,10 +135,11 @@ def drain(name: str, stream) -> None: start_new_session=os.name == "posix", ) if os.name == "posix": - try: - process_group_id = os.getpgid(process.pid) - except ProcessLookupError: - process_group_id = None + # start_new_session makes the child a session leader, so its group + # id is its pid. Asking getpgid() instead fails once a quick leader + # has exited (macOS answers ESRCH for a zombie), which used to skip + # the group kill and leave its descendants running. + process_group_id = process.pid if process.stdout is None or process.stderr is None: raise RuntimeError("external command output pipes unavailable") @@ -161,6 +180,9 @@ def drain(name: str, stream) -> None: returncode = process.wait(timeout=_TERMINATION_GRACE) except subprocess.TimeoutExpired: timed_out = True + if os.name == "posix" and group_cleanup_needed: + # Once the leader is reaped, an empty group reads as gone. + _sweep_group(process_group_id) for stream in (process.stdout, process.stderr): if stream is not None: diff --git a/skills/remove-ai-marks/scripts/heif_meta.py b/skills/remove-ai-marks/scripts/heif_meta.py index 1652b5f..86b4c64 100644 --- a/skills/remove-ai-marks/scripts/heif_meta.py +++ b/skills/remove-ai-marks/scripts/heif_meta.py @@ -14,8 +14,10 @@ from pathlib import Path from typing import Any +import pipeline_actions as act from common import atomic_write_bytes from image_meta import AI_META_HINTS, C2PA_MARKERS, _contains_any +from pipeline_actions import Action, report_actions HEIF_BRANDS = { b"heic", @@ -306,13 +308,20 @@ def _neutralize_runs(buf: bytearray, start: int, length: int, pad: int) -> list[ return sorted(set(hits)) -def inspect_heif(data: bytes) -> tuple[bool, bool, list[str], dict[str, Any]]: +def inspect_heif(data: bytes) -> tuple[bool, bool, list[str], list[str], dict[str, Any]]: + """``(has_c2pa, has_ai, findings, notes, details)`` for a HEIF/AVIF file. + + Findings are C2PA/AI signals and the malformed or unsupported layouts that + leave the verdict unproven. Notes are context only: the brands, and + metadata that is present but carries no AI markers. + """ fmt = detect_heif(data) if fmt == "unknown": - return False, False, ["not a HEIF/AVIF file"], {} + return False, False, ["not a HEIF/AVIF file"], [], {} findings: list[str] = [] + notes: list[str] = [] brands = sorted(b.decode("ascii", "replace") for b in _ftyp_brands(data) if b.strip()) - findings.append(f"brands: {', '.join(brands[:6])}") + notes.append(f"brands: {', '.join(brands[:6])}") has_c2pa = False has_ai = False @@ -334,7 +343,7 @@ def inspect_heif(data: bytes) -> tuple[bool, bool, list[str], dict[str, Any]]: has_c2pa = True findings.append(f"XMP uuid box @ {box.header_start}: {', '.join(hits[:8])}") else: - findings.append(f"XMP uuid box @ {box.header_start}") + notes.append(f"XMP uuid box @ {box.header_start}") for box in _iter_boxes(data, 0, len(data)): if box.type != b"uuid": @@ -355,6 +364,7 @@ def inspect_heif(data: bytes) -> tuple[bool, bool, list[str], dict[str, Any]]: has_c2pa, True, findings, + notes, { "format": fmt, "brands": brands, @@ -387,23 +397,24 @@ def inspect_heif(data: bytes) -> tuple[bool, bool, list[str], dict[str, Any]]: has_c2pa = True findings.append(f"{label} @ {off}: {', '.join(hits[:8])}") else: - findings.append(f"{label} present ({ln} bytes, no AI markers)") + notes.append(f"{label} present ({ln} bytes, no AI markers)") if not items: - findings.append("no iinf item table (or unsupported version)") + notes.append("no iinf item table (or unsupported version)") if not has_c2pa: whole = _contains_any(data, C2PA_MARKERS) if whole: has_c2pa = True findings.append(f"byte-scan C2PA markers: {', '.join(whole[:6])}") - return has_c2pa, has_ai or has_c2pa, findings, {"format": fmt, "brands": brands} + return has_c2pa, has_ai or has_c2pa, findings, notes, {"format": fmt, "brands": brands} -def neutralize_heif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[str]]: +def neutralize_heif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[Action]]: """In-place neutralization on raw bytes. Returns (cleaned, actions).""" - if detect_heif(data) == "unknown": + fmt = detect_heif(data) + if fmt == "unknown": raise ValueError("not a HEIF/AVIF file") - actions: list[str] = [] + actions: list[Action] = [] buf = bytearray(data) # 1. Neutralize C2PA/JUMBF boxes: retype to 'free', zero payload. @@ -412,9 +423,7 @@ def neutralize_heif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[by buf[box.header_start + 4 : box.header_start + 8] = b"free" for i in range(box.payload_start, box.end): buf[i] = 0 - actions.append( - f"neutralized '{name}' box -> free (zeroed {box.end - box.payload_start} payload bytes)" - ) + actions.append(act.neutralize_box(name, box.end - box.payload_start)) # 1b. Neutralize XMP `uuid` boxes (XMP_UUID user-type) at top level or inside meta. for box in _xmp_uuid_boxes(bytes(buf)): @@ -422,15 +431,13 @@ def neutralize_heif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[by pad = 0x20 for i in range(box.payload_start + 16, box.end): buf[i] = pad - actions.append( - f"zeroed XMP uuid box payload ({box.end - box.payload_start - 16} bytes, offsets preserved)" - ) + actions.append(act.zero_uuid_payload(box.end - box.payload_start - 16)) else: hits = _neutralize_runs( buf, box.payload_start + 16, box.end - box.payload_start - 16, 0x20 ) if hits: - actions.append(f"neutralized AI tokens in XMP uuid box: {', '.join(hits[:8])}") + actions.append(act.neutralize_tokens("XMP uuid box", hits[:8])) # 2. Exif / XMP item extents. items, extents = _provenance_items(bytes(buf)) @@ -454,15 +461,15 @@ def neutralize_heif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[by pad = 0x20 if _is_xmp_item(itype, content_type) else 0x00 for i in range(off, off + ln): buf[i] = pad - actions.append(f"zeroed entire {label} payload ({ln} bytes, offsets preserved)") + actions.append(act.zero_item_payload(label, ln)) else: pad = 0x20 if _is_xmp_item(itype, content_type) else 0x00 hits = _neutralize_runs(buf, off, ln, pad) if hits: - actions.append(f"neutralized AI tokens in {label}: {', '.join(hits[:8])}") + actions.append(act.neutralize_tokens(label, hits[:8])) if not actions: - actions.append("no HEIF/AVIF provenance metadata found") + actions.append(act.nothing_removed(fmt, "no HEIF/AVIF provenance metadata found")) return bytes(buf), actions @@ -477,12 +484,12 @@ def clean_heif( cleaned, actions = neutralize_heif(data, strip_all_metadata=strip_all_metadata) atomic_write_bytes(dest, cleaned) - has_c2pa, has_ai, post_findings, _ = inspect_heif(cleaned) + has_c2pa, has_ai, post_findings, _notes, _ = inspect_heif(cleaned) return { "input": str(path), "output": str(dest), "format": fmt, - "actions": actions, + **report_actions(actions), "bytes_in": len(data), "bytes_out": dest.stat().st_size, "still_has_c2pa": has_c2pa, diff --git a/skills/remove-ai-marks/scripts/image_meta.py b/skills/remove-ai-marks/scripts/image_meta.py index f42f1d4..b474f32 100755 --- a/skills/remove-ai-marks/scripts/image_meta.py +++ b/skills/remove-ai-marks/scripts/image_meta.py @@ -21,7 +21,9 @@ from urllib.parse import urlparse import external_command +import pipeline_actions as act from common import atomic_write_bytes, classify_finding_confidence, which +from pipeline_actions import Action, report_actions from png_chunks import iter_png_chunks SCRIPTS_DIR = Path(__file__).resolve().parent @@ -127,6 +129,13 @@ # ...) are free text and are never scanned for product names. _GENERATOR_TEXT_KEYS = ("software", "creator", "parameters") +#: One format scanner's verdict: ``(has_c2pa, has_ai_metadata, findings, notes)``. +#: ``findings`` are AI/provenance signals, plus scan problems (a truncated or +#: malformed structure) that leave the verdict unproven. ``notes`` are context +#: that is neither: container brands, metadata that is present but carries no +#: AI markers. Only findings may be counted or shown as marks. +MetadataScan = tuple[bool, bool, list[str], list[str]] + @dataclass class ImageInspectReport: @@ -172,15 +181,14 @@ def detect_format(data: bytes) -> str: def _contains_any(blob: bytes, needles: tuple[bytes, ...]) -> list[str]: - found = [] + """Return the needles found in ``blob``, matched and listed once regardless of case.""" + found: dict[bytes, str] = {} lower = blob.lower() for n in needles: - if n.lower() in lower: - try: - found.append(n.decode("ascii", errors="replace")) - except Exception: - found.append(repr(n)) - return found + key = n.lower() + if key not in found and key in lower: + found[key] = n.decode("ascii", errors="replace") + return list(found.values()) def _zlib_decompress_bounded(data: bytes, max_bytes: int = MAX_PNG_TEXT_BYTES) -> bytes | None: @@ -297,12 +305,12 @@ def _text_chunk_is_ai(payload: bytes, ctype: bytes) -> bool: return bool(_generator_product_hits(_png_text_entries(payload, ctype))) -def inspect_png(data: bytes) -> tuple[bool, bool, list[str]]: +def inspect_png(data: bytes) -> MetadataScan: findings: list[str] = [] has_c2pa = False has_ai = False if not data.startswith(PNG_SIG): - return False, False, ["not a PNG"] + return False, False, ["not a PNG"], [] try: for chunk in iter_png_chunks(data): ctype = chunk.kind @@ -332,45 +340,45 @@ def inspect_png(data: bytes) -> tuple[bool, bool, list[str]]: if whole and not has_c2pa: has_c2pa = True findings.append(f"byte-scan C2PA markers: {', '.join(whole[:6])}") - return has_c2pa, has_ai or has_c2pa, findings + return has_c2pa, has_ai or has_c2pa, findings, [] -def inspect_avif(data: bytes) -> tuple[bool, bool, list[str]]: +def inspect_avif(data: bytes) -> MetadataScan: """Inspect AVIF (ISO-BMFF) for C2PA/JUMBF boxes and AI-marked Exif/XMP items.""" from heif_meta import inspect_heif - has_c2pa, has_ai, findings, _ = inspect_heif(data) - return has_c2pa, has_ai, findings + has_c2pa, has_ai, findings, notes, _ = inspect_heif(data) + return has_c2pa, has_ai, findings, notes -def inspect_heic(data: bytes) -> tuple[bool, bool, list[str]]: +def inspect_heic(data: bytes) -> MetadataScan: """Inspect HEIC/HEIF (ISO-BMFF) for C2PA/JUMBF boxes and AI-marked Exif/XMP items.""" from heif_meta import inspect_heif - has_c2pa, has_ai, findings, _ = inspect_heif(data) - return has_c2pa, has_ai, findings + has_c2pa, has_ai, findings, notes, _ = inspect_heif(data) + return has_c2pa, has_ai, findings, notes -def strip_avif(data: bytes, *, strip_all: bool = True) -> tuple[bytes, list[str]]: +def strip_avif(data: bytes, *, strip_all: bool = True) -> tuple[bytes, list[Action]]: """Neutralize C2PA/AI metadata in AVIF in place (offsets preserved; pixels untouched).""" from heif_meta import neutralize_heif return neutralize_heif(data, strip_all_metadata=strip_all) -def strip_heic(data: bytes, *, strip_all: bool = True) -> tuple[bytes, list[str]]: +def strip_heic(data: bytes, *, strip_all: bool = True) -> tuple[bytes, list[Action]]: """Neutralize C2PA/AI metadata in HEIC/HEIF in place (offsets preserved; pixels untouched).""" from heif_meta import neutralize_heif return neutralize_heif(data, strip_all_metadata=strip_all) -def inspect_jpeg(data: bytes) -> tuple[bool, bool, list[str]]: +def inspect_jpeg(data: bytes) -> MetadataScan: findings: list[str] = [] has_c2pa = False has_ai = False if not data.startswith(JPEG_SOI): - return False, False, ["not a JPEG"] + return False, False, ["not a JPEG"], [] i = 2 n = len(data) while i + 4 <= n: @@ -415,7 +423,7 @@ def inspect_jpeg(data: bytes) -> tuple[bool, bool, list[str]]: if whole and not has_c2pa: has_c2pa = True findings.append(f"byte-scan C2PA markers: {', '.join(whole[:6])}") - return has_c2pa, has_ai or has_c2pa, findings + return has_c2pa, has_ai or has_c2pa, findings, [] def run_optional_tools(path: Path) -> dict[str, Any]: @@ -582,24 +590,23 @@ def inspect_image( ) -> ImageInspectReport: data = path.read_bytes() fmt = detect_format(data) - notes: list[str] = [] if fmt in ("heif", "avif"): - has_c2pa, has_ai, findings = inspect_heic(data) + has_c2pa, has_ai, findings, notes = inspect_heic(data) elif fmt == "png": - has_c2pa, has_ai, findings = inspect_png(data) + has_c2pa, has_ai, findings, notes = inspect_png(data) elif fmt == "jpeg": - has_c2pa, has_ai, findings = inspect_jpeg(data) + has_c2pa, has_ai, findings, notes = inspect_jpeg(data) elif fmt == "webp": - has_c2pa, has_ai, findings = inspect_webp(data) + has_c2pa, has_ai, findings, notes = inspect_webp(data) elif fmt == "bmp": - has_c2pa, has_ai, findings = inspect_bmp(data) + has_c2pa, has_ai, findings, notes = inspect_bmp(data) elif fmt == "gif": - has_c2pa, has_ai, findings = inspect_gif(data) + has_c2pa, has_ai, findings, notes = inspect_gif(data) elif fmt == "tiff": - has_c2pa, has_ai, findings = inspect_tiff(data) + has_c2pa, has_ai, findings, notes = inspect_tiff(data) else: has_c2pa, has_ai, findings = False, False, ["unsupported format"] - notes.append(f"format '{fmt}' is not inspected") + notes = [f"format '{fmt}' is not inspected"] tools = run_optional_tools(path) # Elevate flags from tools @@ -620,8 +627,8 @@ def inspect_image( ) -def strip_png(data: bytes, *, strip_all_text: bool = True) -> tuple[bytes, list[str]]: - actions: list[str] = [] +def strip_png(data: bytes, *, strip_all_text: bool = True) -> tuple[bytes, list[Action]]: + actions: list[Action] = [] out = bytearray(PNG_SIG) for chunk in iter_png_chunks(data, allow_trailing_data=True): ctype = chunk.kind @@ -631,11 +638,11 @@ def strip_png(data: bytes, *, strip_all_text: bool = True) -> tuple[bytes, list[ drop = False if ctype == b"caBX" or ctype.startswith(b"c2"): drop = True - actions.append(f"drop chunk {name}") + actions.append(act.drop_png_chunk(name)) elif ctype == b"eXIf" or ctype in (b"tEXt", b"zTXt", b"iTXt"): if strip_all_text or _text_chunk_is_ai(bytes(payload), ctype): drop = True - actions.append(f"drop chunk {name}") + actions.append(act.drop_png_chunk(name)) elif _contains_any(ctype + bytes(payload), C2PA_MARKERS) and ctype not in ( b"IHDR", b"IDAT", @@ -649,12 +656,16 @@ def strip_png(data: bytes, *, strip_all_text: bool = True) -> tuple[bytes, list[ b"iCCP", ): drop = True - actions.append(f"drop chunk {name} (C2PA marker in payload)") + actions.append(act.drop_png_chunk(name, c2pa_in_payload=True)) if not drop: out.extend(chunk.raw) if not actions: - actions.append("no PNG metadata chunks removed (already clean or none matched)") + actions.append( + act.nothing_removed( + "png", "no PNG metadata chunks removed (already clean or none matched)" + ) + ) return bytes(out), actions @@ -711,10 +722,10 @@ def _find_jpeg_eoi(data: bytes, start: int) -> int | None: return None -def strip_jpeg(data: bytes, *, strip_all_app: bool = True) -> tuple[bytes, list[str]]: +def strip_jpeg(data: bytes, *, strip_all_app: bool = True) -> tuple[bytes, list[Action]]: if not data.startswith(JPEG_SOI): raise ValueError("not JPEG") - actions: list[str] = [] + actions: list[Action] = [] out = bytearray(JPEG_SOI) i = 2 n = len(data) @@ -753,7 +764,7 @@ def strip_jpeg(data: bytes, *, strip_all_app: bool = True) -> tuple[bytes, list[ raise ValueError("JPEG scan has no complete EOI marker") out.extend(b"\xff\xda") out.extend(data[i : eoi + 2]) - actions.append("preserved entropy-coded scan through EOI") + actions.append(act.preserve_jpeg_scan()) saw_eoi = True break @@ -775,20 +786,22 @@ def strip_jpeg(data: bytes, *, strip_all_app: bool = True) -> tuple[bytes, list[ ) if app11_is_c2pa: drop = True - actions.append("drop APP11 (C2PA/JUMBF)") + actions.append(act.drop_jpeg_segment("APP11", reason="C2PA/JUMBF")) elif strip_all_app and marker != 0xE0: # keep APP0 (JFIF) by default drop = True - actions.append(f"drop APP{marker - 0xE0}") + actions.append(act.drop_jpeg_segment(f"APP{marker - 0xE0}")) elif hits: drop = True - actions.append(f"drop APP{marker - 0xE0} (AI/C2PA markers)") + actions.append( + act.drop_jpeg_segment(f"APP{marker - 0xE0}", reason="AI/C2PA markers") + ) else: keep = True elif marker == 0xFE: # COM if strip_all_app or _contains_any(payload, AI_META_HINTS + C2PA_MARKERS): drop = True - actions.append("drop COM comment") + actions.append(act.drop_jpeg_segment("COM")) else: keep = True else: @@ -802,7 +815,7 @@ def strip_jpeg(data: bytes, *, strip_all_app: bool = True) -> tuple[bytes, list[ if not saw_eoi: raise ValueError("JPEG has no complete EOI marker") if not actions: - actions.append("no JPEG APP segments removed") + actions.append(act.nothing_removed("jpeg", "no JPEG APP segments removed")) return bytes(out), actions @@ -839,10 +852,10 @@ def _webp_chunks(data: bytes) -> tuple[list[tuple[bytes, bytes, bytes]], list[st return chunks, notes -def inspect_webp(data: bytes) -> tuple[bool, bool, list[str]]: +def inspect_webp(data: bytes) -> MetadataScan: chunks, findings = _webp_chunks(data) if not chunks and findings == ["not a WebP"]: - return False, False, findings + return False, False, findings, [] has_c2pa = False has_ai = False @@ -863,17 +876,17 @@ def inspect_webp(data: bytes) -> tuple[bool, bool, list[str]]: ): has_c2pa = True findings.append(f"WebP {name}: {', '.join(hits[:8])}") - return has_c2pa, has_ai or has_c2pa, findings + return has_c2pa, has_ai or has_c2pa, findings, [] -def strip_webp(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[str]]: +def strip_webp(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[Action]]: chunks, notes = _webp_chunks(data) if not chunks and notes == ["not a WebP"]: raise ValueError("not WebP") if notes: raise ValueError("malformed WebP: " + "; ".join(notes)) - actions: list[str] = [] + actions: list[Action] = [] kept: list[tuple[bytes, bytes, bytes]] = [] removed_flags = 0 metadata_flags = {b"ICCP": 0x20, b"EXIF": 0x08, b"XMP ": 0x04} @@ -884,7 +897,7 @@ def strip_webp(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, drop = strip_all_metadata or bool(_contains_any(payload, AI_META_HINTS + C2PA_MARKERS)) if drop: name = fourcc.decode("latin-1", errors="replace") - actions.append(f"drop WebP chunk {name}") + actions.append(act.drop_webp_chunk(name)) removed_flags |= metadata_flags.get(fourcc, 0) else: kept.append((fourcc, payload, padding)) @@ -900,7 +913,11 @@ def strip_webp(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, body.extend(padding if len(chunk) & 1 else b"") if not actions: - actions.append("no WebP metadata chunks removed (already clean or none matched)") + actions.append( + act.nothing_removed( + "webp", "no WebP metadata chunks removed (already clean or none matched)" + ) + ) return WEBP_RIFF + struct.pack(" bytes: return data[end:] if end < len(data) else b"" -def inspect_bmp(data: bytes) -> tuple[bool, bool, list[str]]: +def inspect_bmp(data: bytes) -> MetadataScan: """Inspect a BMP for trailing (non-standard) metadata.""" findings: list[str] = [] + notes: list[str] = [] if len(data) < 14 or data[:2] != BMP_SIG: - return False, False, ["not a BMP"] + return False, False, ["not a BMP"], [] trailing = _bmp_trailing(data) has_c2pa = False has_ai = False @@ -962,31 +980,30 @@ def inspect_bmp(data: bytes) -> tuple[bool, bool, list[str]]: has_c2pa = True findings.append(f"BMP trailing metadata: {', '.join(hits[:6])}") else: - findings.append(f"BMP has {len(trailing)} unrecognized trailing byte(s)") + notes.append(f"BMP has {len(trailing)} unrecognized trailing byte(s)") else: - findings.append("BMP has no metadata (header-only raster format)") - return has_c2pa, has_ai or has_c2pa, findings + notes.append("BMP has no metadata (header-only raster format)") + return has_c2pa, has_ai or has_c2pa, findings, notes -def strip_bmp(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[str]]: +def strip_bmp(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[Action]]: """Strip trailing non-image bytes from a BMP and fix the file-size field.""" if len(data) < 14 or data[:2] != BMP_SIG: raise ValueError("not BMP") extent = _bmp_payload_extent(data) if extent is None: - return data, ["BMP header not fully parsed; left unchanged"] + return data, [act.bmp_unparsed()] pixel_offset, size = extent end = pixel_offset + size if end >= len(data): - return data, ["no BMP trailing metadata to strip"] + return data, [act.nothing_removed("bmp", "no BMP trailing metadata to strip")] trailing = data[end:] hits = _contains_any(trailing, AI_META_HINTS + C2PA_MARKERS) if not strip_all_metadata and not hits: - return data, ["BMP trailing bytes kept (keep-non-ai-metadata)"] + return data, [act.keep_bmp_trailer()] out = bytearray(data[:end]) out[2:6] = struct.pack(" int | None: return None -def inspect_gif(data: bytes) -> tuple[bool, bool, list[str]]: +def inspect_gif(data: bytes) -> MetadataScan: """Inspect GIF comment/application extensions for AI/C2PA markers.""" findings: list[str] = [] if data[:6] not in GIF_SIGS: - return False, False, ["not a GIF"] + return False, False, ["not a GIF"], [] n = len(data) has_c2pa = False has_ai = False @@ -1095,16 +1112,16 @@ def inspect_gif(data: bytes) -> tuple[bool, bool, list[str]]: if whole and not has_c2pa: has_c2pa = True findings.append(f"byte-scan C2PA markers: {', '.join(whole[:6])}") - return has_c2pa, has_ai or has_c2pa, findings + return has_c2pa, has_ai or has_c2pa, findings, [] -def strip_gif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[str]]: +def strip_gif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[Action]]: """Strip GIF comment + XMP/unknown application extensions (in-place, offsets preserved).""" if data[:6] not in GIF_SIGS: raise ValueError("not GIF") n = len(data) out = bytearray(data[:6]) - actions: list[str] = [] + actions: list[Action] = [] pos = 6 while pos + 1 <= n: block = data[pos] @@ -1131,7 +1148,7 @@ def strip_gif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, l name = "application" drop = strip_all_metadata or marker_hit if drop: - actions.append(f"drop GIF {name} extension") + actions.append(act.drop_gif_extension(name)) else: out.extend(data[pos:end]) pos = end @@ -1147,7 +1164,11 @@ def strip_gif(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, l pos += 1 if not actions: - actions.append("no GIF metadata blocks removed (already clean or none matched)") + actions.append( + act.nothing_removed( + "gif", "no GIF metadata blocks removed (already clean or none matched)" + ) + ) return bytes(out), actions @@ -1374,17 +1395,18 @@ def _tiff_entry_payload(data: bytes, ent: dict[str, Any]) -> bytes | None: return data[vo : vo + ent["byte_size"]] -def inspect_tiff(data: bytes) -> tuple[bool, bool, list[str]]: +def inspect_tiff(data: bytes) -> MetadataScan: """Walk the IFD chains (classic or BigTIFF) and report metadata tags.""" findings: list[str] = [] + notes: list[str] = [] has_c2pa = False has_ai = False try: _bo, _big, ifds = _parse_tiff_ifds(data) except Exception: - return False, False, ["not a valid TIFF"] + return False, False, ["not a valid TIFF"], [] if not ifds: - return False, False, ["TIFF with no image file directories"] + return False, False, ["TIFF with no image file directories"], [] for _off, ifd in sorted(ifds.items()): for ent in ifd["entries"]: tag = ent["tag"] @@ -1400,14 +1422,14 @@ def inspect_tiff(data: bytes) -> tuple[bool, bool, list[str]]: name = _TIFF_META_TAG_NAMES.get(tag) if name: label = "sub-IFD" if tag in (34665, 34853, 40965) else "tag" - findings.append(f"TIFF {label} {tag} ({name}) present") + notes.append(f"TIFF {label} {tag} ({name}) present") whole = _contains_any(data, C2PA_MARKERS) if whole and not has_c2pa: has_c2pa = True findings.append(f"byte-scan C2PA markers: {', '.join(whole[:6])}") - if not findings: - findings.append("no TIFF metadata tags found") - return has_c2pa, has_ai or has_c2pa, findings + if not findings and not notes: + notes.append("no TIFF metadata tags found") + return has_c2pa, has_ai or has_c2pa, findings, notes def _collect_tiff_sub_ifd_drops( @@ -1433,7 +1455,7 @@ def _collect_tiff_sub_ifd_drops( drop_ranges.append((vo, vo + vs)) -def strip_tiff(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[str]]: +def strip_tiff(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, list[Action]]: """Drop TIFF metadata tags (classic or BigTIFF) without moving referenced data.""" bo, bigtiff, ifds = _parse_tiff_ifds(data) if not ifds: @@ -1443,7 +1465,7 @@ def strip_tiff(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, off_fmt = bo + ("Q" if bigtiff else "I") count_len = 8 if bigtiff else 2 entry_size = 20 if bigtiff else 12 - actions: list[str] = [] + actions: list[Action] = [] kept: dict[int, list[dict[str, Any]]] = {} drop_ranges: list[tuple[int, int]] = [] drop_ifd_ranges: list[tuple[int, int]] = [] @@ -1476,9 +1498,7 @@ def strip_tiff(data: bytes, *, strip_all_metadata: bool = True) -> tuple[bytes, keep_here.append(ent) continue name = _TIFF_META_TAG_NAMES.get(tag) - actions.append( - f"drop TIFF tag {tag} ({name})" if name else f"drop TIFF tag {tag} (AI markers)" - ) + actions.append(act.drop_tiff_tag(tag, name)) if tag in (34665, 34853): ptr = struct.unpack(off_fmt, ent["value"][:off_len])[0] _collect_tiff_sub_ifd_drops( @@ -1575,7 +1595,11 @@ def _mark_reachable(off: int) -> None: ) if not actions: - actions.append("no TIFF metadata tags removed (already clean or none matched)") + actions.append( + act.nothing_removed( + "tiff", "no TIFF metadata tags removed (already clean or none matched)" + ) + ) return bytes(out), actions @@ -1627,14 +1651,14 @@ def clean_image( output_limit=OPTIONAL_TOOL_OUTPUT_LIMIT, ) if result.returncode == 0 and not (result.stdout_truncated or result.stderr_truncated): - actions.append("exiftool -all= pass") + actions.append(act.exiftool_strip()) else: detail = (result.stderr_text or result.stdout_text).strip()[:300] if result.stdout_truncated or result.stderr_truncated: detail = "output exceeded safety limit" - actions.append(f"exiftool failed (rc={result.returncode}): {detail}") + actions.append(act.exiftool_failed(result.returncode, detail)) except Exception as error: - actions.append(f"exiftool failed: {error}") + actions.append(act.exiftool_failed(None, str(error))) synthid_removal: dict[str, Any] | None = None if remove_synthid: @@ -1662,21 +1686,17 @@ def clean_image( "bytes_out": dest.stat().st_size, "note": "seed-independent mid-frequency band suppression (best-effort)", } - actions.append( - f"SynthID band removal: strength={synthid_strength} (seed-independent DCT suppression)" - ) + actions.append(act.synthid_band_removal(synthid_strength)) if wmct_marker: if fmt != "png": - actions.append("wmCt replacement marker skipped: PNG output only") + actions.append(act.wmct_marker_skipped("PNG output only")) else: # Read dest (post-strip/exiftool/synthid), inject the marker, and # write back so the truthful marker survives every prior pass. marked = add_wmct_marker(dest.read_bytes()) atomic_write_bytes(dest, marked) - actions.append( - "wmCt replacement marker written (strip-without-replacement remains the default)" - ) + actions.append(act.wmct_marker_written()) wmct_marker_present = fmt == "png" else: wmct_marker_present = False @@ -1686,7 +1706,7 @@ def clean_image( "input": str(path), "output": str(dest), "format": fmt, - "actions": actions, + **report_actions(actions), "bytes_in": len(data), "bytes_out": dest.stat().st_size, "still_has_c2pa": after.has_c2pa, diff --git a/skills/remove-ai-marks/scripts/inspect_file.py b/skills/remove-ai-marks/scripts/inspect_file.py index b7746ff..2778c21 100644 --- a/skills/remove-ai-marks/scripts/inspect_file.py +++ b/skills/remove-ai-marks/scripts/inspect_file.py @@ -172,7 +172,7 @@ def _inspect_asset( "kind": "container", "path": str(path), **report.to_dict(), - "suspicious": report.has_c2pa or report.has_ai_metadata, + "suspicious": report.has_c2pa or report.has_ai_metadata or report.layer_a_total > 0, }, None, ) @@ -213,6 +213,8 @@ def _inspect_single(path: Path, args) -> dict: print(f"AI metadata: {result.get('has_ai_metadata')}") for finding in result.get("findings", []): print(f" - {finding}") + for note in result.get("notes", []): + print(f" note: {note}") if kind == "image" and result.get("soft_binding", {}).get("soft_binding", {}).get("found"): print(f"Soft binding: {result['soft_binding']['soft_binding']['labels']}") return result diff --git a/skills/remove-ai-marks/scripts/inspect_image.py b/skills/remove-ai-marks/scripts/inspect_image.py index 750f218..d6dc5f6 100755 --- a/skills/remove-ai-marks/scripts/inspect_image.py +++ b/skills/remove-ai-marks/scripts/inspect_image.py @@ -41,6 +41,10 @@ def main() -> int: print("Findings:") for f in report.findings: print(f" - [{classify_finding_confidence(f)}] {f}") + if report.notes: + print("Notes:") + for note in report.notes: + print(f" - {note}") ct = report.tools.get("c2patool") or {} print(f"c2patool: {'yes' if ct.get('available') else 'no'}") et = report.tools.get("exiftool") or {} diff --git a/skills/remove-ai-marks/scripts/morphomod.py b/skills/remove-ai-marks/scripts/morphomod.py index 01421c8..484e2f7 100644 --- a/skills/remove-ai-marks/scripts/morphomod.py +++ b/skills/remove-ai-marks/scripts/morphomod.py @@ -34,6 +34,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) import external_command +import pipeline_actions as act from common import ( atomic_write_bytes, eprint, @@ -42,6 +43,7 @@ validate_output_path, ) from image_meta import detect_format +from pipeline_actions import Action, report_actions from png_chunks import iter_png_chunks DEFAULT_DILATION_RADIUS = 3 @@ -930,7 +932,7 @@ def remove_visible( _validate_visible_paths(path, dest, plan.mask_path, plan.mask_output) data = _read_bounded(path) fmt = detect_format(data) - actions: list[str] = [] + actions: list[Action] = [] has_source = any( source is not None for source in (plan.mask_path, plan.box, plan.detect_command) @@ -942,10 +944,7 @@ def remove_visible( "output": None, "format": fmt, "backend": plan.backend, - "actions": [ - "supply --mask, --box, or --detect-command", - "then refine/fill holes, dilate d=3, inpaint, restore original outside mask", - ], + **report_actions([act.visible_needs_source(), act.visible_plan()]), "note": "No blind segmenter is bundled; no image bytes were changed.", } raster = decode_to_raster(data, fmt) @@ -1008,17 +1007,16 @@ def remove_visible( mask_label = None # frictionless: effective mask kept in memory only actions.extend( [ - f"mask source: {source}", - f"fill holes + dilate radius={plan.dilation_radius}: {initial.marked}->{refined.marked} pixels", - f"effective mask: {initial.marked}->{refined.marked} pixels" - + (f" (published {mask_label})" if mask_label else " (not published)"), + act.mask_source(source), + act.refine_mask(plan.dilation_radius, initial.marked, refined.marked), + act.effective_mask(initial.marked, refined.marked, mask_label), ] ) if plan.backend == "print-plan": status = "mask-ready" output = None - actions.append("no inpainting run (print-plan backend)") + actions.append(act.inpaint_skipped()) elif plan.backend == "texture": assert raster is not None assert dest is not None @@ -1026,7 +1024,7 @@ def remove_visible( atomic_write_bytes(dest, encode_png(restored)) status, output = "completed", str(dest) actions.append( - f"texture-patch inpaint source=({match.x},{match.y},{match.width},{match.height}) edge_mse={match.score:.2f}" + act.inpaint_texture(match.x, match.y, match.width, match.height, match.score) ) elif plan.backend == "simple": assert raster is not None @@ -1035,7 +1033,7 @@ def remove_visible( restored = composite(raster, filled, refined) atomic_write_bytes(dest, encode_png(restored)) status, output = "completed", str(dest) - actions.append("nearest-boundary inpaint + restore (uniform-background fallback)") + actions.append(act.inpaint_simple()) else: assert plan.backend == "external" assert plan.command is not None @@ -1066,7 +1064,7 @@ def remove_visible( f"does not match source {(raster.width, raster.height, raster.channels)}" ) atomic_write_bytes(dest, encode_png(composite(raster, inpainted, refined))) - actions.append("external inpaint + stdlib restore outside mask") + actions.append(act.inpaint_external()) status, output = "completed", str(dest) return { @@ -1079,7 +1077,7 @@ def remove_visible( "initial_mask_pixels": initial.marked, "refined_mask_pixels": refined.marked, "dilation_radius": plan.dilation_radius, - "actions": actions, + **report_actions(actions), "note": ( "MorphoMod-inspired pipeline; CVPR paper metrics are not this run's metrics. Inspect output fidelity and residual marks manually." ), diff --git a/skills/remove-ai-marks/scripts/optional_deps.py b/skills/remove-ai-marks/scripts/optional_deps.py index d640ccb..dfbf513 100644 --- a/skills/remove-ai-marks/scripts/optional_deps.py +++ b/skills/remove-ai-marks/scripts/optional_deps.py @@ -65,7 +65,6 @@ def check_optional(extra: str) -> BackendAvailability: "quality": "skimage", "ai": "torch", "provenance": "c2pa", - "tui": "textual", } pkg = _checkers.get(extra) if pkg is None: @@ -106,12 +105,8 @@ def has_provenance() -> bool: return check_optional("provenance").available -def has_tui() -> bool: - return check_optional("tui").available - - #: Every extra `check_optional` knows about, for capability reporting. -KNOWN_EXTRAS = ("visible", "quality", "ai", "provenance", "tui") +KNOWN_EXTRAS = ("visible", "quality", "ai", "provenance") def extras_status() -> dict[str, dict[str, object]]: diff --git a/skills/remove-ai-marks/scripts/pipeline_actions.py b/skills/remove-ai-marks/scripts/pipeline_actions.py new file mode 100644 index 0000000..5e03893 --- /dev/null +++ b/skills/remove-ai-marks/scripts/pipeline_actions.py @@ -0,0 +1,794 @@ +#!/usr/bin/env python3 +"""What a clean did to a file, as a stable machine code plus the human line. + +Every stripper and cleaner reports its steps as :class:`Action` values. A +step's ``code`` names *what kind* of step it was and never changes wording; +``params`` carry its counts and names (a part, a chunk, a field, a byte +count); ``text`` is the line the CLI has always printed. ``effect`` says +whether the step changed the output, was only informational, or warns that a +tool failed or fell back. + +A result payload carries the same steps twice, from one list, so the two can +never disagree: + +* ``actions`` -- the human lines, unchanged from before codes existed; +* ``action_details`` -- ``{"code", "effect", "text", "params"}`` per step, in + the same order, for any consumer that must not parse the lines. + +Each code has exactly one constructor (a few codes have more than one, where +the historical wording differs), so a call site never pairs a code with a +text by hand. ``ActionCode`` is the complete list of codes; the TUI keeps a +sentence for every one of them and a test holds it to that. +""" + +from __future__ import annotations + +from collections.abc import Iterable, Mapping, Sequence +from dataclasses import dataclass, field +from enum import Enum +from types import MappingProxyType +from typing import Any + + +class Effect(str, Enum): + """What a step did to the output.""" + + #: The output differs from the input because of this step. + CHANGE = "change" + #: Nothing was removed, or the step was neutral bookkeeping. + INFO = "info" + #: A tool failed, fell back or was missing; the result may be weaker. + WARNING = "warning" + #: The file could not be cleaned at all. + ERROR = "error" + + +class ActionCode(str, Enum): + """Every step a clean can report. Values are the wire codes.""" + + NOTHING_REMOVED = "nothing_removed" + # Raster images + DROP_PNG_CHUNK = "drop_png_chunk" + DROP_JPEG_SEGMENT = "drop_jpeg_segment" + PRESERVE_JPEG_SCAN = "preserve_jpeg_scan" + DROP_WEBP_CHUNK = "drop_webp_chunk" + DROP_GIF_EXTENSION = "drop_gif_extension" + DROP_TIFF_TAG = "drop_tiff_tag" + DROP_BMP_TRAILER = "drop_bmp_trailer" + KEEP_BMP_TRAILER = "keep_bmp_trailer" + BMP_UNPARSED = "bmp_unparsed" + NEUTRALIZE_BOX = "neutralize_box" + ZERO_PAYLOAD = "zero_payload" + NEUTRALIZE_TOKENS = "neutralize_tokens" + EXIFTOOL_STRIP = "exiftool_strip" + SYNTHID_BAND_REMOVAL = "synthid_band_removal" + WMCT_MARKER_WRITTEN = "wmct_marker_written" + WMCT_MARKER_SKIPPED = "wmct_marker_skipped" + # Text-based containers + DROP_FRONTMATTER_KEY = "drop_frontmatter_key" + DROP_EMPTY_FRONTMATTER = "drop_empty_frontmatter" + CLEAN_DATA_URI = "clean_data_uri" + DROP_HTML_META = "drop_html_meta" + DROP_JSON_LD = "drop_json_ld" + DROP_DATA_AI_ATTRIBUTES = "drop_data_ai_attributes" + DROP_SVG_METADATA = "drop_svg_metadata" + DROP_SVG_XMP = "drop_svg_xmp" + DROP_SVG_COMMENT = "drop_svg_comment" + DROP_SVG_GENERATOR_ATTRIBUTES = "drop_svg_generator_attributes" + # Zip containers + CLEAN_EMBEDDED_MEDIA = "clean_embedded_media" + CLEAN_PART = "clean_part" + DROP_PART = "drop_part" + SCRUB_FIELD = "scrub_field" + DROP_CONTENT_TYPE_OVERRIDES = "drop_content_type_overrides" + PRUNE_RELATIONSHIPS = "prune_relationships" + PRUNE_MANIFEST = "prune_manifest" + DROP_GENERATOR_META = "drop_generator_meta" + SCRUB_CREATOR = "scrub_creator" + DROP_OPF_META = "drop_opf_meta" + LAYER_A_TEXT = "layer_a_text" + # PDF + EXIFTOOL_RUN = "exiftool_run" + TOOL_FAILED = "tool_failed" + TOOL_MISSING = "tool_missing" + TRY_FALLBACK = "try_fallback" + PDF_ENCRYPTED = "pdf_encrypted" + PDF_DECRYPTED = "pdf_decrypted" + DROP_PDF_METADATA = "drop_pdf_metadata" + PDF_REWRITTEN = "pdf_rewritten" + PDF_REWRITE_FAILED = "pdf_rewrite_failed" + PDF_COPIED_UNCHANGED = "pdf_copied_unchanged" + C2PATOOL_HINT = "c2patool_hint" + # Visible-mark removal + VISIBLE_NEEDS_SOURCE = "visible_needs_source" + VISIBLE_PLAN = "visible_plan" + MASK_SOURCE = "mask_source" + REFINE_MASK = "refine_mask" + EFFECTIVE_MASK = "effective_mask" + INPAINT_SKIPPED = "inpaint_skipped" + INPAINT = "inpaint" + # Visible-mark dry run + PLAN_LOCALIZE = "plan_localize" + PLAN_REFINE_MASK = "plan_refine_mask" + PLAN_INPAINT = "plan_inpaint" + PLAN_STRIP_METADATA = "plan_strip_metadata" + PLAN_DEGRADE = "plan_degrade" + PLAN_PUBLISH = "plan_publish" + # Whole-file failure + FAILED = "failed" + + +@dataclass(frozen=True) +class Action: + """One step of a clean. Build it with a constructor below, never directly.""" + + code: ActionCode + effect: Effect + text: str + params: Mapping[str, Any] = field(default_factory=dict) + + def __post_init__(self) -> None: + object.__setattr__(self, "params", MappingProxyType(dict(self.params))) + + @property + def changed(self) -> bool: + return self.effect is Effect.CHANGE + + def to_dict(self) -> dict[str, Any]: + return { + "code": self.code.value, + "effect": self.effect.value, + "text": self.text, + "params": {key: _wire(value) for key, value in self.params.items()}, + } + + +def _wire(value: Any) -> Any: + if isinstance(value, Action): + return value.to_dict() + if isinstance(value, (list, tuple)): + return [_wire(item) for item in value] + return value + + +def report_actions(actions: Sequence[Action]) -> dict[str, list[Any]]: + """The two payload fields for *actions*: ``actions`` and ``action_details``.""" + return { + "actions": [action.text for action in actions], + "action_details": [action.to_dict() for action in actions], + } + + +def retarget_report(report: dict[str, Any], old: str, new: str) -> None: + """Rename *old* to *new* in a report's steps, in both fields. + + For a path a step named while it was staged in a temporary directory and + that has since been published elsewhere. + """ + report["actions"] = [text.replace(old, new) for text in report["actions"]] + report["action_details"] = [_retarget(detail, old, new) for detail in report["action_details"]] + + +def _retarget(value: Any, old: str, new: str) -> Any: + if isinstance(value, str): + return value.replace(old, new) + if isinstance(value, list): + return [_retarget(item, old, new) for item in value] + if isinstance(value, dict): + return {key: _retarget(item, old, new) for key, item in value.items()} + return value + + +def any_change(actions: Iterable[Action]) -> bool: + return any(action.changed for action in actions) + + +def _act(code: ActionCode, effect: Effect, text: str, **params: Any) -> Action: + return Action(code, effect, text, {k: v for k, v in params.items() if v is not None}) + + +def _summary(actions: Sequence[Action]) -> str: + """The legacy parenthetical for a nested clean: its first two lines.""" + return ", ".join(action.text for action in actions[:2]) + + +# --- shared ------------------------------------------------------------------ + + +def nothing_removed(fmt: str, text: str) -> Action: + """A cleaner found nothing to remove. *text* is that cleaner's own wording.""" + return _act(ActionCode.NOTHING_REMOVED, Effect.INFO, text, format=fmt) + + +def layer_a_text(removed: int, replaced: int) -> Action: + return _act( + ActionCode.LAYER_A_TEXT, + Effect.CHANGE, + f"layer A text: removed={removed} replaced={replaced}", + removed=removed, + replaced=replaced, + ) + + +def failed(error: object) -> Action: + return _act(ActionCode.FAILED, Effect.ERROR, f"error: {error}", error=str(error)) + + +# --- raster images ----------------------------------------------------------- + + +def drop_png_chunk(chunk: str, *, c2pa_in_payload: bool = False) -> Action: + reason = "C2PA marker in payload" if c2pa_in_payload else None + text = f"drop chunk {chunk}" + (f" ({reason})" if reason else "") + return _act(ActionCode.DROP_PNG_CHUNK, Effect.CHANGE, text, chunk=chunk, reason=reason) + + +def drop_jpeg_segment(segment: str, *, reason: str | None = None) -> Action: + """*segment* is ``APPn`` or ``COM``.""" + name = "COM comment" if segment == "COM" else segment + text = f"drop {name}" + (f" ({reason})" if reason else "") + return _act(ActionCode.DROP_JPEG_SEGMENT, Effect.CHANGE, text, segment=segment, reason=reason) + + +def preserve_jpeg_scan() -> Action: + return _act( + ActionCode.PRESERVE_JPEG_SCAN, Effect.INFO, "preserved entropy-coded scan through EOI" + ) + + +def drop_webp_chunk(chunk: str) -> Action: + return _act(ActionCode.DROP_WEBP_CHUNK, Effect.CHANGE, f"drop WebP chunk {chunk}", chunk=chunk) + + +def drop_gif_extension(extension: str) -> Action: + """*extension* is ``comment``, ``XMP application``, ``application`` or ``extension``.""" + return _act( + ActionCode.DROP_GIF_EXTENSION, + Effect.CHANGE, + f"drop GIF {extension} extension", + extension=extension, + ) + + +def drop_tiff_tag(tag: int, name: str | None) -> Action: + """*name* is the tag's metadata name, or None when AI markers alone dropped it.""" + text = f"drop TIFF tag {tag} ({name})" if name else f"drop TIFF tag {tag} (AI markers)" + return _act(ActionCode.DROP_TIFF_TAG, Effect.CHANGE, text, tag=tag, name=name) + + +def drop_bmp_trailer(size: int, markers: Sequence[str]) -> Action: + reason = f" ({', '.join(markers)})" if markers else "" + return _act( + ActionCode.DROP_BMP_TRAILER, + Effect.CHANGE, + f"drop {size} BMP trailing byte(s){reason}", + bytes=size, + markers=list(markers), + ) + + +def keep_bmp_trailer() -> Action: + return _act( + ActionCode.KEEP_BMP_TRAILER, Effect.INFO, "BMP trailing bytes kept (keep-non-ai-metadata)" + ) + + +def bmp_unparsed() -> Action: + return _act( + ActionCode.BMP_UNPARSED, Effect.WARNING, "BMP header not fully parsed; left unchanged" + ) + + +def neutralize_box(box: str, size: int) -> Action: + return _act( + ActionCode.NEUTRALIZE_BOX, + Effect.CHANGE, + f"neutralized '{box}' box -> free (zeroed {size} payload bytes)", + box=box, + bytes=size, + ) + + +def zero_uuid_payload(size: int) -> Action: + return _act( + ActionCode.ZERO_PAYLOAD, + Effect.CHANGE, + f"zeroed XMP uuid box payload ({size} bytes, offsets preserved)", + target="XMP uuid box", + bytes=size, + ) + + +def zero_item_payload(item: str, size: int) -> Action: + """*item* is ``Exif item`` or ``XMP item``.""" + return _act( + ActionCode.ZERO_PAYLOAD, + Effect.CHANGE, + f"zeroed entire {item} payload ({size} bytes, offsets preserved)", + target=item, + bytes=size, + ) + + +def neutralize_tokens(target: str, tokens: Sequence[str]) -> Action: + return _act( + ActionCode.NEUTRALIZE_TOKENS, + Effect.CHANGE, + f"neutralized AI tokens in {target}: {', '.join(tokens)}", + target=target, + tokens=list(tokens), + ) + + +def exiftool_strip() -> Action: + return _act(ActionCode.EXIFTOOL_STRIP, Effect.CHANGE, "exiftool -all= pass") + + +def synthid_band_removal(strength: float) -> Action: + return _act( + ActionCode.SYNTHID_BAND_REMOVAL, + Effect.CHANGE, + f"SynthID band removal: strength={strength} (seed-independent DCT suppression)", + strength=strength, + ) + + +def wmct_marker_written() -> Action: + return _act( + ActionCode.WMCT_MARKER_WRITTEN, + Effect.CHANGE, + "wmCt replacement marker written (strip-without-replacement remains the default)", + ) + + +def wmct_marker_skipped(reason: str) -> Action: + return _act( + ActionCode.WMCT_MARKER_SKIPPED, + Effect.WARNING, + f"wmCt replacement marker skipped: {reason}", + reason=reason, + ) + + +# --- text-based containers --------------------------------------------------- + + +def drop_frontmatter_key(key: str, *, value_hit: bool = False) -> Action: + text = ( + f"drop frontmatter key (value hit): {key}" if value_hit else f"drop frontmatter key: {key}" + ) + return _act(ActionCode.DROP_FRONTMATTER_KEY, Effect.CHANGE, text, key=key, value_hit=value_hit) + + +def drop_empty_frontmatter() -> Action: + return _act(ActionCode.DROP_EMPTY_FRONTMATTER, Effect.CHANGE, "removed empty frontmatter block") + + +def clean_data_uri(mime: str, sub_actions: Sequence[Action]) -> Action: + return _act( + ActionCode.CLEAN_DATA_URI, + Effect.CHANGE, + f"cleaned embedded data:image/{mime} ({_summary(sub_actions)})", + mime=f"image/{mime}", + actions=list(sub_actions), + ) + + +def drop_html_meta(tag: str) -> Action: + return _act(ActionCode.DROP_HTML_META, Effect.CHANGE, f"drop meta: {tag[:80]}", tag=tag[:80]) + + +def drop_json_ld() -> Action: + return _act(ActionCode.DROP_JSON_LD, Effect.CHANGE, "drop json-ld provenance-like script") + + +def drop_data_ai_attributes(count: int) -> Action: + return _act( + ActionCode.DROP_DATA_AI_ATTRIBUTES, + Effect.CHANGE, + f"drop data-ai* attributes x{count}", + count=count, + ) + + +def drop_svg_metadata(count: int) -> Action: + return _act( + ActionCode.DROP_SVG_METADATA, Effect.CHANGE, f"drop x{count}", count=count + ) + + +def drop_svg_xmp(count: int) -> Action: + return _act(ActionCode.DROP_SVG_XMP, Effect.CHANGE, f"drop xmpmeta x{count}", count=count) + + +def drop_svg_comment() -> Action: + return _act(ActionCode.DROP_SVG_COMMENT, Effect.CHANGE, "drop SVG comment with AI markers") + + +def drop_svg_generator_attributes(count: int) -> Action: + return _act( + ActionCode.DROP_SVG_GENERATOR_ATTRIBUTES, + Effect.CHANGE, + f"drop generator-like attrs x{count}", + count=count, + ) + + +# --- zip containers ---------------------------------------------------------- + + +def clean_embedded_media(part: str, sub_actions: Sequence[Action]) -> Action: + return _act( + ActionCode.CLEAN_EMBEDDED_MEDIA, + Effect.CHANGE, + f"clean embedded media in {part} ({_summary(sub_actions)})", + part=part, + actions=list(sub_actions), + ) + + +def clean_part(part: str, sub_actions: Sequence[Action]) -> Action: + """Steps a nested cleaner took inside one part (an EPUB's XHTML or OPF).""" + return _act( + ActionCode.CLEAN_PART, + Effect.CHANGE if any_change(sub_actions) else Effect.INFO, + f"{part}: {_summary(sub_actions)}", + part=part, + actions=list(sub_actions), + ) + + +def drop_part(part: str, *, markers: bool = False) -> Action: + """*markers*: the part was dropped because it carries AI/C2PA markers.""" + reason = "AI/C2PA markers" if markers else None + text = f"drop part {part}" + (f" ({reason})" if reason else "") + return _act(ActionCode.DROP_PART, Effect.CHANGE, text, part=part, reason=reason) + + +def scrub_field(part: str, field_name: str) -> Action: + return _act( + ActionCode.SCRUB_FIELD, + Effect.CHANGE, + f"scrub {part} field {field_name}", + part=part, + field=field_name, + ) + + +def scrub_opf_field(field_name: str) -> Action: + return _act( + ActionCode.SCRUB_FIELD, + Effect.CHANGE, + f"scrub {field_name} (AI vendor name)", + field=field_name, + reason="AI vendor name", + ) + + +def drop_content_type_overrides(target: str, count: int) -> Action: + """*target* is ``customXml`` (every customXml part) or ``custom.xml``.""" + text = ( + f"drop Content_Types customXml overrides x{count}" + if target == "customXml" + else f"drop Content_Types custom.xml override x{count}" + ) + return _act( + ActionCode.DROP_CONTENT_TYPE_OVERRIDES, Effect.CHANGE, text, target=target, count=count + ) + + +def prune_relationships(part: str, count: int) -> Action: + return _act( + ActionCode.PRUNE_RELATIONSHIPS, + Effect.CHANGE, + f"prune dangling relationships x{count} in {part}", + part=part, + count=count, + ) + + +def prune_odf_manifest(count: int) -> Action: + return _act( + ActionCode.PRUNE_MANIFEST, + Effect.CHANGE, + f"drop manifest entries x{count}", + manifest="ODF", + count=count, + ) + + +def prune_opf_manifest(count: int) -> Action: + return _act( + ActionCode.PRUNE_MANIFEST, + Effect.CHANGE, + f"prune OPF manifest entries x{count}", + manifest="OPF", + count=count, + ) + + +def drop_generator_meta() -> Action: + return _act( + ActionCode.DROP_GENERATOR_META, + Effect.CHANGE, + "drop meta:generator", + field="meta:generator", + ) + + +def scrub_creator() -> Action: + return _act( + ActionCode.SCRUB_CREATOR, Effect.CHANGE, "scrub creator-like meta", field="dc:creator" + ) + + +def drop_opf_meta() -> Action: + return _act(ActionCode.DROP_OPF_META, Effect.CHANGE, "drop OPF meta tag") + + +# --- PDF --------------------------------------------------------------------- + + +def exiftool_run(returncode: int) -> Action: + """exiftool ran; a nonzero exit means it did not strip cleanly.""" + return _act( + ActionCode.EXIFTOOL_RUN, + Effect.CHANGE if returncode == 0 else Effect.WARNING, + f"exiftool -all= (rc={returncode})", + returncode=returncode, + ) + + +def tool_failed( + tool: str, + text: str, + *, + returncode: int | None = None, + detail: str | None = None, + fallback: str | None = None, +) -> Action: + """*tool* failed. *text* is the historical line for that failure.""" + return _act( + ActionCode.TOOL_FAILED, + Effect.WARNING, + text, + tool=tool, + returncode=returncode, + detail=detail, + fallback=fallback, + ) + + +def exiftool_failed(returncode: int | None, detail: str) -> Action: + """The raster exiftool pass failed.""" + text = ( + f"exiftool failed (rc={returncode}): {detail}" + if returncode is not None + else f"exiftool failed: {detail}" + ) + return tool_failed("exiftool", text, returncode=returncode, detail=detail) + + +def tool_missing(tool: str, text: str) -> Action: + return _act(ActionCode.TOOL_MISSING, Effect.WARNING, text, tool=tool) + + +def try_fallback(tool: str) -> Action: + return _act(ActionCode.TRY_FALLBACK, Effect.INFO, f"trying {tool} fallback", tool=tool) + + +def pdf_encrypted() -> Action: + return _act( + ActionCode.PDF_ENCRYPTED, + Effect.WARNING, + "encrypted PDF (password required); copied as-is", + ) + + +def pdf_decrypted() -> Action: + return _act(ActionCode.PDF_DECRYPTED, Effect.INFO, "decrypted with empty password") + + +def drop_pdf_page_metadata(pages: int) -> Action: + return _act( + ActionCode.DROP_PDF_METADATA, + Effect.CHANGE, + f"pypdf: drop per-page /Metadata x{pages}", + target="page metadata", + pages=pages, + ) + + +def drop_pdf_docinfo() -> Action: + return _act( + ActionCode.DROP_PDF_METADATA, + Effect.CHANGE, + "pypdf: drop document info dictionary", + target="document info dictionary", + ) + + +def drop_pdf_catalog_xmp() -> Action: + return _act( + ActionCode.DROP_PDF_METADATA, + Effect.CHANGE, + "pypdf: drop catalog XMP packet", + target="catalog XMP packet", + ) + + +def pdf_rewritten_pypdf() -> Action: + return _act( + ActionCode.PDF_REWRITTEN, + Effect.CHANGE, + "pypdf: cloned full document graph; removed docinfo/XMP", + tool="pypdf", + ) + + +def pdf_rewritten_qpdf(returncode: int) -> Action: + return _act( + ActionCode.PDF_REWRITTEN, + Effect.CHANGE, + f"qpdf --linearize structural rewrite (rc={returncode})", + tool="qpdf", + returncode=returncode, + ) + + +def pdf_rewrite_failed( + text: str, *, returncode: int | None = None, detail: str | None = None +) -> Action: + """No structural rewrite: the old metadata bytes may remain recoverable.""" + return _act( + ActionCode.PDF_REWRITE_FAILED, + Effect.WARNING, + text, + tool="qpdf", + returncode=returncode, + detail=detail, + ) + + +def pdf_copied_unchanged() -> Action: + return _act( + ActionCode.PDF_COPIED_UNCHANGED, + Effect.WARNING, + "no structural PDF cleaner succeeded; copied unchanged", + ) + + +def c2patool_hint() -> Action: + return _act( + ActionCode.C2PATOOL_HINT, + Effect.INFO, + "c2patool available for inspect; strip via exiftool/re-export", + ) + + +# --- visible-mark removal ---------------------------------------------------- + + +def visible_needs_source() -> Action: + return _act( + ActionCode.VISIBLE_NEEDS_SOURCE, Effect.INFO, "supply --mask, --box, or --detect-command" + ) + + +def visible_plan() -> Action: + return _act( + ActionCode.VISIBLE_PLAN, + Effect.INFO, + "then refine/fill holes, dilate d=3, inpaint, restore original outside mask", + ) + + +def mask_source(source: str) -> Action: + return _act(ActionCode.MASK_SOURCE, Effect.INFO, f"mask source: {source}", source=source) + + +def refine_mask(dilation_radius: int, before: int, after: int) -> Action: + return _act( + ActionCode.REFINE_MASK, + Effect.INFO, + f"fill holes + dilate radius={dilation_radius}: {before}->{after} pixels", + dilation_radius=dilation_radius, + pixels_before=before, + pixels_after=after, + ) + + +def effective_mask(before: int, after: int, published: str | None) -> Action: + """*published* is the mask file written, or None when it stayed in memory.""" + where = f" (published {published})" if published else " (not published)" + return _act( + ActionCode.EFFECTIVE_MASK, + Effect.INFO, + f"effective mask: {before}->{after} pixels{where}", + pixels_before=before, + pixels_after=after, + published=published, + ) + + +def inpaint_skipped() -> Action: + return _act(ActionCode.INPAINT_SKIPPED, Effect.INFO, "no inpainting run (print-plan backend)") + + +def inpaint_texture(x: int, y: int, width: int, height: int, edge_mse: float) -> Action: + return _act( + ActionCode.INPAINT, + Effect.CHANGE, + f"texture-patch inpaint source=({x},{y},{width},{height}) edge_mse={edge_mse:.2f}", + backend="texture", + source_patch=[x, y, width, height], + edge_mse=round(edge_mse, 2), + ) + + +def inpaint_simple() -> Action: + return _act( + ActionCode.INPAINT, + Effect.CHANGE, + "nearest-boundary inpaint + restore (uniform-background fallback)", + backend="simple", + ) + + +def inpaint_external() -> Action: + return _act( + ActionCode.INPAINT, + Effect.CHANGE, + "external inpaint + stdlib restore outside mask", + backend="external", + ) + + +# --- visible-mark dry run ---------------------------------------------------- + + +def plan_localize(source: str) -> Action: + return _act( + ActionCode.PLAN_LOCALIZE, + Effect.INFO, + f"localize visible mark via {source}", + source=source, + ) + + +def plan_refine_mask(dilation_radius: int) -> Action: + return _act( + ActionCode.PLAN_REFINE_MASK, + Effect.INFO, + f"fill holes + dilate radius={dilation_radius}", + dilation_radius=dilation_radius, + ) + + +def plan_inpaint(backend: str) -> Action: + return _act( + ActionCode.PLAN_INPAINT, Effect.INFO, f"inpaint with {backend} backend", backend=backend + ) + + +def plan_strip_metadata() -> Action: + return _act(ActionCode.PLAN_STRIP_METADATA, Effect.INFO, "strip requested metadata") + + +def plan_degrade(strategy: str) -> Action: + return _act( + ActionCode.PLAN_DEGRADE, + Effect.INFO, + f"apply {strategy} degradation", + strategy=strategy, + ) + + +def plan_publish(mask: str, image: str) -> Action: + return _act( + ActionCode.PLAN_PUBLISH, + Effect.INFO, + f"publish mask to {mask} and image to {image}", + mask=mask, + image=image, + ) diff --git a/skills/remove-ai-marks/scripts/rewrite_text.py b/skills/remove-ai-marks/scripts/rewrite_text.py index 3055fd0..a817077 100644 --- a/skills/remove-ai-marks/scripts/rewrite_text.py +++ b/skills/remove-ai-marks/scripts/rewrite_text.py @@ -96,8 +96,13 @@ REASONING_EFFORTS = ("none", "low", "medium", "high", "off") #: Hosts that may receive document content without an explicit opt-in. LOOPBACK_HOSTS = frozenset({"localhost", "127.0.0.1", "::1"}) -#: Private alias kept for in-module call sites predating the export. -_LOOPBACK_HOSTS = LOOPBACK_HOSTS +#: Where Layer B looks for a backend when nothing names one: Ollama's port. +DEFAULT_BASE_URL = "http://127.0.0.1:11434" + + +def resolve_base_url(explicit: str | None) -> str: + """The base URL a run contacts: *explicit*, else the environment, else the default.""" + return explicit or os.environ.get("WATERMARKS_REWRITE_BASE_URL", DEFAULT_BASE_URL) class RewriteConfigurationError(ValueError): @@ -246,8 +251,7 @@ def live_from_environment( return cls( backend=resolved_backend, model=resolved_model, - base_url=base_url - or os.environ.get("WATERMARKS_REWRITE_BASE_URL", "http://127.0.0.1:11434"), + base_url=resolve_base_url(base_url), api_key=api_key or os.environ.get("WATERMARKS_REWRITE_API_KEY"), strength=strength, lang=lang if lang is not None else defaults.lang, @@ -339,7 +343,7 @@ def _check_remote(base_url: str, allow_remote: bool) -> None: f"error: rewrite base URL must be http(s), got scheme '{u.scheme}': {base_url}" ) host = u.hostname or "" - if host in _LOOPBACK_HOSTS: + if host in LOOPBACK_HOSTS: return if not allow_remote: raise SystemExit( @@ -915,7 +919,7 @@ def main() -> int: p.add_argument("--model", default=_env("WATERMARKS_REWRITE_MODEL")) p.add_argument( "--base-url", - default=_env("WATERMARKS_REWRITE_BASE_URL", "http://127.0.0.1:11434"), + default=_env("WATERMARKS_REWRITE_BASE_URL", DEFAULT_BASE_URL), ) p.add_argument( "--allow-remote", diff --git a/skills/remove-ai-marks/scripts/tui.py b/skills/remove-ai-marks/scripts/tui.py index 2a7d746..21f484b 100644 --- a/skills/remove-ai-marks/scripts/tui.py +++ b/skills/remove-ai-marks/scripts/tui.py @@ -1,25 +1,22 @@ #!/usr/bin/env python3 -"""wm-tui — interactive terminal UI over the watermark-remover pipeline. - -The loop this exists for is inspect -> clean -> re-inspect, with Layer A's exact -removal counts sitting next to Layer B's detector scores so "cleanly" is a -measured before/after rather than a claim. - -Design rules this module is held to (see ``tasks/tui-plan.md``): - -* It never builds a ``CleanPlan`` itself. Every run fills a ``CleanRequest`` - and goes through ``clean_request.plan_work`` and ``clean_file.run_clean_item`` - — the same seam, and the same refusals, as the CLI. -* It never speaks HTTP. Rewrites go through ``rewrite_text``; model discovery - goes through ``layer_b_discovery``, which goes through ``layer_b_http``. There - is no ``urllib`` import in this file, and a test enforces that. -* It never renders a best-effort result as verified. Layer A and Layer M are - Verifiable, Layer B and Layer V are Best-effort, soft binding is - Detection-only, and the badge follows the layer, not the outcome. -* It never displays or persists an API key. - -This module is deliberately one file: ``[tool.setuptools] packages`` is an -explicit list, so a subpackage would silently not ship in the wheel. +"""wm-tui: launch the interactive terminal UI over the watermark-remover pipeline. + +The UI itself is a TypeScript program on Bun (``tui/``); every decision about a +clean is made by the Python bridge it spawns (``tui_bridge.py``, contract in +``tui/PROTOCOL.md``). This module only does what must happen before a +full-screen program takes the terminal: + +* refuse a missing path the way ``wm`` does, instead of opening a UI whose only + content is an error; +* find ``bun`` and say how to get it when it is absent, rather than dying with + ``FileNotFoundError``; +* install the frontend's dependencies once, into a writable copy when the + package lives somewhere read-only (a wheel in site-packages); +* hand the frontend the interpreter it must spawn the bridge with, so the + bridge never runs on some other ``python`` from PATH. + +It imports nothing from the pipeline on purpose: this runs on every start, and +the bridge pays for the pipeline import once it is actually needed. """ from __future__ import annotations @@ -27,446 +24,230 @@ import argparse import json import os +import shutil +import stat +import subprocess import sys -from dataclasses import asdict, dataclass, field, replace +from importlib.metadata import PackageNotFoundError, version from pathlib import Path -from typing import get_args, get_type_hints - -sys.path.insert(0, str(Path(__file__).resolve().parent)) - -from asset_kind import SUPPORTED_EXTENSIONS -from batch_inputs import select_inputs -from clean_request import CleanRequest -from common import atomic_write_text -from optional_deps import check_optional - -TUI_EXTRA = "tui" - -#: Result classes from CONTEXT.md. The badge follows the *layer*, never the -#: outcome: a Layer B rewrite that "worked" is still best-effort. -VERIFIABLE = "Verifiable" -BEST_EFFORT = "Best-effort" -DETECTION_ONLY = "Detection-only" - -LAYER_RESULT_CLASS: dict[str, str] = { - "A": VERIFIABLE, - "M": VERIFIABLE, - "B": BEST_EFFORT, - "V": BEST_EFFORT, - "soft-binding": DETECTION_ONLY, - "synthid": BEST_EFFORT, - # Character perturbation adds noise to defeat a detector rather than - # removing a carrier that can be counted afterwards. There is nothing to - # verify, so it cannot be badged with the layer that strips zero-width. - "perturb": BEST_EFFORT, -} - -RESULT_CLASS_STYLE = { - VERIFIABLE: "bold green", - BEST_EFFORT: "bold yellow", - DETECTION_ONLY: "bold cyan", -} - -#: Default Layer B timeout used for the batch cost estimate when none is set. -DEFAULT_REWRITE_TIMEOUT = 120.0 - -#: Border title for the Layer B stream pane when no rewrite is running. An -#: always-visible box that only fills on one code path reads as broken, so it -#: says why it is empty rather than hiding. -IDLE_STREAM_TITLE = "Layer B stream — no live rewrite in this plan" - -#: Characters kept in the live Layer B stream view. A rewrite of a long -#: document would otherwise grow the widget without bound while it runs. -STREAM_VIEW_CHARS = 8000 - -#: Worst-case seconds above which a run is worth stopping to confirm. A single -#: file with one candidate sits under this; a batch, or any TSAPA search, does -#: not. A gate that fires on every rewrite is a gate nobody reads. -COST_CONFIRM_SECONDS = 300.0 - - -def layer_for_result(request: CleanRequest, kind: str) -> str: - """The layer that did the work on one asset. Follows the work, not the outcome. - - ``CleanRequest.visible_requested`` covers mask, box, dilation and the - external inpainter, but ``--degrade``, ``--morpho`` and - ``--remove-synthid`` are pixel-domain operations too: routing them to the - metadata layer badged a frequency-domain perturbation *Verifiable*, which - is exactly the claim this project does not make. - """ - if kind == "text": - if request.rewrite_strength: - return "B" - return "perturb" if request.char_perturb else "A" - if kind == "image": - if request.visible_requested() or request.degrade or request.morpho: - return "V" - if request.remove_synthid: - return "synthid" - return "M" - - -def result_class_for(layer: str) -> str: - """The honesty label for a layer. Unknown layers are never called verified.""" - return LAYER_RESULT_CLASS.get(layer, BEST_EFFORT) - - -def format_badge(layer: str) -> str: - """Rich markup badge naming the layer's result class.""" - label = result_class_for(layer) - return f"[{RESULT_CLASS_STYLE[label]}]{label}[/]" - - -def estimate_rewrite_seconds(request: CleanRequest, file_count: int) -> float: - """Worst-case wall clock for a batch that runs a live rewrite. - - Sequential execution (matching ``clean_file.main``) is what makes this - honest: files x candidates x per-call timeout is a real ceiling, not an - optimistic one. - """ - if request.rewrite_strength is None or file_count <= 0: - return 0.0 - timeout = request.rewrite_timeout or DEFAULT_REWRITE_TIMEOUT - candidates = max(1, request.rewrite_candidates or 1) - if request.rewrite_strength == "tsapa": - # TSAPA issues roughly population calls per generation, per file. - calls = max(1, request.tsapa_generations) * max(2, request.tsapa_population) - else: - calls = candidates - return float(file_count) * calls * timeout - - -def should_confirm_cost(request: CleanRequest, file_count: int) -> bool: - """Whether this run is expensive enough to stop and confirm.""" - return estimate_rewrite_seconds(request, file_count) > COST_CONFIRM_SECONDS - - -def format_duration(seconds: float) -> str: - if seconds < 90: - return f"{seconds:.0f}s" - if seconds < 5400: - return f"{seconds / 60:.0f}m" - return f"{seconds / 3600:.1f}h" +BUN_INSTALL_HINT = ( + "wm-tui needs Bun (https://bun.sh) to run its terminal frontend.\n" + "Install it with: curl -fsSL https://bun.sh/install | bash\n" + '(Windows: powershell -c "irm bun.sh/install.ps1 | iex"), then run wm-tui again.\n' + "The plain CLI (`wm FILE`) works without it." +) -@dataclass -class HistoryEntry: - """One command this session generated, kept in memory only. +#: Files copied when the frontend has to be staged into a writable cache. +#: ``node_modules`` is never copied: it is platform-specific and rebuilt there. +_COPY_IGNORE = shutil.ignore_patterns("node_modules", ".git", "*.log") - Persisting history would turn it into a preset store and put the - secret-serialization question on the table; it stays in memory in v1. - """ - when: str - summary: str - command: str - request: CleanRequest = field(repr=False) +def build_parser() -> argparse.ArgumentParser: + """The ``wm-tui`` argument surface. The bridge re-parses ``WM_TUI_ARGV`` with it.""" + parser = argparse.ArgumentParser( + prog="wm-tui", + description="Interactive terminal UI for watermark-remover.", + ) + parser.add_argument( + "path", + nargs="*", + help="File(s) or director(ies) to open. Defaults to the current directory.", + ) + parser.add_argument( + "--recursive", + action="store_true", + help="Recurse into directories when listing files", + ) + parser.add_argument("--glob", default="*", help="Directory glob for the file list") + parser.add_argument("--extensions", default=None, help="Comma-separated extension allow-list") + setup = parser.add_mutually_exclusive_group() + setup.add_argument( + "--setup", + action="store_true", + help="Open the first-run setup screen even though a saved setup exists", + ) + setup.add_argument( + "--no-setup", + action="store_true", + help="Never open the setup screen, even on a first run", + ) + parser.add_argument( + "--print-env", + action="store_true", + help="Debug: print the WM_TUI_* environment the frontend would get, as JSON, and exit", + ) + return parser -def discover_files(request: CleanRequest) -> tuple[list[Path], str | None]: - """Resolve the request's selection through the CLI's own input selector.""" - if not request.paths: - return [], None +def package_version() -> str: + """The installed distribution's version, or ``dev`` from a bare checkout.""" try: - selection = select_inputs( - request.paths, - recursive=request.recursive, - pattern=request.glob, - extensions=request.allowed_extensions(SUPPORTED_EXTENSIONS), - ) - except ValueError as error: - return [], str(error) - return [item.path for item in selection.items], None + return version("watermark-remover") + except PackageNotFoundError: + return "dev" -# --- presets ----------------------------------------------------------------- +def cache_root() -> Path: + """The per-user cache root: ``XDG_CACHE_HOME``, else the platform default.""" + base = os.environ.get("XDG_CACHE_HOME") + if base: + return Path(base) + if os.name == "nt" and os.environ.get("LOCALAPPDATA"): + return Path(os.environ["LOCALAPPDATA"]) + return Path.home() / ".cache" -@dataclass(frozen=True) -class Preset: - """One named starting point for a clean. +def frontend_dir() -> Path: + """Where the frontend lives: ``WM_TUI_DIR``, else ``tui/`` beside ``scripts/``. - A preset is a claim, not just a shortcut. Choosing "Deep clean" is - choosing a best-effort result, so the result class is part of the label - the operator reads *before* running — not something they only learn from - the results table afterwards. + The fallback resolves in a checkout (``skills/remove-ai-marks/tui``) and in + a wheel (``watermark_remover/tui``) alike, because package-data ships the + directory at the same place relative to this file. """ + override = os.environ.get("WM_TUI_DIR") + if override: + return Path(override).resolve() + return Path(__file__).resolve().parent.parent / "tui" - key: str - label: str - description: str - #: The weakest layer this preset turns on. A preset is only as verifiable - #: as its least verifiable step, so this is the honest badge for the whole - #: thing: adding a Layer B rewrite to a Layer A clean makes it best-effort. - layer: str - overrides: dict[str, object] - #: True when the preset cannot run without a reachable Layer B endpoint. - requires_endpoint: bool = False - #: The extra this preset needs installed, if any. - requires_extra: str | None = None - - def badge(self) -> str: - return format_badge(self.layer) - - def headline(self) -> str: - """Label and result class together, for the point of choice.""" - return f"{self.label} — {result_class_for(self.layer)}" - - -#: Every field a preset is allowed to set. Each preset assigns all of them, so -#: switching presets replaces the previous choice instead of layering on top of -#: it — a half-applied preset runs something nobody selected. -PRESET_FIELDS: tuple[str, ...] = ( - "nfkc", - "aggressive_homoglyphs", - "keep_non_ai_metadata", - "rewrite", - "char_perturb", - "remove_synthid", - "degrade", - "morpho", -) - -#: Fields a preset must never set. Each one either overwrites the operator's -#: input, changes what the text means, or turns the run into a description -#: instead of a clean. They are deliberate, per-run decisions with their own -#: confirmation gates; a one-click convenience control must not reach for them. -PRESET_FORBIDDEN_FIELDS: tuple[str, ...] = ( - "in_place", - "strip_semantic_format", - "dry_run", -) - -PRESETS: tuple[Preset, ...] = ( - Preset( - key="hidden", - label="Hidden marks", - description=( - "Zero-width carriers, bidi controls and AI metadata. " - "Counted before and after — nothing is rephrased. " - "Identical to a bare `wm FILE`." - ), - layer="A", - overrides={ - "nfkc": False, - "aggressive_homoglyphs": False, - "keep_non_ai_metadata": False, - "rewrite": None, - "char_perturb": False, - "remove_synthid": False, - "degrade": None, - "morpho": None, - }, - ), - Preset( - key="hidden-aggressive", - label="Hidden marks, aggressive", - description=( - "Adds NFKC normalisation and homoglyph folding: Cyrillic and Greek " - "look-alikes become ASCII. Can change genuinely mixed-script text." - ), - layer="A", - overrides={ - "nfkc": True, - "aggressive_homoglyphs": True, - "keep_non_ai_metadata": False, - "rewrite": None, - "char_perturb": False, - "remove_synthid": False, - "degrade": None, - "morpho": None, - }, - ), - Preset( - key="rewrite", - label="Deep clean (LLM rewrite)", - description=( - "Hidden marks, then a local model rephrases the text to break " - "token-level watermarks. No detector guarantee. Needs an endpoint." - ), - layer="B", - overrides={ - "nfkc": False, - "aggressive_homoglyphs": False, - "keep_non_ai_metadata": False, - "rewrite": "paraphrase", - "char_perturb": False, - "remove_synthid": False, - "degrade": None, - "morpho": None, - }, - requires_endpoint=True, - ), - Preset( - key="image", - label="Images: metadata + degrade", - description=( - "Strips C2PA and AI metadata, then perturbs the frequency domain " - "where invisible image marks live. Best-effort; the pixels change." - ), - layer="V", - overrides={ - "nfkc": False, - "aggressive_homoglyphs": False, - "keep_non_ai_metadata": False, - "rewrite": None, - "char_perturb": False, - "remove_synthid": False, - "degrade": "freq-dct", - "morpho": None, - }, - ), -) - - -def preset_for(key: str | None) -> Preset | None: - """The preset with this key, or None. An unknown key is never guessed at.""" - for preset in PRESETS: - if preset.key == key: - return preset - return None - - -def apply_preset(request: CleanRequest, preset: Preset) -> CleanRequest: - """Return ``request`` with the preset's fields — and only those — applied.""" - return replace(request, **preset.overrides) - - -# --- persisted setup --------------------------------------------------------- - -#: Environment override for the settings file, so a test never touches the -#: real one and an operator can keep per-project setups side by side. -SETTINGS_ENV = "WATERMARKS_TUI_SETTINGS" +def bridge_path() -> Path: + return Path(__file__).resolve().parent / "tui_bridge.py" -@dataclass(frozen=True) -class TuiSettings: - """The setup wm-tui remembers between runs. - Deliberately not routed through ``configuration``: that module is the - shared CLI/server config seam with its own precedence rules, and this is - one UI's memory of which endpoint you last pointed it at. The generated - command still carries every value explicitly, so a command copied out of - the TUI runs the same way on a machine that has no settings file. +def frontend_env(argv: list[str]) -> dict[str, str]: + """The ``WM_TUI_*`` variables the frontend receives (PROTOCOL "Environment"). - There is no API key field, and there never will be one. The key is read - from the environment at run time and is never rendered, copied or written - to disk — ``rewrite_api_key`` exists on ``CleanRequest`` and is absent - here on purpose. + ``WM_TUI_CWD`` is the one addition: the frontend must run with its own + directory as cwd (Bun reads ``bunfig.toml`` from there), so the bridge needs + to be told where the operator's relative paths are relative to. """ - - preset: str | None = None - rewrite_backend: str | None = None - rewrite_base_url: str | None = None - rewrite_model: str | None = None - rewrite_reasoning_effort: str | None = None - rewrite_allow_remote: bool | None = None - - def seed(self, request: CleanRequest) -> CleanRequest: - """Apply the remembered endpoint to a fresh request.""" - remembered = { - field_name: value - for field_name, value in asdict(self).items() - if field_name != "preset" and value is not None - } - return replace(request, **remembered) - - -def settings_path() -> Path: - """Where the setup file lives, honouring the usual per-platform roots.""" - override = os.environ.get(SETTINGS_ENV) - if override: - return Path(override) - base = os.environ.get("XDG_CONFIG_HOME") or os.environ.get("APPDATA") - root = Path(base) if base else Path.home() / ".config" - return root / "watermark-remover" / "tui.json" + log = os.environ.get("WM_TUI_LOG") or str(cache_root() / "watermark-remover" / "tui.log") + return { + "WM_TUI_PYTHON": sys.executable, + "WM_TUI_BRIDGE": str(bridge_path()), + "WM_TUI_ARGV": json.dumps(argv), + "WM_TUI_LOG": log, + "WM_TUI_CWD": os.getcwd(), + } -def load_settings(path: Path | None = None) -> TuiSettings: - """Read the setup file. Anything unreadable means "no saved setup". +def _stage_frontend(source: Path) -> Path: + """Copy a read-only frontend into the cache so ``bun install`` has somewhere to write. - Fail-soft on purpose: a corrupt or hand-edited settings file must not stop - the operator from starting the TUI, and every value in it is a convenience - with a visible control behind it. + Keyed by package version: an upgrade gets a fresh copy instead of running + new sources against the previous version's dependency tree. """ - target = path or settings_path() - try: - raw = json.loads(target.read_text(encoding="utf-8")) - except (OSError, ValueError): - return TuiSettings() - if not isinstance(raw, dict): - return TuiSettings() - hints = get_type_hints(TuiSettings) - allowed = { - name: tuple(kind for kind in get_args(hint) if kind is not type(None)) - for name, hint in hints.items() - } - # A value of the wrong type is as unusable as an absent key: drop it, or - # it reaches CleanRequest and fails late in classify_endpoint instead of - # failing soft here. - return TuiSettings( - **{ - key: value - for key, value in raw.items() - if key in allowed and isinstance(value, allowed[key]) - } - ) - - -def save_settings(settings: TuiSettings, path: Path | None = None) -> Path: - """Write the setup file atomically and return where it went.""" - target = path or settings_path() + target = cache_root() / "watermark-remover" / f"tui-{package_version()}" target.parent.mkdir(parents=True, exist_ok=True) - atomic_write_text(target, json.dumps(asdict(settings), indent=2, sort_keys=True) + "\n") + # copyfile, not copy2: a read-only source (a Nix-style store, 0o444 files) + # must not produce a read-only copy that the next launch cannot refresh. + shutil.copytree( + source, + target, + ignore=_COPY_IGNORE, + dirs_exist_ok=True, + copy_function=shutil.copyfile, + ) + # copytree still copies each directory's mode; bun install writes into + # the root, and the next refresh writes into every directory. + for directory, _subdirs, _files in os.walk(target): + mode = os.stat(directory).st_mode + os.chmod(directory, mode | stat.S_IWUSR | stat.S_IXUSR) return target -def _build_parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser( - prog="wm-tui", - description="Interactive terminal UI for watermark-remover.", +def _install(directory: Path, bun: str) -> int: + """``bun install --frozen-lockfile``: the lockfile, not the network, decides versions.""" + print( + f"wm-tui: installing frontend dependencies in {directory} (first run only)...", + file=sys.stderr, ) - parser.add_argument( - "path", - nargs="*", - type=Path, - help="File(s) or director(ies) to open. Defaults to the current directory.", + result = subprocess.run( + [bun, "install", "--frozen-lockfile"], + cwd=directory, + check=False, ) - parser.add_argument( - "--recursive", - action="store_true", - help="Recurse into directories when listing files", - ) - parser.add_argument("--glob", default="*", help="Directory glob for the file list") - parser.add_argument("--extensions", default=None, help="Comma-separated extension allow-list") - return parser + if result.returncode != 0: + print( + f"wm-tui: `bun install --frozen-lockfile` failed in {directory} " + f"(exit {result.returncode})", + file=sys.stderr, + ) + return result.returncode + + +def prepare_frontend(source: Path, bun: str) -> tuple[Path | None, int]: + """Return a directory with installed dependencies to run from, or an exit code.""" + if not (source / "package.json").is_file() or not (source / "src" / "index.tsx").is_file(): + print( + f"wm-tui: the terminal frontend is missing from {source} " + "(expected package.json and src/index.tsx). Set WM_TUI_DIR to its directory.", + file=sys.stderr, + ) + return None, 1 + run_dir = source + if not (source / "node_modules").is_dir(): + if not os.access(source, os.W_OK): + try: + run_dir = _stage_frontend(source) + except OSError as error: + print(f"wm-tui: cannot stage the frontend into the cache: {error}", file=sys.stderr) + return None, 1 + if not (run_dir / "node_modules").is_dir(): + code = _install(run_dir, bun) + if code != 0: + return None, 1 + return run_dir, 0 def main(argv: list[str] | None = None) -> int: - args = _build_parser().parse_args(argv) - availability = check_optional(TUI_EXTRA) - if not availability.available: - print(availability.hint, file=sys.stderr) + raw = list(sys.argv[1:] if argv is None else argv) + args = build_parser().parse_args(raw) + + # Refuse a typo here, the way ``wm`` does, rather than open a full-screen + # UI whose only content is an error in the status bar. + missing = [path for path in (args.path or ["."]) if not Path(path).exists()] + if missing: + for path in missing: + print(f"wm-tui: no such file or directory: {path}", file=sys.stderr) return 2 - paths = tuple(args.path) if args.path else (Path.cwd(),) - settings = load_settings() - # The saved setup only seeds the endpoint fields. Anything the operator - # typed on the command line stays exactly as typed. - request = settings.seed( - CleanRequest( - paths=paths, - recursive=args.recursive, - glob=args.glob, - extensions=args.extensions, - ) - ) - # Imported here, not at module scope: the guard above must be able to print - # an install hint on a default install where textual is absent. - from tui_app import WatermarkTuiApp - - WatermarkTuiApp(request, preset=preset_for(settings.preset)).run() - return 0 + # The frontend gets the arguments minus the launcher's own debug switch. + passed = [item for item in raw if item != "--print-env"] + wm_env = frontend_env(passed) + if args.print_env: + # Only the WM_TUI_* values: the inherited environment can carry the + # rewrite API key, and a debug flag must not be how it leaks. + print(json.dumps(wm_env, indent=2, sort_keys=True)) + return 0 + + bun = shutil.which("bun") + if bun is None: + print(BUN_INSTALL_HINT, file=sys.stderr) + return 1 + + run_dir, code = prepare_frontend(frontend_dir(), bun) + if run_dir is None: + return code + + log_dir = Path(wm_env["WM_TUI_LOG"]).parent + try: + log_dir.mkdir(parents=True, exist_ok=True) + except OSError as error: + print(f"wm-tui: cannot create the log directory {log_dir}: {error}", file=sys.stderr) + return 1 + + env = {**os.environ, **wm_env} + command = [bun, "run", "src/index.tsx"] + if os.name == "nt": + # No exec on Windows that keeps the console; wait and pass the code on. + return subprocess.run(command, cwd=run_dir, env=env, check=False).returncode + os.chdir(run_dir) + # Replace this process: the frontend owns the terminal and its signals, + # and a Python parent waiting on it would only be one more thing to kill. + os.execvpe(bun, command, env) # noqa: S606 - argv list, no shell + return 0 # pragma: no cover - execvpe does not return if __name__ == "__main__": diff --git a/skills/remove-ai-marks/scripts/tui_app.py b/skills/remove-ai-marks/scripts/tui_app.py deleted file mode 100644 index dfe80fe..0000000 --- a/skills/remove-ai-marks/scripts/tui_app.py +++ /dev/null @@ -1,1924 +0,0 @@ -#!/usr/bin/env python3 -"""Textual widgets for wm-tui. - -Split from ``tui.py`` so the entry point can print an install hint on a default -install where textual is absent, and so the pure logic in ``tui`` stays testable -without a terminal. - -Every rule in ``tui``'s module docstring applies here. In particular: no -``urllib``, no ``CleanPlan`` construction, no verified badge on a best-effort -layer, and no API key ever reaching a widget or a copied string. -""" - -from __future__ import annotations - -import difflib -import os -import sys -import time -from dataclasses import dataclass, field, replace -from pathlib import Path -from typing import ClassVar - -from rich.markup import escape -from textual import on, work -from textual.app import App, ComposeResult -from textual.containers import Horizontal, Vertical, VerticalScroll -from textual.screen import ModalScreen -from textual.widgets import ( - Button, - Checkbox, - DataTable, - Footer, - Header, - Input, - Label, - RichLog, - Select, - SelectionList, - Static, - TabbedContent, - TabPane, - TextArea, -) -from textual.widgets._select import NoSelection - -sys.path.insert(0, str(Path(__file__).resolve().parent)) - -from asset_kind import SUPPORTED_EXTENSIONS -from batch_inputs import select_inputs -from clean_asset import DEGRADE_CLI_CHOICES, MORPHO_CLI_CHOICES -from clean_file import dry_run_payload, run_clean_item -from clean_request import ( - FORCED_KINDS, - QUALITY_PROFILES, - REWRITE_CLI_CHOICES, - CleanPlanPreflightError, - CleanRequest, - build_rewrite_plan, - describe_dropped_text_transforms, - dropped_text_transforms, - plan_work, - resolve_kind, -) -from common import atomic_write_text -from inspect_file import inspect_asset -from layer_b_discovery import classify_endpoint, probe_backend -from morphomod import VISIBLE_CLEAN_BACKENDS -from optional_deps import KNOWN_EXTRAS, check_optional -from perturb_text import MODES as PERTURB_MODES -from rewrite_text import ( - LIVE_REWRITE_BACKENDS, - REASONING_EFFORTS, - RewriteConfigurationError, - generate_candidates, -) -from score_stylometry import score_text_stylometry -from tui import ( - DEFAULT_REWRITE_TIMEOUT, - IDLE_STREAM_TITLE, - PRESETS, - STREAM_VIEW_CHARS, - HistoryEntry, - Preset, - TuiSettings, - apply_preset, - discover_files, - estimate_rewrite_seconds, - format_badge, - format_duration, - layer_for_result, - preset_for, - result_class_for, - save_settings, - should_confirm_cost, -) - -REWRITE_ENV_KEY = "WATERMARKS_REWRITE_API_KEY" - -#: Column label and share of the leftover width, per table. ``DataTable`` -#: sizes columns to their content, which leaves a table of short rows stopping -#: a third of the way across its pane with a dark strip after it; these weights -#: are what spread the columns over the whole width instead. -CAPABILITY_COLUMNS: tuple[tuple[str, int], ...] = ( - ("capability", 2), - ("state", 1), - ("detail", 5), -) -RUN_COLUMNS: tuple[tuple[str, int], ...] = ( - ("file", 4), - ("kind", 1), - ("result", 1), - ("class", 2), - ("residual", 1), - ("note", 4), -) -HISTORY_COLUMNS: tuple[tuple[str, int], ...] = ( - ("time", 1), - ("summary", 8), -) - - -def _column_widths(total: int, spec: tuple[tuple[str, int], ...]) -> list[int]: - """Split ``total`` columns across ``spec``, exactly and without overflow. - - Every column gets at least its own label, so a narrow terminal degrades to - a readable header rather than a row of ellipses. Anything left over is - shared by weight and the remainder lands on the last column, which is what - makes the widths add up to ``total`` rather than to one or two less. - """ - floors = [max(len(label), 4) for label, _ in spec] - if total <= sum(floors): - return floors - weights = [weight for _, weight in spec] - extra = total - sum(floors) - share = sum(weights) or 1 - widths = [ - floor + extra * weight // share for floor, weight in zip(floors, weights, strict=True) - ] - widths[-1] += total - sum(widths) - return widths - - -@dataclass -class _TableModel: - """The rows a ``DataTable`` is showing, kept so it can be re-laid out. - - Column widths can only be given when a column is added, so filling the - pane's width after a resize means re-adding the columns — and therefore - re-adding the rows. Holding them here is what makes that possible without - reading them back out of the widget. - """ - - spec: tuple[tuple[str, int], ...] - rows: list[tuple[str, ...]] = field(default_factory=list) - #: Width the columns were last laid out for. Negative forces a redraw. - width: int = 0 - - -class ConfirmModal(ModalScreen[bool]): - """A yes/no gate that defaults to No. - - Used for every irreversible or outward-facing action: remote egress, - in-place overwrite, semantic stripping, and the batch cost ceiling. - """ - - BINDINGS: ClassVar[list] = [("escape", "dismiss_false", "Cancel")] - - def __init__(self, title: str, body: str, confirm_label: str = "Proceed") -> None: - super().__init__() - self._title = title - self._body = body - self._confirm_label = confirm_label - - def compose(self) -> ComposeResult: - with Vertical(id="modal-body"): - yield Label(self._title, id="modal-title") - yield Static(self._body, id="modal-text") - with Horizontal(id="modal-buttons"): - yield Button("No", variant="primary", id="modal-no") - yield Button(self._confirm_label, variant="warning", id="modal-yes") - - def on_mount(self) -> None: - # Default focus lands on No: confirmation must be a deliberate act. - self.query_one("#modal-no", Button).focus() - - @on(Button.Pressed, "#modal-yes") - def _yes(self) -> None: - self.dismiss(True) - - @on(Button.Pressed, "#modal-no") - def _no(self) -> None: - self.dismiss(False) - - def action_dismiss_false(self) -> None: - self.dismiss(False) - - -class CandidateModal(ModalScreen[int | None]): - """Pick among generated rewrites instead of accepting the auto-selected one.""" - - BINDINGS: ClassVar[list] = [("escape", "dismiss_none", "Cancel")] - - def __init__(self, candidates) -> None: - super().__init__() - self._candidates = candidates - - def compose(self) -> ComposeResult: - with Vertical(id="modal-body"): - yield Label( - f"{len(self._candidates)} candidates — {format_badge('B')}", - id="modal-title", - ) - table = DataTable(id="candidate-table", cursor_type="row") - yield table - yield TextArea("", read_only=True, id="candidate-preview") - with Horizontal(id="modal-buttons"): - yield Button("Cancel", id="cand-cancel") - yield Button("Use selected", variant="primary", id="cand-ok") - - def on_mount(self) -> None: - table = self.query_one("#candidate-table", DataTable) - table.add_columns("#", "divergence", "score", "auto-pick", "chars") - for candidate in self._candidates: - table.add_row( - str(candidate.index + 1), - f"{candidate.lexical_divergence:.3f}", - f"{candidate.selection_score:.3f}", - "yes" if candidate.selected else "", - str(len(candidate.text)), - ) - table.focus() - self._show(0) - - def _show(self, row: int) -> None: - if 0 <= row < len(self._candidates): - self.query_one("#candidate-preview", TextArea).text = self._candidates[row].text - - @on(DataTable.RowHighlighted, "#candidate-table") - def _highlight(self, event: DataTable.RowHighlighted) -> None: - self._show(event.cursor_row) - - @on(Button.Pressed, "#cand-ok") - def _ok(self) -> None: - self.dismiss(self.query_one("#candidate-table", DataTable).cursor_row) - - @on(Button.Pressed, "#cand-cancel") - def _cancel(self) -> None: - self.dismiss(None) - - def action_dismiss_none(self) -> None: - self.dismiss(None) - - -class WatermarkTuiApp(App): - """inspect -> clean -> re-inspect, over the same seam the CLI uses.""" - - TITLE = "wm-tui" - SUB_TITLE = "watermark-remover" - - CSS = """ - Screen { layout: vertical; } - /* Header and Footer are docked, so the flow area is two rows shorter than - the screen. Left at ``height: auto`` the tab container claimed all of - it and pushed the status bar out, which cost the screen a permanent - vertical scrollbar -- two columns stolen from every pane, on a screen - that had nothing to scroll. */ - #tabs { height: 1fr; } - - /* -- the shared row grid --------------------------------------------- */ - .row { height: auto; } - /* ``.muted``, not a bare ``Static``: ``Checkbox`` subclasses ``Static``, - so the type selector also caught every checkbox and stretched it to - fill the row -- "recursive" rendered 46 columns wide next to a - 16-column button. These are the inline status labels only. */ - .row > .muted { width: 1fr; height: 3; content-align: left middle; padding: 0 1; } - /* Checkboxes are ``width: auto`` by default, which left-packs them and - leaves the rest of the row empty. Give them the same bounded share as - the input fields so a row of controls reads as one grid. */ - .row > Checkbox { width: 1fr; } - /* Share the row rather than claiming a fixed 32 columns each: four - fields at a fixed width overflow an 80- or 120-column terminal and - the last one is simply unreachable. ``1fr`` cannot overflow, so no - ``max-width`` is needed to stay safe -- and capping it left a ragged - gap at the end of every row on a wide terminal while the uncapped - neighbours stretched past it. */ - .field { width: 1fr; } - .wide { width: 1fr; } - - /* -- type ------------------------------------------------------------- */ - .muted { color: $text-muted; } - /* Names the controls in the row beneath it. A placeholder is gone the - moment a field is filled, and "10" in a row of three inputs says - nothing about which field it is. */ - .caption { color: $text-muted; height: 1; padding: 0 1; } - /* Breathing room between groups; a form with no rhythm reads as one - undifferentiated wall of controls. */ - .section { text-style: bold; margin-top: 1; } - .section:first-of-type { margin-top: 0; } - - /* -- panes ------------------------------------------------------------ */ - TabPane { padding: 1 1 0 1; } - /* Every scrolling surface gets the same titled frame, so a pane reads as - a labelled thing rather than text floating on the terminal. */ - DataTable { border: round $panel; } - DataTable:focus { border: round $accent; } - #files-list { height: 1fr; border: round $panel; } - #files-list:focus { border: round $accent; } - #inspect-report { height: 1fr; border: round $panel; padding: 1; } - #command-preview { height: auto; min-height: 3; border: round $accent; padding: 0 1; } - #command-copyable { height: 5; border: round $panel; } - #run-table { height: 1fr; min-height: 6; } - #run-log { height: 1fr; min-height: 6; border: round $panel; } - #history-table { height: 1fr; } - /* Side by side: the stream is what the model is writing now, the diff is - what changed — reading one without the other is half the answer, and - stacking them starved the result table above. */ - #run-panes { height: 10; } - #diff-view { width: 1fr; border: round $panel; } - #stream-view { width: 1fr; border: round $accent; } - #status-bar { height: 1; padding: 0 1; background: $panel; color: $text-muted; } - - /* -- the Start pane --------------------------------------------------- */ - /* One bordered, titled block per step. The onboarding is four things to - do in order; a flat column of controls does not say that. */ - .step { height: auto; border: round $primary 50%; padding: 0 1 1 1; margin-bottom: 1; } - .step:focus-within { border: round $accent; } - /* One line, not three: a summary padded to the height of a button row - left a hole in the middle of the first thing the operator reads. */ - #start-files { width: 1fr; height: auto; padding: 0 1; } - #start-ready { width: 1fr; height: 3; content-align: left middle; padding: 0 1; } - #preset-detail { height: auto; min-height: 2; padding: 0 1; } - #backend-table { height: auto; max-height: 12; } - #install-command { height: 3; border: round $panel; } - #btn-start-run { min-width: 22; } - - /* -- modals ----------------------------------------------------------- */ - #modal-body { - width: 84; height: auto; max-height: 90%; - border: thick $warning; background: $surface; padding: 1 2; - } - #modal-title { text-style: bold; } - #modal-text { padding: 1 0; } - #modal-buttons { height: auto; align-horizontal: right; } - #candidate-table { height: 10; } - #candidate-preview { height: 12; } - """ - - BINDINGS: ClassVar[list] = [ - ("q", "quit", "Quit"), - ("r", "rescan", "Rescan"), - ("i", "inspect", "Inspect"), - ("ctrl+r", "run", "Run"), - ] - - def __init__(self, request: CleanRequest, *, preset: Preset | None = None) -> None: - super().__init__() - self.request = request - # The saved setup's preset, or the safest one. Landing on a preset is - # what makes "add files, press Clean" work without a tour of the Plan - # tab; the pane names the preset and the command preview shows exactly - # what it turned on, so nothing about it is implicit. - self.preset: Preset = preset or PRESETS[0] - self.files: list[Path] = [] - self.selected: list[Path] = [] - self.history: list[HistoryEntry] = [] - self._cancel_requested = False - # Namespaced deliberately: textual's App owns a private `_running`, and - # reusing that name silently reads the framework's lifecycle state. - self._clean_running = False - # Kept so rebuilding the capability table does not discard a probe the - # operator just ran. - self._last_probe = None - # Install command per capability row, by row index; "" for the rows - # that are not an extra. - self._install_commands: list[str] = [] - # Which extras are installed, cached. The endpoint row is rebuilt on - # every keystroke in the base-URL field — the control and the row it - # describes are on the same pane now, so a stale row would contradict - # the field right next to it — and re-importing five optional packages - # per keystroke to learn nothing new is not worth that. - self._extra_rows: list[tuple[tuple[str, ...], str]] | None = None - # Asset kind per discovered file, from the last rescan. Classifying - # reads the file, so it happens once per rescan rather than on every - # keystroke that redraws the readiness line. - self._kinds: dict[Path, str] = {} - # Guards every handler that reaches into the widget tree. Textual - # mounts children as ``compose`` yields them and processes messages in - # between, so a ``Select`` constructed with a value posts ``Changed`` - # before the widgets its handler would go on to query exist — and the - # same messages can still be in flight while the app is tearing down, - # where the query raises instead of finding a widget. Both only ever - # showed up on a loaded runner, which is exactly why the flag is - # explicit rather than a timing assumption. - self._widgets_live = False - self._tables: dict[str, _TableModel] = { - "#run-table": _TableModel(RUN_COLUMNS), - "#backend-table": _TableModel(CAPABILITY_COLUMNS), - "#history-table": _TableModel(HISTORY_COLUMNS), - } - - # -- layout ------------------------------------------------------------ - - def compose(self) -> ComposeResult: - yield Header() - # Start first, and it is the setup pane: the old landing tab was a file - # list over a plan nobody had configured yet, and the capability table - # that told you what was missing was five tabs away with no way to act - # on any of it. - with TabbedContent(initial="tab-start", id="tabs"): - with TabPane("Start", id="tab-start"): - yield from self._compose_start() - with TabPane("Files", id="tab-files"): - yield from self._compose_files() - with TabPane("Inspect", id="tab-inspect"): - yield from self._compose_inspect() - with TabPane("Plan", id="tab-plan"): - yield from self._compose_plan() - with TabPane("Run", id="tab-run"): - yield from self._compose_run() - with TabPane("History", id="tab-history"): - yield from self._compose_history() - yield Static("", id="status-bar") - yield Footer() - - def _compose_start(self) -> ComposeResult: - """Add files, choose what to remove, point at a backend, clean.""" - with VerticalScroll(id="start"): - with Vertical(classes="step") as step: - step.border_title = "1 · Add files" - yield Static( - "add as many as you like · filters on the Files tab", classes="caption" - ) - with Horizontal(classes="row"): - yield Input( - placeholder="path to a file or a folder", - id="in-add-path", - classes="wide", - ) - yield Button("Add", variant="primary", id="btn-add-path") - yield Button("Clear", id="btn-clear-paths") - yield Static("", id="start-files", classes="muted") - - with Vertical(classes="step") as step: - step.border_title = "2 · Choose what to remove" - with Horizontal(classes="row"): - yield Select( - [(preset.label, preset.key) for preset in PRESETS], - value=self.preset.key, - allow_blank=False, - id="sel-preset", - classes="field", - ) - yield Static( - "every option a preset sets stays visible on the Plan tab", - classes="muted", - ) - yield Static("", id="preset-detail") - - with Vertical(classes="step") as step: - step.border_title = "3 · Clean" - with Horizontal(classes="row"): - yield Button("Clean now", variant="primary", id="btn-start-run") - yield Button("Advanced options", id="btn-advanced") - yield Static("", id="start-ready", classes="muted") - - with Vertical(classes="step") as step: - step.border_title = "Layer B endpoint — only the rewrite preset needs one" - yield Static("backend · base URL · model", classes="caption") - with Horizontal(classes="row"): - yield Select( - [(name, name) for name in LIVE_REWRITE_BACKENDS], - prompt="backend", - allow_blank=True, - id="sel-backend", - classes="field", - ) - yield Input( - placeholder="base URL (http://127.0.0.1:11434)", - id="in-base-url", - classes="field", - ) - # Free text, not a discovery-only picker: an endpoint that - # does not list models (or is not reachable yet) must still - # be usable. - yield Input(placeholder="model", id="in-model", classes="field") - yield Button("Probe", id="btn-probe") - yield Static("discovered models · reasoning effort", classes="caption") - with Horizontal(classes="row"): - yield Select( - [], prompt="discovered", allow_blank=True, id="sel-model", classes="field" - ) - yield Select( - [(name, name) for name in REASONING_EFFORTS], - prompt="reasoning effort", - allow_blank=True, - id="sel-effort", - classes="field", - ) - yield Checkbox("disable thinking", False, id="cb-disable-thinking") - with Horizontal(classes="row"): - yield Checkbox("allow remote endpoint", False, id="cb-allow-remote") - yield Button("Save setup", id="btn-save-settings") - yield Static("", id="endpoint-state", classes="muted") - yield Static("", id="settings-state", classes="muted") - yield Static("", id="api-key-state", classes="muted") - - with Vertical(classes="step") as step: - step.border_title = "Installed capabilities" - yield DataTable(id="backend-table", cursor_type="row", zebra_stripes=True) - yield Static("select a row for the command that installs it", classes="caption") - yield TextArea("", read_only=True, id="install-command") - with Horizontal(classes="row"): - yield Button("Refresh", id="btn-refresh-backends") - yield Static( - "the core clean needs nothing extra; the rows above are opt-in backends", - classes="muted", - ) - - def _compose_files(self) -> ComposeResult: - with Horizontal(classes="row"): - yield Input(value=self.request.glob, placeholder="glob", id="in-glob", classes="field") - yield Input( - value=self.request.extensions or "", - placeholder="extensions (.md,.txt)", - id="in-extensions", - classes="field", - ) - yield Checkbox("recursive", self.request.recursive, id="cb-recursive") - yield Button("Rescan", id="btn-rescan") - listing = SelectionList[str](id="files-list") - listing.border_title = "space toggles · every file is selected after a rescan" - yield listing - yield Static("", id="files-summary", classes="muted") - - def _compose_inspect(self) -> ComposeResult: - with Horizontal(classes="row"): - yield Button("Inspect selected", variant="primary", id="btn-inspect") - yield Checkbox("soft binding", False, id="cb-inspect-soft") - yield VerticalScroll(Static("", id="inspect-report")) - - def _compose_plan(self) -> ComposeResult: - with VerticalScroll(): - yield Label("Routing", classes="section") - yield Static("asset kind · force text · audit record", classes="caption") - with Horizontal(classes="row"): - yield Select( - [(name, name) for name in FORCED_KINDS], - value="auto", - allow_blank=False, - id="sel-force-type", - classes="field", - ) - # The skipped-transform note tells the operator to force text on - # a container; it would be a poor UI that said so and then made - # them leave for the CLI. - yield Checkbox("force text", False, id="cb-force-text") - yield Input(placeholder="audit JSON path", id="in-audit", classes="field") - - yield Label("Layer A — hidden Unicode " + format_badge("A"), classes="section") - with Horizontal(classes="row"): - yield Checkbox("NFKC", self.request.nfkc, id="cb-nfkc") - yield Checkbox("aggressive homoglyphs", False, id="cb-homoglyphs") - yield Checkbox("strip semantic format", False, id="cb-semantic") - - yield Label("Layer M — metadata " + format_badge("M"), classes="section") - with Horizontal(classes="row"): - yield Checkbox("keep non-AI metadata", False, id="cb-keep-meta") - yield Checkbox("detect soft binding", False, id="cb-soft") - - yield Label("Layer B — LLM rewrite " + format_badge("B"), classes="section") - # The endpoint, the model and the key state are set-once settings - # and live on the Start tab; what stays here is what changes per - # run. Splitting them that way is what lets Start be short enough - # to read. - yield Static( - "strength · candidates · temperature · per-call timeout (s) — " - "backend and model are on the Start tab", - classes="caption", - ) - with Horizontal(classes="row"): - yield Select( - [(name, name) for name in REWRITE_CLI_CHOICES], - prompt="strength (off)", - allow_blank=True, - id="sel-rewrite", - classes="field", - ) - yield Input( - placeholder="candidates", value="1", id="in-candidates", classes="field" - ) - yield Input(placeholder="temperature", id="in-temperature", classes="field") - yield Input(placeholder="timeout s", id="in-rewrite-timeout", classes="field") - yield Static( - "pivot language · original language · tsapa generations · tsapa population", - classes="caption", - ) - with Horizontal(classes="row"): - yield Input(placeholder="pivot lang", id="in-lang", classes="field") - yield Input(placeholder="original lang", id="in-original-lang", classes="field") - yield Input(placeholder="tsapa generations", id="in-generations", classes="field") - yield Input(placeholder="tsapa population", id="in-population", classes="field") - with Horizontal(classes="row"): - yield Button("Pick from candidates", id="btn-candidates") - yield Static( - "generates N candidates for one text file and lets you choose", - classes="muted", - ) - - yield Label("Character perturbation", classes="section") - with Horizontal(classes="row"): - yield Checkbox("char perturb", False, id="cb-perturb") - yield Select( - [(name, name) for name in PERTURB_MODES], - prompt="mode", - allow_blank=True, - id="sel-perturb-mode", - classes="field", - ) - yield Input(placeholder="strength 0-1", id="in-perturb-strength", classes="field") - yield Input(placeholder="seed", id="in-seed", classes="field") - - yield Label("Layer V — visible marks " + format_badge("V"), classes="section") - yield Static("mask · box · inpaint backend · dilation radius", classes="caption") - with Horizontal(classes="row"): - yield Input(placeholder="mask path", id="in-mask", classes="field") - yield Input(placeholder="box x,y,w,h", id="in-box", classes="field") - yield Select( - [(name, name) for name in VISIBLE_CLEAN_BACKENDS], - prompt="visible backend", - allow_blank=True, - id="sel-visible-backend", - classes="field", - ) - yield Input(placeholder="dilate radius", id="in-dilate", classes="field") - yield Static( - "detector command · inpainter command · prompt · external timeout (s)", - classes="caption", - ) - with Horizontal(classes="row"): - yield Input( - placeholder="detect cmd {input} {mask}", - id="in-detect-command", - classes="field", - ) - yield Input( - placeholder="inpaint cmd {input} {mask} {output}", - id="in-inpaint-command", - classes="field", - ) - yield Input(placeholder="visible prompt", id="in-visible-prompt", classes="field") - yield Input(placeholder="external timeout", id="in-timeout", classes="field") - with Horizontal(classes="row"): - yield Select( - [(name, name) for name in QUALITY_PROFILES], - prompt="quality", - allow_blank=True, - id="sel-quality", - classes="field", - ) - yield Checkbox("remove SynthID", False, id="cb-synthid") - yield Checkbox("dry run", False, id="cb-dry-run") - yield Static("", id="visible-state", classes="muted") - - yield Label("Image degradation " + format_badge("V"), classes="section") - yield Static( - "frequency · morphological · strength · seed · SynthID strength", - classes="caption", - ) - with Horizontal(classes="row"): - yield Select( - [(name, name) for name in DEGRADE_CLI_CHOICES], - prompt="degrade", - allow_blank=True, - id="sel-degrade", - classes="field", - ) - yield Select( - [(name, name) for name in MORPHO_CLI_CHOICES], - prompt="morpho", - allow_blank=True, - id="sel-morpho", - classes="field", - ) - yield Input( - placeholder="degrade strength", id="in-degrade-strength", classes="field" - ) - yield Input(placeholder="degrade seed", id="in-degrade-seed", classes="field") - yield Input( - placeholder="synthid strength", id="in-synthid-strength", classes="field" - ) - - yield Label("Output", classes="section") - with Horizontal(classes="row"): - yield Input(placeholder="output path or directory", id="in-output", classes="wide") - yield Checkbox("in place", False, id="cb-in-place") - yield Checkbox("keep artifacts", False, id="cb-artifacts") - yield Checkbox("wmCt marker", False, id="cb-wmct") - - yield Label("Equivalent command", classes="section") - yield Static("", id="command-preview") - with Horizontal(classes="row"): - yield Button("Copy command", id="btn-copy-command") - yield Static("", id="copy-state", classes="muted") - yield TextArea("", read_only=True, id="command-copyable") - - def _compose_run(self) -> ComposeResult: - with Horizontal(classes="row"): - yield Button("Run", variant="primary", id="btn-run") - yield Button("Stop after current file", id="btn-cancel", disabled=True) - yield Static("", id="run-state", classes="muted") - results = DataTable(id="run-table", zebra_stripes=True) - results.border_title = "results — one row per file, badged by layer" - yield results - with Horizontal(id="run-panes"): - stream = TextArea("", read_only=True, id="stream-view") - stream.border_title = IDLE_STREAM_TITLE - yield stream - diff = TextArea("", read_only=True, id="diff-view") - diff.border_title = "before / after" - yield diff - log = RichLog(id="run-log", markup=True, wrap=True) - log.border_title = "log — the after-state of every file, re-inspected" - yield log - - def _compose_history(self) -> ComposeResult: - with Horizontal(classes="row"): - yield Button("Copy", id="btn-history-copy") - yield Button("Reuse", id="btn-history-reuse") - yield Static("in-memory, this session only", classes="muted") - table = DataTable(id="history-table", cursor_type="row", zebra_stripes=True) - table.border_title = "every command this session generated" - yield table - yield TextArea("", read_only=True, id="history-copyable") - - # -- lifecycle --------------------------------------------------------- - - def on_mount(self) -> None: - self._widgets_live = True - for selector in self._tables: - self._relayout_table(selector) - self._refresh_api_key_state() - # The preset is applied, not merely displayed: the Plan widgets and the - # command preview must say what pressing "Clean now" would actually do. - self.apply_request(apply_preset(self.request, self.preset)) - self.action_rescan() - self.refresh_backends() - self._sync_preview() - - def on_unmount(self) -> None: - self._widgets_live = False - - def on_resize(self) -> None: - """Re-lay the tables when the terminal changes size.""" - self._relayout_tables() - - def _relayout_tables(self) -> None: - """Re-lay every table that now knows how wide it is. - - A table in a hidden ``TabPane`` has no width to divide up, so the one - on the tab you open second would keep the fallback columns it was given - at mount — the very dark strip this is here to remove. - """ - if not self._widgets_live: - return - for selector in self._tables: - if self.query(selector): - self._relayout_table(selector) - - # -- tables ------------------------------------------------------------ - - def _relayout_table(self, selector: str) -> bool: - """Rebuild a table so its columns span the pane. True when it redrew. - - ``DataTable`` sizes each column to its content and column widths can - only be given when the column is added, so a table of short rows stops - halfway across the pane and the header band ends in a dark strip. - Filling the width means re-adding the columns, which means re-adding - the rows — which is why the rows are kept in ``_TableModel``. - """ - model = self._tables[selector] - table = self.query_one(selector, DataTable) - width = table.size.width - if width == model.width and table.columns: - return False - model.width = width - usable = width - 2 * table.cell_padding * len(model.spec) - # Before the first layout the table has no width to divide up; add the - # columns anyway so a row written now is not dropped for want of one. - widths = _column_widths(usable, model.spec) if usable > 0 else [None] * len(model.spec) - table.clear(columns=True) - for (label, _), column_width in zip(model.spec, widths, strict=True): - table.add_column(label, width=column_width) - for row in model.rows: - table.add_row(*row) - return True - - def _add_table_row(self, selector: str, *cells: str) -> None: - self._tables[selector].rows.append(cells) - if not self._widgets_live: - return - # A relayout re-adds every row itself; appending twice would double it. - if not self._relayout_table(selector): - self.query_one(selector, DataTable).add_row(*cells) - - def _set_table_rows(self, selector: str, rows: list[tuple[str, ...]]) -> None: - model = self._tables[selector] - model.rows = rows - model.width = -1 # no real width is negative, so this forces the redraw - if not self._widgets_live: - return - self._relayout_table(selector) - - def _status(self, message: str) -> None: - if not self._widgets_live: - return - self.query_one("#status-bar", Static).update(message) - - def _log(self, message: str) -> None: - if not self._widgets_live: - return - self.query_one("#run-log", RichLog).write(message) - - # -- files ------------------------------------------------------------- - - def action_rescan(self) -> None: - if not self._widgets_live: - return - self.request = replace( - self.request, - glob=self.query_one("#in-glob", Input).value or "*", - extensions=self.query_one("#in-extensions", Input).value or None, - recursive=self.query_one("#cb-recursive", Checkbox).value, - ) - files, error = discover_files(self.request) - listing = self.query_one("#files-list", SelectionList) - listing.clear_options() - if error: - self._status(f"selection error: {error}") - self.files = [] - self.selected = [] - self._update_files_summary() - self._sync_preview() - return - self.files = files - self._kinds = {path: self._classify(path) for path in files} - listing.add_options([(str(path), str(path), True) for path in files]) - # Everything the rescan found is selected: the list it replaced no - # longer matches, so carrying a previous deselection forward would - # silently apply it to different files. ``_update_files_summary`` - # says so rather than leaving it to be discovered. - self.selected = list(files) - self._update_files_summary() - self._sync_preview() - - def _classify(self, path: Path) -> str: - """Best-effort asset kind. A file we cannot classify is not a warning.""" - try: - return resolve_kind(path, self.request) - except (ValueError, OSError): - return "unknown" - - def _update_files_summary(self) -> None: - filters = [f"glob {self.request.glob}"] - if self.request.recursive: - filters.append("recursive") - if self.request.extensions: - filters.append(f"ext {self.request.extensions}") - self.query_one("#files-summary", Static).update( - f"{len(self.selected)} of {len(self.files)} selected · " + " · ".join(filters) - ) - self._sync_start() - - # -- start pane -------------------------------------------------------- - - def _sync_start(self) -> None: - """Keep the onboarding pane's three answers current.""" - if not self._widgets_live: - return - roots = len(self.request.paths) - self.query_one("#start-files", Static).update( - f"{len(self.files)} file(s) under {roots} path(s) · {len(self.selected)} selected" - ) - self.query_one("#preset-detail", Static).update( - f"{self.preset.badge()} {escape(self.preset.description)}" - ) - self.query_one("#start-ready", Static).update(self._readiness()) - # The Run pane's own state line, which nothing wrote to before: the - # operator arrives there from another tab and has to be told what - # pressing Run would act on. - self.query_one("#run-state", Static).update( - f"{len(self.selected)} file(s) · {escape(self.preset.label)} {self.preset.badge()}" - ) - # The endpoint controls and the row that reports the endpoint policy - # are on the same pane; a row that lags the field above it by a tab - # switch is a pane arguing with itself. - if self.is_mounted: - self.refresh_backends(recheck=False) - - def _readiness(self) -> str: - """What still stands between this form and a run. Never a bare "ready".""" - blockers = [] - if not self.selected: - blockers.append("step 1: no files selected") - if self.preset.requires_endpoint and not self._value("#in-base-url"): - blockers.append("this preset needs a Layer B endpoint — set one below") - extra = self.preset.requires_extra - if extra and not check_optional(extra).available: - blockers.append(f"needs watermark-remover[{extra}]") - blockers.extend(self._selection_warnings()) - if blockers: - return "[yellow]" + escape(" · ".join(blockers)) + "[/]" - return f"[green]ready[/] — {len(self.selected)} file(s), {self.preset.badge()}" - - def _selection_warnings(self) -> list[str]: - """What this plan would do to this selection that the operator has not seen. - - Two shapes, and the difference matters. A text transform on a - container is a *silent no-op*: ``.md`` and ``.html`` route to the - container pipeline, the rewrite chosen in step 2 never runs, and the - old UI said so only in one row of a results table afterwards. Image - degradation on a non-image is a *refusal*, and it aborts in preflight — - so one ``.md`` in the folder means the whole batch writes nothing. - Both belong in front of the operator before the run, not after it. - """ - try: - request = self.collect_request() - except ValueError: - return [] - warnings = [] - counted: dict[str, int] = {} - for path in self.selected: - for name in dropped_text_transforms(request, self._kinds.get(path, "unknown")): - counted[name] = counted.get(name, 0) + 1 - warnings.extend( - f"{count} file(s) would skip {name} — tick “force text” on Plan" - for name, count in sorted(counted.items()) - ) - if request.degrade or request.morpho: - others = sum(1 for path in self.selected if self._kinds.get(path, "unknown") != "image") - if others: - warnings.append( - f"{others} selected file(s) are not images — image degradation " - "refuses the whole run in preflight, writing nothing" - ) - return warnings - - @on(Button.Pressed, "#btn-add-path") - @on(Input.Submitted, "#in-add-path") - def _add_path(self) -> None: - raw = self._value("#in-add-path") - if raw is None: - self._status("type a file or folder path first") - return - path = Path(raw).expanduser() - if not path.exists(): - self._status(f"no such path: {path}") - return - # ``..`` segments and symlinks hide re-adds from plain Path equality. - if path.resolve() in {root.resolve() for root in self.request.paths}: - self._status(f"already added: {path}") - return - self.request = replace(self.request, paths=(*self.request.paths, path)) - self.query_one("#in-add-path", Input).value = "" - self.action_rescan() - self._status(f"added {path}") - - @on(Button.Pressed, "#btn-clear-paths") - def _clear_paths(self) -> None: - self.request = replace(self.request, paths=()) - self.action_rescan() - self._status("cleared — add a file or folder to start again") - - @on(Select.Changed, "#sel-preset") - def _preset_changed(self, event: Select.Changed) -> None: - """Apply a preset through the same widgets everything else reads.""" - if not self._widgets_live: - return - chosen = preset_for(None if isinstance(event.value, NoSelection) else str(event.value)) - if chosen is None: - return - self.preset = chosen - # Round-trip through the form so the preset lands in the Plan widgets: - # a preset that only changed a private field would not show up in the - # command preview, and the run would not match what the pane claims. - try: - current = self.collect_request() - except ValueError: - current = self.request - self.apply_request(apply_preset(current, chosen)) - self._status(f"{chosen.label} — {result_class_for(chosen.layer)}") - - @on(Button.Pressed, "#btn-advanced") - def _show_advanced(self) -> None: - self.query_one("#tabs", TabbedContent).active = "tab-plan" - - @on(Button.Pressed, "#btn-start-run") - def _start_run(self) -> None: - self.query_one("#tabs", TabbedContent).active = "tab-run" - self.action_run() - - @on(Button.Pressed, "#btn-save-settings") - def _save_setup(self) -> None: - """Remember the endpoint. Never the key — it has no field to land in.""" - try: - request = self.collect_request() - except ValueError as error: - self._status(f"invalid options: {error}") - return - settings = TuiSettings( - preset=self.preset.key, - rewrite_backend=request.rewrite_backend, - rewrite_base_url=request.rewrite_base_url, - rewrite_model=request.rewrite_model, - rewrite_reasoning_effort=request.rewrite_reasoning_effort, - rewrite_allow_remote=request.rewrite_allow_remote, - ) - try: - where = save_settings(settings) - except OSError as error: - self.query_one("#settings-state", Static).update( - f"[red]could not save: {escape(str(error))}[/]" - ) - return - self.query_one("#settings-state", Static).update( - f"saved to {escape(str(where))} — the API key is never written" - ) - - @on(DataTable.RowHighlighted, "#backend-table") - def _capability_highlighted(self, event: DataTable.RowHighlighted) -> None: - """Show the command that installs the highlighted capability. - - ``refresh_backends`` strips the shared "Install watermark-remover[...]" - preamble so the import that actually failed stays visible in a narrow - detail column — but that preamble was the actionable half. It belongs - here, in something selectable. - """ - if not self._widgets_live: - return - row = event.cursor_row - command = self._install_commands[row] if 0 <= row < len(self._install_commands) else "" - self.query_one("#install-command", TextArea).text = ( - command or "# nothing to install for this row" - ) - - @on(SelectionList.SelectedChanged, "#files-list") - def _files_changed(self) -> None: - if not self._widgets_live: - return - listing = self.query_one("#files-list", SelectionList) - self.selected = [Path(value) for value in listing.selected] - self._update_files_summary() - self._sync_preview() - - @on(Button.Pressed, "#btn-rescan") - def _rescan_pressed(self) -> None: - self.action_rescan() - - # -- inspect ----------------------------------------------------------- - - def action_inspect(self) -> None: - if not self.selected: - self._status("nothing selected") - return - self._status("inspecting…") - self.inspect_worker(list(self.selected)) - - @on(Button.Pressed, "#btn-inspect") - def _inspect_pressed(self) -> None: - self.action_inspect() - - @work(thread=True, exclusive=True, group="inspect") - def inspect_worker(self, paths: list[Path]) -> None: - soft = self.query_one("#cb-inspect-soft", Checkbox).value - blocks = [self._inspect_block(path, soft_binding=soft) for path in paths] - self.call_from_thread(self.query_one("#inspect-report", Static).update, "\n\n".join(blocks)) - self.call_from_thread(self._status, f"inspected {len(paths)} file(s)") - - def _inspect_block(self, path: Path, *, soft_binding: bool, label: str = "") -> str: - # Filenames, findings and error text are not ours: a name like - # "report[1].md" would otherwise be parsed as Rich markup and vanish. - name = escape(path.name) - try: - report = inspect_asset(path, soft_binding=soft_binding) - except Exception as error: # inspection must never take the UI down - return f"[bold]{name}[/] — [red]inspect failed: {escape(str(error))}[/]" - kind = escape(str(report.get("kind", "unknown"))) - lines = [f"[bold]{name}[/]{label} kind: {kind}"] - - if kind == "text": - total = report.get("suspicious_total", 0) - lines.append(f" A hidden Unicode: {total} carrier(s) {format_badge('A')}") - for carrier, count in sorted((report.get("counts") or {}).items()): - if count: - lines.append(f" {escape(str(carrier))} x{count}") - lines.append(self._stylometry_line(path)) - else: - lines.append( - f" M metadata: C2PA={report.get('has_c2pa')} " - f"AI={report.get('has_ai_metadata')} {format_badge('M')}" - ) - for finding in (report.get("findings") or [])[:12]: - lines.append(f" {escape(str(finding))}") - if report.get("note"): - lines.append(f" {escape(str(report['note']))}") - - soft = (report.get("soft_binding") or {}).get("soft_binding") - if soft: - state = "found" if soft.get("found") else "none" - lines.append(f" soft binding: {state} {format_badge('soft-binding')}") - return "\n".join(lines) - - def _stylometry_line(self, path: Path) -> str: - """Layer B's detector view. Always labelled best-effort.""" - try: - text = path.read_text(encoding="utf-8", errors="surrogateescape") - report = score_text_stylometry(text, str(path)) - except Exception as error: - return f" B token watermark: unavailable ({escape(str(error))}) {format_badge('B')}" - markers = len(report.matched_markers or ()) - detail = f" B token watermark: stylometry {report.score:.2f} ({report.confidence_level})" - if report.status != "ok": - # Parentheses, not brackets: brackets are Rich markup and would be - # parsed away, leaving a stray gap where the status should be. - detail += f" ({escape(report.status)})" - return f"{detail}, {markers} AI phrase(s) {format_badge('B')}" - - # -- plan -------------------------------------------------------------- - - def _refresh_api_key_state(self) -> None: - # The key itself is never rendered — only whether one is present. - present = "set" if os.environ.get(REWRITE_ENV_KEY) else "not set" - self.query_one("#api-key-state", Static).update( - f"API key: {present} (read from {REWRITE_ENV_KEY}; never displayed or copied)" - ) - - def _value(self, widget_id: str) -> str | None: - raw = self.query_one(widget_id, Input).value.strip() - return raw or None - - def _number(self, widget_id: str, cast) -> object | None: - """Parse a numeric field: empty means "unset", garbage means invalid. - - Raising rather than returning ``None`` on garbage is the point. A - field the operator typed something into but got wrong must not quietly - become the default — that runs a clean with a timeout or a strength - nobody asked for, and the copyable command would show the substituted - value as though it had been chosen. - """ - raw = self._value(widget_id) - if raw is None: - return None - try: - return cast(raw) - except ValueError as error: - field = widget_id.removeprefix("#in-") - raise ValueError(f"{field}: not a number ({raw})") from error - - def _selected_value(self, widget_id: str) -> str | None: - """Read a Select, treating "nothing chosen" as None. - - Textual's empty sentinel is ``Select.NULL`` (a ``NoSelection``); - ``Select.BLANK`` is the bool ``False`` and is *not* it. Comparing - against the wrong one lets the sentinel leak into the request and - stringify as "Select.NULL". - """ - value = self.query_one(widget_id, Select).value - return None if isinstance(value, NoSelection) else str(value) - - def collect_request(self) -> CleanRequest: - """Read every Plan widget into a CleanRequest. The only request builder.""" - values: dict[str, object] = {} - for binding in PLAN_BINDINGS: - values[binding.field] = binding.read(self) - return replace( - self.request, - paths=tuple(self.selected), - visible_box=self._box(), - # ``--tsapa`` is an alias for ``--rewrite tsapa``; the picker above - # already carries it, so setting both would double the flag. - tsapa=False, - rewrite_disable_thinking=( - True if self.query_one("#cb-disable-thinking", Checkbox).value else None - ), - **values, - ) - - def _box(self) -> tuple[int, int, int, int] | None: - raw = self._value("#in-box") - if not raw: - return None - try: - parts = tuple(int(part) for part in raw.split(",")) - except ValueError: - parts = () - if len(parts) != 4: - self._status("box must be x,y,w,h") - return None - return parts - - @on(Input.Changed) - @on(Select.Changed) - @on(Checkbox.Changed) - def _any_change(self) -> None: - self._sync_preview() - - def _sync_preview(self) -> None: - if not self._widgets_live: - return - try: - request = self.collect_request() - except ValueError as error: - message = f"invalid options: {error}" - self.query_one("#command-preview", Static).update(f"[red]{escape(message)}[/]") - # Clear the copyable box rather than leave the last valid command - # in it: a box that still offers a runnable command while the form - # is invalid hands the operator something the form no longer says. - self.query_one("#command-copyable", TextArea).text = f"# {message}" - return - command = request.command_string() - self.query_one("#command-preview", Static).update(escape(command)) - # The selectable copy of the command is the fallback for terminals that - # ignore OSC 52 — it must always carry the same string as the button. - self.query_one("#command-copyable", TextArea).text = command - self._sync_endpoint_state(request) - self._sync_visible_state(request) - self._sync_start() - - def _sync_endpoint_state(self, request: CleanRequest) -> None: - if request.rewrite_strength is None: - self.query_one("#endpoint-state", Static).update("Layer B off") - return - policy = classify_endpoint( - request.rewrite_base_url, allow_remote=request.rewrite_allow_remote - ) - if policy.loopback: - state = f"[green]loopback {escape(policy.host)}[/]" - elif policy.allowed: - state = f"[yellow]REMOTE {escape(policy.host)} — text leaves this machine[/]" - else: - state = f"[red]blocked: {escape(str(policy.reason))}[/]" - self.query_one("#endpoint-state", Static).update(state) - - def _sync_visible_state(self, request: CleanRequest) -> None: - note = "" - if len(self.selected) > 1 and (request.visible_mask or request.visible_box): - note = "[yellow]mask/box are single-file; use --detect-command for batch[/]" - elif not check_optional("visible").available and request.visible_requested(): - note = check_optional("visible").hint - self.query_one("#visible-state", Static).update(note) - - @on(Select.Changed, "#sel-model") - def _discovered_model_chosen(self, event: Select.Changed) -> None: - """A discovered model fills the free-text field, which stays the source.""" - if not self._widgets_live: - return - if not isinstance(event.value, NoSelection): - self.query_one("#in-model", Input).value = str(event.value) - - @on(Button.Pressed, "#btn-copy-command") - def _copy_command(self) -> None: - command = self.query_one("#command-copyable", TextArea).text - self.copy_to_clipboard(command) - # OSC 52 is fire-and-forget: the terminal never acknowledges it, and - # macOS Terminal.app ignores it outright. Never claim a verified copy. - self.query_one("#copy-state", Static).update( - "copied via OSC 52 — if your terminal ignored it, the box below is selectable" - ) - - # -- candidate picker --------------------------------------------------- - - @on(Button.Pressed, "#btn-candidates") - def _candidates_pressed(self) -> None: - try: - request = self.collect_request() - except ValueError as error: - self._status(f"invalid options: {error}") - return - if request.rewrite_strength is None: - self._status("choose a Layer B strength first") - return - if len(self.selected) != 1: - self._status("candidate picking works on exactly one file") - return - path = self.selected[0] - try: - if resolve_kind(path, request) != "text": - self._status(f"{path.name} is not a text asset") - return - except ValueError as error: - self._status(str(error)) - return - try: - plan = build_rewrite_plan(request) - except RewriteConfigurationError as error: - self._status(str(error)) - return - self.run_worker(self._pick_candidate(path, plan), exclusive=False) - - async def _pick_candidate(self, path: Path, plan) -> None: - policy = classify_endpoint(plan.base_url, allow_remote=plan.allow_remote) - if not policy.loopback and policy.allowed: - approved = await self.push_screen_wait( - ConfirmModal( - "Send document text to a remote endpoint?", - f"{policy.warning}\n\nEndpoint host: {policy.host}", - "Send anyway", - ) - ) - if not approved: - self._status("cancelled: remote endpoint not approved") - return - self._status(f"generating {plan.candidates} candidate(s)…") - self._begin_stream(path.name) - candidates = await self._generate_candidates(path, plan) - if not candidates: - return - chosen = await self.push_screen_wait(CandidateModal(candidates)) - if chosen is None: - self._status("candidate discarded; nothing written") - return - destination = path.with_name(f"{path.stem}.rewritten{path.suffix}") - try: - atomic_write_text(destination, candidates[chosen].text) - except OSError as error: - self._status(f"write failed: {error}") - return - self._log( - f"wrote {destination} from candidate {chosen + 1} " - f"{format_badge('B')} — no detector guarantee" - ) - self._status(f"wrote {destination.name}") - - @work(thread=True, exclusive=True, group="candidates") - async def _generate_candidates(self, path: Path, plan): - """Generate off the UI thread; a model call must never block the terminal.""" - try: - text = path.read_text(encoding="utf-8", errors="surrogateescape") - return generate_candidates(text, plan, on_token=self._append_stream) - except Exception as error: - self.call_from_thread( - self._status, f"candidate generation failed: {escape(str(error))}" - ) - return [] - - def _append_stream(self, fragment: str) -> None: - """Token sink: called from the reader thread, so hop to the UI thread.""" - self.call_from_thread(self._write_stream, fragment) - - def _write_stream(self, fragment: str) -> None: - if not self._widgets_live: - return - view = self.query_one("#stream-view", TextArea) - # Bounded on purpose: a long document would otherwise grow the widget's - # document without limit while the run is still going. - view.text = (view.text + fragment)[-STREAM_VIEW_CHARS:] - view.scroll_end(animate=False) - - def _begin_stream(self, filename: str | None) -> None: - """Reset the stream box, and label why it is empty when nothing streams. - - An always-visible box that only ever fills on one code path reads as - broken; naming the reason is cheaper than hiding it. - """ - if not self._widgets_live: - return - view = self.query_one("#stream-view", TextArea) - view.text = "" - view.border_title = IDLE_STREAM_TITLE if filename is None else f"generating {filename}" - - # -- backends ---------------------------------------------------------- - - @on(Button.Pressed, "#btn-refresh-backends") - def _refresh_pressed(self) -> None: - self.refresh_backends() - - @on(TabbedContent.TabActivated, "#tabs") - def _tab_activated(self, event: TabbedContent.TabActivated) -> None: - """Rebuild the capability table when it comes into view. - - Its Layer B rows are derived from the *current* plan, so a table left - as it was at mount reports "no base URL set" over a plan that has one — - stale state presented as fact. - """ - if not self._widgets_live: - return - if event.pane.id == "tab-start": - self.refresh_backends() - # A pane that was hidden until now has only just been given a width. - self.call_after_refresh(self._relayout_tables) - - def refresh_backends(self, *, recheck: bool = True) -> None: - """Rebuild the capability table. ``recheck`` re-imports the extras.""" - if not self._widgets_live: - return - if recheck or self._extra_rows is None: - self._extra_rows = [ - ( - ( - f"extra: {extra}", - "available" if availability.available else "missing", - # The hint's shared "install watermark-remover[...]" - # preamble repeats on every row and pushes the part that - # differs — the import that actually failed — off the - # visible width. The preamble is not lost: it becomes - # the row's install command. - availability.hint.replace("Reason: ", "").split(". ")[-1], - ), - ("" if availability.available else f'pip install "watermark-remover[{extra}]"'), - ) - for extra, availability in ((name, check_optional(name)) for name in KNOWN_EXTRAS) - ] - rows: list[tuple[str, ...]] = [row for row, _ in self._extra_rows] - commands: list[str] = [command for _, command in self._extra_rows] - # A half-typed number must not blank the capability pane: fall back to - # the last valid request so the endpoint row still says something true. - try: - request = self.collect_request() - except ValueError: - request = self.request - policy = classify_endpoint( - request.rewrite_base_url, allow_remote=request.rewrite_allow_remote - ) - rows.append( - ( - "layer B endpoint", - "allowed" if policy.allowed else "blocked", - policy.reason or f"host {policy.host or '-'} (loopback={policy.loopback})", - ) - ) - commands.append("") - rows.append( - ( - "layer B api key", - "set" if os.environ.get(REWRITE_ENV_KEY) else "not set", - f"{REWRITE_ENV_KEY} — never displayed", - ) - ) - # An export line, not the key: this is the shape of the thing to set, - # and the value stays somewhere this process never reads it back out. - commands.append(f"export {REWRITE_ENV_KEY}=... # set it in your shell, not here") - if self._last_probe is not None: - rows.append( - ( - f"probe: {escape(self._last_probe.backend)}", - "reachable" if self._last_probe.reachable else "unreachable", - escape(self._last_probe.summary), - ) - ) - commands.append("") - self._install_commands = commands - self._set_table_rows("#backend-table", rows) - - @on(Button.Pressed, "#btn-probe") - def _probe_pressed(self) -> None: - try: - request = self.collect_request() - except ValueError as error: - self._status(f"invalid options: {error}") - return - backend = request.rewrite_backend - if not backend: - self._status("choose a Layer B backend first") - return - self._status(f"probing {backend}…") - self.probe_worker(backend, request.rewrite_base_url, request.rewrite_allow_remote) - - @work(thread=True, exclusive=True, group="probe") - def probe_worker(self, backend: str, base_url: str | None, allow_remote: bool) -> None: - probe = probe_backend( - backend, - base_url, - api_key=os.environ.get(REWRITE_ENV_KEY), - allow_remote=allow_remote, - ) - self.call_from_thread(self._apply_probe, probe) - - def _apply_probe(self, probe) -> None: - self._last_probe = probe - if not self._widgets_live: - return - self._status(f"{probe.backend}: {probe.summary}") - if probe.models: - model_select = self.query_one("#sel-model", Select) - model_select.set_options([(name, name) for name in probe.models]) - self.refresh_backends() - - # -- run --------------------------------------------------------------- - - def action_run(self) -> None: - if self._clean_running: - self._status("already running") - return - try: - request = self.collect_request() - except ValueError as error: - self._status(f"invalid options: {error}") - return - if not request.paths: - self._status("nothing selected") - return - self.run_worker(self._confirm_then_run(request), exclusive=False) - - @on(Button.Pressed, "#btn-run") - def _run_pressed(self) -> None: - self.action_run() - - @on(Button.Pressed, "#btn-cancel") - def _cancel_pressed(self) -> None: - self._cancel_requested = True - self._status("stopping after the current file…") - - async def _confirm_then_run(self, request: CleanRequest) -> None: - """Gate every irreversible or outward-facing aspect before writing.""" - policy = classify_endpoint( - request.rewrite_base_url, allow_remote=request.rewrite_allow_remote - ) - if request.rewrite_strength is not None and not policy.loopback and policy.allowed: - # Render the rewrite path's own warning verbatim: the operator is - # agreeing to that statement, not to a paraphrase of it. - approved = await self.push_screen_wait( - ConfirmModal( - "Send document text to a remote endpoint?", - f"{policy.warning}\n\nEndpoint host: {policy.host}\n" - "The full text of every selected file will be sent there.", - "Send anyway", - ) - ) - if not approved: - self._status("cancelled: remote endpoint not approved") - return - - if request.in_place: - approved = await self.push_screen_wait( - ConfirmModal( - "Overwrite the source files?", - f"{len(request.paths)} file(s) will be rewritten in place. " - "A .bak backup is created for each.", - "Overwrite", - ) - ) - if not approved: - self._status("cancelled: in-place not approved") - return - - if request.strip_semantic_format: - approved = await self.push_screen_wait( - ConfirmModal( - "Strip semantic formatting?", - "Contextual ZWJ, variation selectors, and balanced bidi controls " - "are preserved by default because removing them can change how " - "text renders or what it means.", - "Strip anyway", - ) - ) - if not approved: - self._status("cancelled: semantic stripping not approved") - return - - estimate = estimate_rewrite_seconds(request, len(request.paths)) - if should_confirm_cost(request, len(request.paths)): - approved = await self.push_screen_wait( - ConfirmModal( - "This run calls a model repeatedly", - f"{len(request.paths)} file(s) run sequentially.\n" - f"Worst case: {format_duration(estimate)} " - f"(files x calls x {request.rewrite_timeout or DEFAULT_REWRITE_TIMEOUT:.0f}s " - "timeout).\nCancelling stops after the current file; an in-flight " - "model call cannot be interrupted.", - "Run", - ) - ) - if not approved: - self._status("cancelled: cost not approved") - return - - self.clean_worker(request) - - @work(thread=True, exclusive=True, group="clean") - def clean_worker(self, request: CleanRequest) -> None: - self._cancel_requested = False - self.call_from_thread(self._set_running, True) - self.call_from_thread(self._set_table_rows, "#run-table", []) - - batch = len(request.paths) > 1 - # Preflight every destination before the first write: a failing plan - # aborts the whole run rather than cleaning the first N files. - try: - # The selection is already resolved to explicit files, but it still - # goes through select_inputs so the TUI inherits the CLI's symlink - # and regular-file refusals rather than reimplementing them. - selection = select_inputs( - request.paths, - recursive=request.recursive, - pattern=request.glob, - extensions=request.allowed_extensions(SUPPORTED_EXTENSIONS), - ) - work_items = plan_work(selection.items, request, batch) - except CleanPlanPreflightError as error: - self._fail_preflight( - f"preflight failed on {escape(str(error.path))}: {escape(str(error))}" - ) - return - except Exception as error: - # Anything raised before the first write means nothing was written; - # surface it rather than letting the worker die silently. - self._fail_preflight(f"preflight failed: {escape(str(error))}") - return - - if request.dry_run: - # Mirrors clean_file.main: a dry run describes and returns before - # any directory is created or any byte is written. - self.call_from_thread(self._render_dry_run, request, work_items) - return - - if batch and request.output and not request.in_place: - request.output.mkdir(parents=True, exist_ok=True) - - # json=True keeps run_clean_item silent so this UI renders the payload - # itself instead of the CLI printing over the terminal. - silent = replace(request, json=True, quiet=True) - done = 0 - for item, output, plan in work_items: - if self._cancel_requested: - self.call_from_thread(self._log, "[yellow]stopped after the current file[/]") - break - before = self._read_text(item.path) - self.call_from_thread(self._status, f"cleaning {item.path.name}…") - streaming = plan.text.rewrite_plan is not None - self.call_from_thread(self._begin_stream, item.path.name if streaming else None) - payload = run_clean_item( - item.path, - output, - silent, - plan, - on_token=self._append_stream if streaming else None, - ) - done += 1 - self.call_from_thread(self._render_result, request, item.path, payload, before) - - self.call_from_thread(self._set_running, False) - self.call_from_thread(self._status, f"done: {done} of {len(work_items)} file(s)") - self.call_from_thread(self._record_history, request) - - def _render_dry_run(self, request: CleanRequest, work_items) -> None: - """Show what a visible-mark clean would do. Nothing is written.""" - self._begin_stream(None) - for item, output, plan in work_items: - payload = dry_run_payload(item.path, output, plan, request.in_place) - self._add_table_row( - "#run-table", - escape(item.path.name), - "image", - "dry-run", - format_badge("V"), - "-", - escape(str(payload["output"])), - ) - self._log(f"[bold]dry-run {escape(str(payload['input']))}[/]") - for action in payload["actions"]: - self._log(f" - {escape(str(action))}") - self._set_running(False) - self._status(f"dry run: {len(work_items)} file(s) described, nothing written") - self._record_history(request) - - def _fail_preflight(self, message: str) -> None: - self.call_from_thread(self._log, f"[red]{message}[/]") - self.call_from_thread(self._status, f"{message} — nothing was written") - self.call_from_thread(self._set_running, False) - - def _set_running(self, running: bool) -> None: - self._clean_running = running - if not self._widgets_live: - return - self.query_one("#btn-run", Button).disabled = running - self.query_one("#btn-cancel", Button).disabled = not running - - @staticmethod - def _read_text(path: Path) -> str | None: - try: - return path.read_text(encoding="utf-8", errors="surrogateescape") - except (OSError, UnicodeDecodeError): - return None - - def _render_result( - self, - request: CleanRequest, - path: Path, - payload: dict, - before: str | None, - ) -> None: - if not self._widgets_live: - return - kind = payload.get("kind", "unknown") - failed = payload.get("exit_code", 0) != 0 - # The badge follows the layer that did the work, never the outcome. - layer = layer_for_result(request, str(kind)) - note = payload.get("error") or "" - skipped = payload.get("skipped_text_transforms") - if skipped: - note = f"skipped {', '.join(skipped)}" - self._add_table_row( - "#run-table", - escape(path.name), - escape(str(kind)), - "error" if failed else "written", - "-" if failed else result_class_for(layer), - "yes" if payload.get("residual") else "", - escape(note), - ) - if failed: - self._log(f"[red]{escape(path.name)}: {escape(str(payload.get('error')))}[/]") - return - if skipped: - # Carries the file name, so it is operator-controlled text: escape - # it or a name like report[1].md is swallowed as a style tag. - self._log(f"[yellow]{escape(describe_dropped_text_transforms(request, kind, path))}[/]") - output = payload.get("output") - if output and before is not None: - after = self._read_text(Path(output)) - if after is not None and after != before: - diff = "\n".join( - difflib.unified_diff( - before.splitlines(), - after.splitlines(), - fromfile=f"{path.name} (before)", - tofile=f"{path.name} (after)", - lineterm="", - n=2, - ) - ) - self.query_one("#diff-view", TextArea).text = diff or "(no line-level change)" - # Re-inspect: the point of the loop is the measured after-state. - self._log(self._inspect_block(Path(output), soft_binding=False, label=" (after)")) - - def _record_history(self, request: CleanRequest) -> None: - summary_parts = [f"{len(request.paths)} file(s)"] - layers = [] - if request.nfkc or request.aggressive_homoglyphs or request.strip_semantic_format: - layers.append("A+") - if request.rewrite_strength: - layers.append(f"B({request.rewrite_strength})") - if request.visible_requested(): - layers.append("V") - if not request.keep_non_ai_metadata: - layers.append("M") - if layers: - summary_parts.append("·".join(layers)) - entry = HistoryEntry( - when=time.strftime("%H:%M"), - summary=" · ".join(summary_parts), - command=request.command_string(), - request=request, - ) - self.history.insert(0, entry) - if not self._widgets_live: - return - self._set_table_rows("#history-table", [(item.when, item.summary) for item in self.history]) - self.query_one("#history-copyable", TextArea).text = entry.command - - # -- history ----------------------------------------------------------- - - @on(DataTable.RowHighlighted, "#history-table") - def _history_highlighted(self, event: DataTable.RowHighlighted) -> None: - if not self._widgets_live: - return - if 0 <= event.cursor_row < len(self.history): - self.query_one("#history-copyable", TextArea).text = self.history[ - event.cursor_row - ].command - - @on(Button.Pressed, "#btn-history-copy") - def _history_copy(self) -> None: - command = self.query_one("#history-copyable", TextArea).text - if not command: - return - self.copy_to_clipboard(command) - self._status("copied via OSC 52 — the box below is selectable if it was ignored") - - @on(Button.Pressed, "#btn-history-reuse") - def _history_reuse(self) -> None: - row = self.query_one("#history-table", DataTable).cursor_row - if not (0 <= row < len(self.history)): - return - self.apply_request(self.history[row].request) - self._status("plan repopulated from history") - - def apply_request(self, request: CleanRequest) -> None: - """Push a saved request back into the Plan widgets. - - Driven by the same table ``collect_request`` reads, so "Reuse" cannot - quietly drop an option that only one of the two knows about. - """ - if not self._widgets_live: - return - for binding in PLAN_BINDINGS: - binding.write(self, getattr(request, binding.field)) - self.query_one("#cb-disable-thinking", Checkbox).value = bool( - request.rewrite_disable_thinking - ) - self.query_one("#in-box", Input).value = ( - ",".join(str(part) for part in request.visible_box) if request.visible_box else "" - ) - self._sync_preview() - - -@dataclass(frozen=True) -class PlanBinding: - """One Plan widget bound to one ``CleanRequest`` field. - - Declaring the binding once is what keeps reading the form and repopulating - it symmetric. Two hand-written lists drift, and the drift is invisible: - an option that only ``collect_request`` knows about is silently dropped by - "Reuse", and an option only ``apply_request`` knows about is never read. - A test asserts every field is bound here or listed as deliberately absent. - """ - - selector: str - field: str - kind: str - default: object = None - - def read(self, app: WatermarkTuiApp) -> object: - if self.kind == "bool": - return app.query_one(self.selector, Checkbox).value - if self.kind == "select": - return app._selected_value(self.selector) or self.default - if self.kind == "text": - return app._value(self.selector) or self.default - if self.kind == "path": - raw = app._value(self.selector) - return Path(raw) if raw else None - if self.kind in ("int", "float"): - # ``is None``, not truthiness: 0, 0.0 and a temperature of 0.0 are - # all values an operator can legitimately mean, and falling back to - # the default for them silently runs something else. - value = app._number(self.selector, int if self.kind == "int" else float) - return self.default if value is None else value - raise AssertionError(f"unknown binding kind: {self.kind}") - - def write(self, app: WatermarkTuiApp, value: object) -> None: - if self.kind == "bool": - app.query_one(self.selector, Checkbox).value = bool(value) - return - if self.kind == "select": - app.query_one(self.selector, Select).value = value if value is not None else Select.NULL - return - app.query_one(self.selector, Input).value = "" if value is None else str(value) - - -#: Every ``CleanRequest`` field the Plan pane owns. Fields absent from this -#: table are listed in ``UNBOUND_REQUEST_FIELDS`` with the reason. -PLAN_BINDINGS: tuple[PlanBinding, ...] = ( - # Routing - PlanBinding("#sel-force-type", "force_type", "select", "auto"), - PlanBinding("#cb-force-text", "force_text", "bool"), - PlanBinding("#in-audit", "audit", "text"), - # Layer A - PlanBinding("#cb-nfkc", "nfkc", "bool"), - PlanBinding("#cb-homoglyphs", "aggressive_homoglyphs", "bool"), - PlanBinding("#cb-semantic", "strip_semantic_format", "bool"), - # Layer M - PlanBinding("#cb-keep-meta", "keep_non_ai_metadata", "bool"), - PlanBinding("#cb-soft", "soft_binding", "bool"), - # Layer B - PlanBinding("#sel-rewrite", "rewrite", "select"), - PlanBinding("#sel-backend", "rewrite_backend", "select"), - PlanBinding("#in-base-url", "rewrite_base_url", "text"), - PlanBinding("#in-model", "rewrite_model", "text"), - PlanBinding("#in-candidates", "rewrite_candidates", "int"), - PlanBinding("#in-temperature", "rewrite_temperature", "float"), - PlanBinding("#in-rewrite-timeout", "rewrite_timeout", "float"), - PlanBinding("#sel-effort", "rewrite_reasoning_effort", "select"), - PlanBinding("#cb-allow-remote", "rewrite_allow_remote", "bool"), - PlanBinding("#in-lang", "rewrite_lang", "text"), - PlanBinding("#in-original-lang", "rewrite_original_lang", "text"), - PlanBinding("#in-generations", "tsapa_generations", "int", 5), - PlanBinding("#in-population", "tsapa_population", "int", 12), - # Character perturbation - PlanBinding("#cb-perturb", "char_perturb", "bool"), - PlanBinding("#sel-perturb-mode", "char_mode", "select", "zero-width"), - PlanBinding("#in-perturb-strength", "char_strength", "float", 0.1), - PlanBinding("#in-seed", "seed", "int"), - # Layer V - PlanBinding("#in-mask", "visible_mask", "path"), - PlanBinding("#sel-visible-backend", "visible_backend", "select", "texture"), - PlanBinding("#in-dilate", "dilate", "int"), - PlanBinding("#in-detect-command", "detect_command", "text"), - PlanBinding("#in-inpaint-command", "inpaint_command", "text"), - PlanBinding( - "#in-visible-prompt", - "visible_prompt", - "text", - "Remove watermark, fill with background", - ), - PlanBinding("#in-timeout", "timeout", "float", 1800.0), - PlanBinding("#sel-quality", "quality", "select", "balanced"), - PlanBinding("#cb-synthid", "remove_synthid", "bool"), - PlanBinding("#in-synthid-strength", "synthid_strength", "float", 0.6), - PlanBinding("#cb-dry-run", "dry_run", "bool"), - # Image degradation - PlanBinding("#sel-degrade", "degrade", "select"), - PlanBinding("#sel-morpho", "morpho", "select"), - PlanBinding("#in-degrade-strength", "degrade_strength", "float", 0.6), - PlanBinding("#in-degrade-seed", "degrade_seed", "int"), - # Output - PlanBinding("#in-output", "output", "path"), - PlanBinding("#cb-in-place", "in_place", "bool"), - PlanBinding("#cb-artifacts", "keep_artifacts", "bool"), - PlanBinding("#cb-wmct", "wmct_marker", "bool"), -) - -#: Fields the Plan pane deliberately does not own, and why. -UNBOUND_REQUEST_FIELDS: dict[str, str] = { - "paths": "the Files pane's selection", - "recursive": "the Files pane's filters", - "glob": "the Files pane's filters", - "extensions": "the Files pane's filters", - "visible_box": "parsed from x,y,w,h rather than read straight through", - "rewrite_disable_thinking": "tri-state: unchecked means unset, not False", - "tsapa": "an alias for --rewrite tsapa, which the strength picker carries", - "rewrite_api_key": "read from the environment; never rendered or persisted", - "json": "CLI presentation; the TUI renders payloads itself", - "quiet": "CLI presentation; the TUI renders payloads itself", -} diff --git a/skills/remove-ai-marks/scripts/tui_bridge.py b/skills/remove-ai-marks/scripts/tui_bridge.py new file mode 100644 index 0000000..f77450d --- /dev/null +++ b/skills/remove-ai-marks/scripts/tui_bridge.py @@ -0,0 +1,833 @@ +#!/usr/bin/env python3 +"""The Python half of wm-tui: a JSON-Lines server over stdin and stdout. + +The TypeScript frontend (``tui/``) spawns this with ``$WM_TUI_PYTHON`` and +talks to it one JSON object per line; ``tui/PROTOCOL.md`` is the contract. +The frontend owns the terminal, and this process owns every decision about a +clean. The decisions themselves live in ``tui_core`` so they can be tested +without a process; this module is the transport around them. + +Three properties this module exists to hold: + +* **Frames are never corrupted.** The pipeline prints: warnings, progress, a + stray debug line, sometimes while it is still being imported. Before any + pipeline import, fd 1 is duplicated for frames only, and fd 1 plus + ``sys.stdout`` are pointed at stderr, so anything else that writes to + "stdout", from Python or from C, lands in the frontend's log instead of in + the middle of a frame. +* **The frontend never freezes.** Requests run on worker threads, so a + ``cancel`` or ``detect_endpoints`` is answered while a clean runs. Only one + clean or inspect runs at a time; a second is answered ``busy`` rather than + queued, because a queued clean the operator cannot see is a surprise write. +* **The key never leaves.** Nothing here puts the API key in a frame, and the + frame writer redacts it from every payload as a last line of defence. +""" + +from __future__ import annotations + +import contextlib +import json +import os +import sys +import threading +import time +import traceback +from collections.abc import Callable, Mapping +from pathlib import Path +from typing import Any, BinaryIO + + +def claim_stdout() -> BinaryIO: + """Keep fd 1 for frames; send every other write to stderr. + + ``os.dup2`` moves the *file descriptor*, so C extensions and subprocesses + that inherit fd 1 are redirected too, not only Python's ``print``. + """ + with contextlib.suppress(Exception): + sys.stdout.flush() + frames_fd = os.dup(1) + os.dup2(2, 1) + sys.stdout = sys.stderr + return os.fdopen(frames_fd, "wb") + + +# Claimed before the imports below: the pipeline probes optional libraries at +# import time, and one that prints would otherwise write ahead of ``ready``. +_FRAMES = claim_stdout() if __name__ == "__main__" else None + +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +import tui as launcher +from asset_kind import classify_asset +from batch_inputs import InputItem, InputSelection +from clean_asset import CleanPlan +from clean_file import dry_run_payload, run_clean_item +from clean_request import CleanRequest, select_request_inputs +from inspect_file import inspect_asset +from rewrite_text import LIVE_REWRITE_BACKENDS +from tui_core import ( + API_KEY_ENV, + PRESETS, + BadRequest, + InvalidOptions, + apply_settings_update, + compose_request, + confirm_gates, + describe_output, + detect_local_endpoints, + display_path, + environment_checks, + estimate_rewrite_seconds, + failed_finding, + finding_from_report, + history_summary, + layer_for_result, + load_settings, + plan_layer, + plan_warnings, + preflight_work, + preview_command, + result_class_for, + result_lines, + save_settings, + settings_path, + should_onboard, + text_for_diff, + unified_diff, +) + +#: Minimum interval between two ``token`` frames. A model can emit hundreds +#: of fragments a second; one frame each would spend the frontend's render +#: budget on JSON parsing. +TOKEN_INTERVAL = 0.05 + +#: Most files ``plan`` lists. ``plan`` runs on every keystroke (debounced); +#: pointing it at a home directory must not produce a megabyte frame. +PLAN_FILES_LIMIT = 2000 + +#: How long exit waits for in-flight requests to answer. +DRAIN_SECONDS = 5.0 + +#: Keys shorter than this are not redacted: replacing every "ab" in every +#: frame would mangle the protocol, and a string that short is not a secret. +REDACT_MIN_LENGTH = 8 +REDACTED = "[redacted]" + + +class BridgeError(Exception): + """A request failed with a PROTOCOL error code.""" + + def __init__(self, code: str, message: str, data: Mapping[str, Any] | None = None) -> None: + super().__init__(message) + self.code = code + self.message = message + self.data = dict(data) if data is not None else None + + +# --- frames ------------------------------------------------------------------ + + +def _redact(value: Any, secret: str) -> Any: + if isinstance(value, str): + return value.replace(secret, REDACTED) if secret in value else value + if isinstance(value, dict): + return {_redact(key, secret): _redact(item, secret) for key, item in value.items()} + if isinstance(value, (list, tuple)): + return [_redact(item, secret) for item in value] + return value + + +def _error_frame( + rid: Any, code: str, message: str, data: Mapping[str, Any] | None = None +) -> dict[str, Any]: + error: dict[str, Any] = {"code": code, "message": message} + if data is not None: + error["data"] = dict(data) + return {"id": rid, "error": error} + + +class FrameWriter: + """Serialises frames onto the protocol stream, one whole line at a time. + + Worker threads, the token flusher and the reader thread all write; the lock + is what keeps two frames from interleaving mid-line. ASCII-escaped JSON is + still UTF-8, and it survives file names that are not valid Unicode (lone + surrogates from ``surrogateescape``), which raw UTF-8 would refuse. + """ + + def __init__(self, stream: BinaryIO, *, redact: bool = True) -> None: + self._stream = stream + self._lock = threading.Lock() + self._redact = redact + self._closed = False + + def write(self, frame: dict[str, Any]) -> None: + if self._redact: + secret = os.environ.get(API_KEY_ENV) or "" + if len(secret) >= REDACT_MIN_LENGTH: + # Payloads only: the envelope (id, event names) is ours, and + # redacting a substring of "file_done" would break framing. + frame = { + key: _redact(value, secret) if key in ("result", "error", "data") else value + for key, value in frame.items() + } + line = json.dumps(frame, ensure_ascii=True, default=str) + "\n" + with self._lock: + if self._closed: + return + try: + self._stream.write(line.encode("ascii")) + self._stream.flush() + except (OSError, ValueError): + # The frontend is gone; the reader will see EOF and exit. + self._closed = True + + +class TokenCoalescer: + """Batch Layer B fragments into at most one frame per ``TOKEN_INTERVAL``. + + Emission happens on one flusher thread only, so frames stay in order. + ``close`` stops and joins that thread and then emits whatever is left, + waiting out the interval first, so the last fragment is never lost, never + sent after the file's ``file_done``, and never breaks the rate limit. + """ + + def __init__(self, emit: Callable[[str], None]) -> None: + self._emit = emit + self._parts: list[str] = [] + self._lock = threading.Lock() + self._last = float("-inf") + self._stop = threading.Event() + self._thread = threading.Thread(target=self._loop, name="wm-tui-tokens", daemon=True) + self._thread.start() + + def add(self, fragment: str) -> None: + with self._lock: + self._parts.append(fragment) + + def _take(self) -> str: + with self._lock: + text = "".join(self._parts) + self._parts.clear() + return text + + def _send(self, text: str) -> None: + self._last = time.monotonic() + self._emit(text) + + def _loop(self) -> None: + while not self._stop.wait(TOKEN_INTERVAL / 5): + if time.monotonic() - self._last >= TOKEN_INTERVAL and (text := self._take()): + self._send(text) + + def close(self) -> None: + self._stop.set() + self._thread.join() + text = self._take() + if not text: + return + wait = TOKEN_INTERVAL - (time.monotonic() - self._last) + if wait > 0: + time.sleep(wait) + self._send(text) + + +# --- helpers ----------------------------------------------------------------- + + +def _state(params: Mapping[str, Any]) -> Mapping[str, Any]: + state = params.get("state") + if not isinstance(state, Mapping): + raise BadRequest("params.state must be an object") + return state + + +def _select(request: CleanRequest) -> InputSelection: + """The request's files, with a selector refusal reported as ``invalid_options``.""" + try: + return select_request_inputs(request) + except ValueError as error: + raise InvalidOptions(str(error)) from error + + +def _file_done( + index: int, + display: str, + layer: str, + *, + output: str | None, + exit_code: int, + before: Mapping[str, int], + after: Mapping[str, int], + lines: list[str], + diff: str | None, + error: str | None, +) -> dict[str, Any]: + return { + "index": index, + "display": display, + "output": output, + "layer": layer, + "result_class": result_class_for(layer), + "exit_code": exit_code, + "before": dict(before), + "after": dict(after), + "lines": lines, + "diff": diff, + "error": error, + } + + +def _require_confirmed(gates: list[dict[str, str]], confirmed: list[str]) -> None: + """Refuse with ``needs_confirm`` unless every gate was agreed to.""" + missing = [gate for gate in gates if gate["kind"] not in confirmed] + if missing: + raise BridgeError( + "needs_confirm", + f"confirm before cleaning: {', '.join(gate['kind'] for gate in missing)}", + {"confirm": missing}, + ) + + +def _kinds(work: list[tuple[InputItem, Path | None, CleanPlan]]) -> list[str]: + """Each planned file's resolved kind (``plan_work`` records it on the plan).""" + return [plan.forced_kind for _item, _output, plan in work] + + +def _parse_launch_argv(raw: str | None) -> Any: + """Re-parse the launcher's arguments, falling back to defaults. + + The launcher already validated them; a failure here means the variable was + set by hand, and the right response is a working UI on defaults plus a + line in the log, not a dead bridge. + """ + parser = launcher.build_parser() + defaults = parser.parse_args([]) + if not raw: + return defaults + try: + argv = json.loads(raw) + if not isinstance(argv, list) or not all(isinstance(item, str) for item in argv): + raise ValueError("WM_TUI_ARGV must be a JSON list of strings") + + def fail(message: str) -> None: + raise ValueError(message) + + parser.error = fail # type: ignore[method-assign] + return parser.parse_args(argv) + except (ValueError, SystemExit) as error: + print(f"wm-tui-bridge: ignoring WM_TUI_ARGV: {error}", file=sys.stderr) + return defaults + + +# --- the server -------------------------------------------------------------- + + +class Bridge: + """Dispatches requests to handlers, one worker thread per request.""" + + #: Methods answered on the reader thread: they must never wait behind work. + INLINE = frozenset({"cancel", "shutdown"}) + #: Methods that touch many files; one at a time. + EXCLUSIVE = frozenset({"clean", "inspect"}) + + def __init__(self, writer: FrameWriter, *, launch_argv: str | None = None) -> None: + self.writer = writer + self.launch = _parse_launch_argv(launch_argv) + self.cancel = threading.Event() + self.stopping = False + self._exclusive = threading.Lock() + self._history: list[dict[str, str]] = [] + self._history_lock = threading.Lock() + self._workers: set[threading.Thread] = set() + self._workers_lock = threading.Lock() + self._handlers: dict[str, Callable[[Any, Mapping[str, Any]], Any]] = { + "hello": self.hello, + "plan": self.plan, + "inspect": self.inspect, + "clean": self.clean, + "cancel": self.cancel_request, + "detect_endpoints": self.detect_endpoints, + "checks": self.checks, + "save_settings": self.save_settings, + "history": self.history, + "shutdown": self.shutdown, + } + + # -- framing ----------------------------------------------------------- + + def event(self, rid: Any, name: str, data: Mapping[str, Any]) -> None: + self.writer.write({"id": rid, "event": name, "data": dict(data)}) + + def fail(self, rid: Any, code: str, message: str) -> None: + self.writer.write(_error_frame(rid, code, message)) + + def ready(self) -> None: + self.writer.write({"event": "ready", "data": {"version": launcher.package_version()}}) + + # -- dispatch ---------------------------------------------------------- + + def handle_line(self, line: str | bytes) -> threading.Thread | None: + """Parse one request and start it. Returns the worker, for tests to join.""" + try: + message = json.loads(line) + except (ValueError, UnicodeDecodeError) as error: + self.fail(None, "bad_request", f"not valid JSON: {error}") + return None + if not isinstance(message, dict): + self.fail(None, "bad_request", "a request must be a JSON object") + return None + rid = message.get("id") + if rid is not None and (isinstance(rid, bool) or not isinstance(rid, (int, str))): + self.fail(None, "bad_request", "id must be a number, a string or null") + return None + method = message.get("method") + handler = self._handlers.get(method) if isinstance(method, str) else None + if handler is None: + self.fail(rid, "bad_request", f"unknown method: {method!r}") + return None + params = message.get("params") + if params is None: + params = {} + if not isinstance(params, dict): + self.fail(rid, "bad_request", "params must be an object") + return None + + if method in self.INLINE: + self.writer.write(self._answer(rid, handler, params)) + return None + exclusive = method in self.EXCLUSIVE + if exclusive: + # Claimed here, on the reader thread, not in the worker: otherwise + # a cancel sent right after the clean could be cleared by a worker + # that had not started yet, and two cleans could both get in. + if not self._exclusive.acquire(blocking=False): + self.fail(rid, "busy", "a clean or inspect is already running") + return None + self.cancel.clear() + worker = threading.Thread( + target=self._work, + args=(rid, handler, params, exclusive), + name=f"wm-tui-{method}", + daemon=True, + ) + with self._workers_lock: + self._workers.add(worker) + worker.start() + return worker + + def _answer( + self, rid: Any, handler: Callable[[Any, Mapping[str, Any]], Any], params: Mapping[str, Any] + ) -> dict[str, Any]: + """Run a handler and return its response frame. Never raises. + + ``BaseException`` on purpose: parts of the pipeline still raise + ``SystemExit`` for a CLI refusal, and inside a thread that would end + the worker silently, never answering and never releasing the + exclusive lock, which leaves every later clean ``busy`` forever. + """ + try: + return {"id": rid, "result": handler(rid, params)} + except BridgeError as error: + return _error_frame(rid, error.code, error.message, error.data) + except BadRequest as error: + return _error_frame(rid, "bad_request", str(error)) + except InvalidOptions as error: + return _error_frame(rid, "invalid_options", str(error)) + except BaseException as error: + traceback.print_exc(file=sys.stderr) + return _error_frame(rid, "internal", f"{type(error).__name__}: {error}") + + def _work( + self, + rid: Any, + handler: Callable[[Any, Mapping[str, Any]], Any], + params: Mapping[str, Any], + exclusive: bool, + ) -> None: + frame = self._answer(rid, handler, params) + # Released before the response is written: a frontend that sends the + # next clean the moment it reads this answer must not be told busy. + if exclusive: + self._exclusive.release() + self.writer.write(frame) + # Only now, so drain waits for the response to be written. + with self._workers_lock: + self._workers.discard(threading.current_thread()) + + def serve(self, stdin: BinaryIO) -> int: + """Read requests until EOF or ``shutdown``. Returns the exit code.""" + self.ready() + while not self.stopping: + line = stdin.readline() + if not line: + break + if line.strip(): + self.handle_line(line) + self.drain() + return 0 + + def drain(self, timeout: float = DRAIN_SECONDS) -> None: + """Let in-flight requests answer before exiting, within a bound. + + A clean is cancelled first, so it stops after the file it is on: the + frontend that asked for it is gone or leaving, and nobody is watching + the rest. The bound is there because an in-flight model call cannot + be interrupted, and a bridge the frontend has let go of must not + linger for a whole rewrite timeout. + """ + self.cancel.set() + deadline = time.monotonic() + timeout + with self._workers_lock: + workers = list(self._workers) + for worker in workers: + worker.join(max(0.0, deadline - time.monotonic())) + + # -- methods ----------------------------------------------------------- + + def hello(self, rid: Any, params: Mapping[str, Any]) -> dict[str, Any]: + target = settings_path() + args = self.launch + flags: list[str] = [] + if args.recursive: + flags.append("--recursive") + if args.glob != "*": + flags += ["--glob", args.glob] + if args.extensions: + flags += ["--extensions", args.extensions] + return { + "version": launcher.package_version(), + "presets": [preset.to_dict() for preset in PRESETS], + "settings": load_settings(target).to_wire(), + "settings_path": str(target), + "onboard": should_onboard( + settings_exist=target.exists(), force=args.setup, skip=args.no_setup + ), + "initial": {"paths": list(args.path or ["."]), "flags": flags}, + "backends": list(LIVE_REWRITE_BACKENDS), + } + + def plan(self, rid: Any, params: Mapping[str, Any]) -> dict[str, Any]: + """The cheap preview. Reads, never writes.""" + state = _state(params) + try: + composition = compose_request(state) + except InvalidOptions as error: + return { + "ok": False, + "error": str(error), + "command": preview_command(state), + "files": [], + "files_total": 0, + "discover_error": None, + "layer": None, + "result_class": None, + "output": None, + "confirm": [], + "estimate_seconds": 0, + "warnings": [], + "preflight_error": None, + } + request = composition.request + preview, kinds = self._preview_selection(request) + total = preview["files_total"] + # Unknown kinds (nothing selected, or preflight refused) count every + # step the options ask for, and every file as rewritten. + layer = plan_layer(request, kinds or ()) + text_count = total if kinds is None else kinds.count("text") + return { + "ok": True, + "error": None, + "command": composition.command, + **preview, + "layer": layer, + "result_class": result_class_for(layer), + "output": describe_output(request, total), + "confirm": confirm_gates(request, total, text_count), + "estimate_seconds": estimate_rewrite_seconds(request, text_count), + "warnings": plan_warnings(request, kinds), + } + + def _preview_selection(self, request: CleanRequest) -> tuple[dict[str, Any], list[str] | None]: + """``plan``'s file fields, plus each file's resolved kind when preflight passes.""" + try: + selection = select_request_inputs(request) + except ValueError as error: + return { + "files": [], + "files_total": 0, + "discover_error": str(error), + "preflight_error": None, + }, None + preview: dict[str, Any] = { + "files": [ + self._file_entry(item.path, request) for item in selection.items[:PLAN_FILES_LIMIT] + ], + "files_total": len(selection.items), + "discover_error": None, + "preflight_error": None, + } + try: + return preview, _kinds(preflight_work(request, selection)) + except ValueError as error: + preview["preflight_error"] = str(error) + return preview, None + + @staticmethod + def _file_entry(path: Path, request: CleanRequest) -> dict[str, Any]: + try: + kind = classify_asset(path, forced_kind=request.force_type) + size = path.stat().st_size + except (OSError, ValueError): + kind, size = "unknown", None + return { + "path": str(path.resolve()), + "display": display_path(path), + "kind": kind, + "size": size, + } + + @staticmethod + def _inspect_one(path: Path, force_type: str, soft: bool) -> dict[str, Any]: + display = display_path(path) + try: + report = inspect_asset(path, force_type=force_type, soft_binding=soft) + except (Exception, SystemExit) as error: + return failed_finding(path, display, error) + return finding_from_report(report, path, display) + + def inspect(self, rid: Any, params: Mapping[str, Any]) -> dict[str, Any]: + state = _state(params) + soft = params.get("soft", False) + if not isinstance(soft, bool): + raise BadRequest("params.soft must be a boolean") + request = compose_request(state).request + findings = [] + cancelled = False + for item in _select(request).items: + if self.cancel.is_set(): + cancelled = True + break + finding = self._inspect_one(item.path, request.force_type, soft) + self.event(rid, "inspected", finding) + findings.append(finding) + return {"files": findings, "cancelled": cancelled} + + def clean(self, rid: Any, params: Mapping[str, Any]) -> dict[str, Any]: + state = _state(params) + confirmed = params.get("confirmed") or [] + if not isinstance(confirmed, list) or not all(isinstance(c, str) for c in confirmed): + raise BadRequest("params.confirmed must be a list of strings") + # A confirmed "remote" is this run's egress permission and nothing + # more: it is composed in here and never saved anywhere. + composition = compose_request(state, allow_remote="remote" in confirmed) + request = composition.request + selection = _select(request) + try: + work = preflight_work(request, selection) + except ValueError as error: + raise InvalidOptions(f"preflight failed, nothing was written: {error}") from error + kinds = _kinds(work) + _require_confirmed(confirm_gates(request, len(work), kinds.count("text")), confirmed) + + if request.dry_run: + self._dry_run(rid, request, work) + errors, cancelled = 0, False + else: + errors, cancelled = self._clean_all(rid, request, selection.batch, work) + total = len(work) + return { + "total": total, + "errors": errors, + "cancelled": cancelled, + "command": composition.command, + "history": self._record( + composition.command, total, errors, plan_layer(request, kinds), cancelled + ), + } + + def _clean_all( + self, + rid: Any, + request: CleanRequest, + batch: bool, + work: list[tuple[InputItem, Path | None, CleanPlan]], + ) -> tuple[int, bool]: + """Clean file by file until done or cancelled. Returns (errors, cancelled).""" + if batch and request.output and not request.in_place: + request.output.mkdir(parents=True, exist_ok=True) + errors = 0 + for index, (item, output, plan) in enumerate(work): + if self.cancel.is_set(): + return errors, True + done = self._clean_one(rid, index, len(work), item.path, output, request, plan) + if done["error"] is not None or done["exit_code"] != 0: + errors += 1 + return errors, False + + def _clean_one( + self, + rid: Any, + index: int, + total: int, + path: Path, + output: Path | None, + request: CleanRequest, + plan: CleanPlan, + ) -> dict[str, Any]: + display = display_path(path) + self.event(rid, "file_start", {"index": index, "total": total, "display": display}) + before = self._inspect_one(path, request.force_type, False) + before_text = text_for_diff(path) + + payload = self._run_item(rid, index, path, output, request, plan) + + error = payload.get("error") + written = Path(payload["output"]) if error is None and payload.get("output") else None + after_counts: dict[str, int] = {} + diff = None + if written is not None and written.is_file(): + after_counts = self._inspect_one(written, request.force_type, False)["counts"] + diff = unified_diff(before_text, text_for_diff(written), display) + done = _file_done( + index, + display, + layer_for_result(request, str(payload.get("kind") or before["kind"])), + output=str(written.resolve()) if written is not None else None, + exit_code=int(payload.get("exit_code") or 0), + before=before["counts"], + after=after_counts, + lines=result_lines(payload), + diff=diff, + error=str(error) if error is not None else None, + ) + self.event(rid, "file_done", done) + return done + + def _run_item( + self, + rid: Any, + index: int, + path: Path, + output: Path | None, + request: CleanRequest, + plan: CleanPlan, + ) -> dict[str, Any]: + """``run_clean_item`` with its Layer B stream sent as ``token`` frames. + + One file failing is a row, not the end of the batch, so an exception + becomes an error payload. + """ + coalescer = None + if plan.text.rewrite_plan is not None: + coalescer = TokenCoalescer( + lambda text: self.event(rid, "token", {"index": index, "text": text}) + ) + try: + return run_clean_item( + path, + output, + request, + plan, + on_token=coalescer.add if coalescer is not None else None, + ) + except (Exception, SystemExit) as error: + return {"exit_code": 1, "error": f"{type(error).__name__}: {error}"} + finally: + # Joined before file_done, so no token frame can follow it. + if coalescer is not None: + coalescer.close() + + def _dry_run( + self, rid: Any, request: CleanRequest, work: list[tuple[InputItem, Path | None, CleanPlan]] + ) -> None: + """Describe a visible-mark clean without writing, as ``wm --dry-run`` does.""" + total = len(work) + for index, (item, output, plan) in enumerate(work): + display = display_path(item.path) + self.event(rid, "file_start", {"index": index, "total": total, "display": display}) + payload = dry_run_payload(item.path, output, plan, request.in_place) + done = _file_done( + index, + display, + layer_for_result(request, payload["kind"]), + output=None, + exit_code=0, + before={}, + after={}, + lines=[ + f"Dry run: would write {payload['output']}.", + *result_lines(payload), + ], + diff=None, + error=None, + ) + self.event(rid, "file_done", done) + + def _record( + self, command: str, total: int, errors: int, layer: str, cancelled: bool + ) -> dict[str, str]: + entry = { + "time": time.strftime("%H:%M:%S"), + "command": command, + "summary": history_summary(total, errors, layer, cancelled=cancelled), + } + with self._history_lock: + self._history.insert(0, entry) + return entry + + def cancel_request(self, rid: Any, params: Mapping[str, Any]) -> dict[str, bool]: + self.cancel.set() + return {"ok": True} + + def detect_endpoints(self, rid: Any, params: Mapping[str, Any]) -> dict[str, Any]: + return {"endpoints": detect_local_endpoints()} + + def checks(self, rid: Any, params: Mapping[str, Any]) -> dict[str, Any]: + return {"checks": [vars(check) for check in environment_checks()]} + + def save_settings(self, rid: Any, params: Mapping[str, Any]) -> dict[str, str]: + target = settings_path() + updated = apply_settings_update(load_settings(target), params.get("settings")) + return {"path": str(save_settings(updated, target))} + + def history(self, rid: Any, params: Mapping[str, Any]) -> dict[str, Any]: + with self._history_lock: + return {"entries": [dict(entry) for entry in self._history]} + + def shutdown(self, rid: Any, params: Mapping[str, Any]) -> dict[str, bool]: + self.cancel.set() + self.stopping = True + return {"ok": True} + + +# --- process setup ----------------------------------------------------------- + + +def main(frames: BinaryIO | None = None) -> int: + """Serve on stdin until EOF or ``shutdown``. + + *frames* is the stream claimed before the pipeline import when this file + runs as a script; any other caller has fd 1 claimed here. + """ + stream = frames if frames is not None else claim_stdout() + # The launcher ran the frontend from its own directory; the operator's + # relative paths are relative to where wm-tui was started. + cwd = os.environ.get("WM_TUI_CWD") + if cwd: + try: + os.chdir(cwd) + except OSError as error: + print(f"wm-tui-bridge: cannot enter WM_TUI_CWD {cwd}: {error}", file=sys.stderr) + bridge = Bridge(FrameWriter(stream), launch_argv=os.environ.get("WM_TUI_ARGV")) + return bridge.serve(sys.stdin.buffer) + + +if __name__ == "__main__": + code = main(_FRAMES) + with contextlib.suppress(Exception): + sys.stderr.flush() + # Skip interpreter teardown: a daemon worker mid-clean or a probe pool + # waiting on a hung port must not keep a bridge the frontend already let + # go of alive. + os._exit(code) diff --git a/skills/remove-ai-marks/scripts/tui_core.py b/skills/remove-ai-marks/scripts/tui_core.py new file mode 100644 index 0000000..a37cde6 --- /dev/null +++ b/skills/remove-ai-marks/scripts/tui_core.py @@ -0,0 +1,1520 @@ +#!/usr/bin/env python3 +"""Pure decision logic behind ``wm-tui``, shared by the bridge and its tests. + +``wm-tui`` is a TypeScript frontend over a Python bridge (see +``tui/PROTOCOL.md``). The frontend only presents; every decision about a clean +is made here, in Python, next to the pipeline it describes. Keeping the logic +in a module with no I/O loop of its own is what lets it be unit-tested without +spawning anything. + +Design rules this module is held to: + +* A request is only ever built through ``clean_file._build_parser`` and + ``CleanRequest.from_args``. Presets are flag lists for that reason: a preset + is exactly what a person could type after ``wm``, so the command preview is + the command that runs. Plans come from ``clean_request.plan_work``; this + module never assembles one itself. +* It never speaks HTTP. Discovery goes through ``layer_b_discovery``, which + goes through ``layer_b_http``, so every probe inherits that module's + hardening. A test enforces that no HTTP library is imported here. +* It never labels a best-effort result as verified. The badge follows the + layer that did the work, never the outcome. +* It never holds an API key in anything it returns. The key is read from the + environment only when a run starts. +""" + +from __future__ import annotations + +import argparse +import difflib +import json +import os +import shlex +import sys +import unicodedata +from collections.abc import Callable, Collection, Mapping, Sequence +from concurrent.futures import ThreadPoolExecutor +from dataclasses import asdict, dataclass, fields, replace +from pathlib import Path +from typing import Any, NoReturn + +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +from batch_inputs import InputItem, InputSelection +from clean_asset import CleanPlan +from clean_file import _build_parser as _build_clean_parser +from clean_request import ( + CleanPlanPreflightError, + CleanRequest, + dropped_text_transforms, + plan_work, +) +from common import atomic_write_text, looks_binary +from layer_b_discovery import MODEL_ROUTES, EndpointPolicy, classify_endpoint, probe_backend +from optional_deps import KNOWN_EXTRAS, check_optional +from pipeline_actions import ActionCode, Effect +from rewrite_text import DEFAULT_BASE_URL, LIVE_REWRITE_BACKENDS, RewritePlan, resolve_base_url + +#: The environment variable the key is read from at run time, and only then. +API_KEY_ENV = "WATERMARKS_REWRITE_API_KEY" + +# --- result classes ---------------------------------------------------------- + +#: Result classes from CONTEXT.md. The badge follows the *layer*, never the +#: outcome: a Layer B rewrite that "worked" is still best-effort. +VERIFIABLE = "Verifiable" +BEST_EFFORT = "Best-effort" + +LAYER_RESULT_CLASS: dict[str, str] = { + "A": VERIFIABLE, + "M": VERIFIABLE, + "B": BEST_EFFORT, + "V": BEST_EFFORT, + "synthid": BEST_EFFORT, + # Character perturbation adds noise to defeat a detector rather than + # removing a carrier that can be counted afterwards. There is nothing to + # verify, so it cannot be badged with the layer that strips zero-width. + "perturb": BEST_EFFORT, +} + + +def result_class_for(layer: str) -> str: + """The honesty label for a layer. Unknown layers are never called verified.""" + return LAYER_RESULT_CLASS.get(layer, BEST_EFFORT) + + +def _changes_pixels(request: CleanRequest) -> bool: + """Whether the request edits image pixels rather than image metadata. + + ``CleanRequest.visible_requested`` covers mask, box, dilation and the + external inpainter; ``--degrade`` and ``--morpho`` are pixel-domain + operations too, and badging them as metadata work would call a + frequency-domain perturbation *Verifiable*. + """ + return request.visible_requested() or bool(request.degrade) or bool(request.morpho) + + +def layer_for_result(request: CleanRequest, kind: str) -> str: + """The layer that did the work on one asset. Follows the work, not the outcome.""" + if kind == "text": + if request.rewrite_strength: + return "B" + return "perturb" if request.char_perturb else "A" + if kind == "image": + if _changes_pixels(request): + return "V" + if request.remove_synthid: + return "synthid" + return "M" + + +#: ``layer_for_result``'s layers as a plan reports them (the wire carries A, +#: B or V there), weakest last. +_PLAN_LAYER = {"A": "A", "M": "A", "V": "V", "perturb": "V", "synthid": "V", "B": "B"} +_PLAN_LAYER_ORDER = ("A", "V", "B") + + +def plan_layer(request: CleanRequest, kinds: Collection[str] = ()) -> str: + """The weakest layer a whole plan turns on. + + A plan is only as verifiable as its least verifiable step, so this is the + honest badge to show *before* a run: adding a rewrite to a Layer A clean + makes the whole thing best-effort. Given the resolved kinds of the + selected files, only the work those files will get counts, so a rewrite + over Markdown files alone (which it skips) does not make the plan + best-effort. Without kinds, every step the request asks for counts. + """ + layers = {_PLAN_LAYER[layer_for_result(request, kind)] for kind in kinds or ("text", "image")} + return max(layers, key=_PLAN_LAYER_ORDER.index) + + +# --- the batch cost gate ----------------------------------------------------- + +#: ``RewritePlan``'s own defaults, so the estimate uses the values a run would. +_REWRITE_DEFAULTS: dict[str, Any] = {item.name: item.default for item in fields(RewritePlan)} + +#: Worst-case seconds above which a run is worth stopping to confirm. A single +#: file with one candidate sits under this; a batch, or any TSAPA search, does +#: not. A gate that fires on every rewrite is a gate nobody reads. +COST_CONFIRM_SECONDS = 300.0 + + +def _rewrite_timeout(request: CleanRequest) -> float: + return request.rewrite_timeout or _REWRITE_DEFAULTS["timeout"] + + +def estimate_rewrite_seconds(request: CleanRequest, file_count: int) -> float: + """Worst-case wall clock for a batch that runs a live rewrite. + + Sequential execution (matching ``clean_file.main``) is what makes this + honest: files x candidates x per-call timeout is a real ceiling, not an + optimistic one. + """ + if request.rewrite_strength is None or file_count <= 0: + return 0.0 + if request.rewrite_strength == "tsapa": + # TSAPA issues roughly population calls per generation, per file. + calls = max(1, request.tsapa_generations) * max(2, request.tsapa_population) + else: + calls = max(1, request.rewrite_candidates or _REWRITE_DEFAULTS["candidates"]) + return float(file_count) * calls * _rewrite_timeout(request) + + +def format_duration(seconds: float) -> str: + if seconds < 90: + return f"{seconds:.0f}s" + if seconds < 5400: + return f"{seconds / 60:.0f}m" + return f"{seconds / 3600:.1f}h" + + +# --- presets ----------------------------------------------------------------- + + +@dataclass(frozen=True) +class Preset: + """One named starting point for a clean, expressed as ``wm`` flags. + + A preset is a claim, not just a shortcut. Choosing "Deep clean" is + choosing a best-effort result, so the result class travels with the preset + and the operator reads it *before* running. + + The flags are CLI flags rather than request fields so a preset can only do + what a person could type, and so the command preview shows it verbatim. + A preset never carries a flag that has its own confirmation gate. + """ + + key: str + label: str + description: str + #: The weakest layer this preset turns on. + layer: str + flags: tuple[str, ...] + #: True when the preset cannot run without a reachable Layer B endpoint. + requires_endpoint: bool = False + + def to_dict(self) -> dict[str, Any]: + return { + "key": self.key, + "label": self.label, + "description": self.description, + "layer": self.layer, + "result_class": result_class_for(self.layer), + "flags": list(self.flags), + "requires_endpoint": self.requires_endpoint, + } + + +PRESETS: tuple[Preset, ...] = ( + Preset( + key="hidden", + label="Hidden marks", + description=( + "Zero-width carriers, bidi controls and AI metadata. " + "Counted before and after; nothing is rephrased. " + "Identical to a bare `wm FILE`." + ), + layer="A", + flags=(), + ), + Preset( + key="hidden-aggressive", + label="Hidden marks, aggressive", + description=( + "Adds NFKC normalisation and homoglyph folding: Cyrillic and Greek " + "look-alikes become ASCII. Can change genuinely mixed-script text." + ), + layer="A", + flags=("--nfkc", "--aggressive-homoglyphs"), + ), + Preset( + key="rewrite", + label="Deep clean (LLM rewrite)", + description=( + "Hidden marks, then a local model rephrases the text to break " + "token-level watermarks. No detector guarantee. Needs an endpoint." + ), + layer="B", + flags=("--rewrite", "paraphrase"), + requires_endpoint=True, + ), + Preset( + key="image", + label="Images: metadata + degrade", + description=( + "Strips C2PA and AI metadata, then perturbs the frequency domain " + "where invisible image marks live. Best-effort; the pixels change." + ), + layer="V", + flags=("--degrade", "freq-dct"), + ), +) + + +def preset_for(key: str | None) -> Preset | None: + """The preset with this key, or None. An unknown key is never guessed at.""" + return next((preset for preset in PRESETS if preset.key == key), None) + + +# --- composing a request ----------------------------------------------------- + + +class BadRequest(ValueError): + """The message itself is malformed: wrong types, unknown keys.""" + + +class InvalidOptions(ValueError): + """argparse or ``CleanRequest`` refused the options. The message is theirs.""" + + +#: Wire endpoint key -> ``CleanRequest`` field and ``tui.json`` key (they are +#: the same names). An allow-list, so nothing else a client sends (an +#: ``api_key``, say) can be poured into a request or a settings file. +ENDPOINT_FIELDS: dict[str, str] = { + "backend": "rewrite_backend", + "base_url": "rewrite_base_url", + "model": "rewrite_model", +} + + +@dataclass(frozen=True) +class Composition: + """A composed request plus the ``wm`` command that is equivalent to it. + + ``request`` has ``json`` and ``quiet`` forced on so ``run_clean_item`` stays + silent; ``command`` is rendered *before* that, so the preview is what an + operator would type, not what the bridge needs. Neither carries a key. + """ + + request: CleanRequest + command: str + + +def _string_list(value: object, name: str) -> list[str]: + if value is None: + return [] + if not isinstance(value, list) or not all(isinstance(item, str) for item in value): + raise BadRequest(f"state.{name} must be a list of strings") + return list(value) + + +def _read_state(state: object) -> tuple[list[str], list[str], str | None]: + """The state's paths, its flags with the preset's in front, and the preset key. + + Only shapes are checked here. An unknown preset contributes no flags and + is refused by ``compose_request``, so a refused plan still previews + everything else that was typed. + """ + if not isinstance(state, Mapping): + raise BadRequest("state must be an object") + paths = _string_list(state.get("paths"), "paths") + flags = _string_list(state.get("flags"), "flags") + key = state.get("preset") + if key is not None and not isinstance(key, str): + raise BadRequest("state.preset must be a string or null") + preset = preset_for(key) + return paths, [*(preset.flags if preset else ()), *flags], key + + +def _raising_parser() -> argparse.ArgumentParser: + """``clean_file``'s own parser, made to raise instead of printing and exiting. + + argparse reports a bad flag by printing usage and calling ``sys.exit``, and + ``--help`` (or any abbreviation of it) prints the whole help first. In a + long-lived bridge the exit would kill the process and the text would land + in the log; the message is what the operator needs, so it is carried out + as ``InvalidOptions`` instead. + """ + parser = _build_clean_parser() + parser.prog = "wm" + + def refuse(message: str) -> NoReturn: + raise InvalidOptions(message) + + def leave(status: int = 0, message: str | None = None) -> NoReturn: + refuse(message or "--help is not available here") + + parser.error = refuse # type: ignore[method-assign] + parser.exit = leave # type: ignore[method-assign] + parser.print_help = lambda file=None: None # type: ignore[method-assign] + return parser + + +def preview_command(state: Mapping[str, object]) -> str: + """The ``wm ...`` line for a state that did not compose. + + A refused flag should still be visible in the preview, next to the error + that names it, rather than vanishing. + """ + paths, flags, _ = _read_state(state) + return shlex.join(["wm", *paths, *flags]) + + +def compose_request(state: Mapping[str, object], *, allow_remote: bool = False) -> Composition: + """Compose the ``CleanRequest`` for one frontend ``state`` (PROTOCOL "State"). + + 1. Parse ``[*preset.flags, *flags, "--", *paths]`` with ``clean_file``'s + parser, so a later flag wins over an earlier one and any CLI flag is + valid. Paths go after ``--`` so a file named ``-x.md`` is a path, not + an option; argparse reads the same request either way. + 2. Build the request with ``CleanRequest.from_args``. + 3. Apply ``endpoint`` only when the request rewrites, which keeps the + preview free of endpoint flags that would do nothing. *allow_remote* + is applied the same way: it is how ``clean`` passes a per-run + ``confirmed: ["remote"]`` in, and it is never persisted. + 4. Force ``json`` and ``quiet``. + + The API key is never part of a composition, which is shown and + remembered: the rewrite reads ``WATERMARKS_REWRITE_API_KEY`` itself, at + run time, exactly as it does under ``wm``. + + Raises ``BadRequest`` for a malformed state and ``InvalidOptions`` for + anything argparse or ``CleanRequest`` refuses. + """ + paths, flags, preset_key = _read_state(state) + if preset_key is not None and preset_for(preset_key) is None: + raise InvalidOptions(f"unknown preset: {preset_key}") + endpoint = state.get("endpoint") + if endpoint is not None and not isinstance(endpoint, Mapping): + raise BadRequest("state.endpoint must be an object or null") + # Validated even when it will not be applied, so a malformed endpoint is + # reported while the operator is still editing, not first on the run that + # turns a rewrite on. An unset field leaves WATERMARKS_REWRITE_* in charge. + overrides = { + target: value + for target, value in _endpoint_values(endpoint or {}, "state.endpoint").items() + if value is not None + } + + args = _raising_parser().parse_args([*flags, "--", *paths]) + try: + request = CleanRequest.from_args(args) + except (ValueError, TypeError) as error: + raise InvalidOptions(str(error)) from error + + # Egress permission comes from the per-run confirmation and nowhere else: + # not a typed --rewrite-allow-remote, and not WATERMARKS_REWRITE_ALLOW_REMOTE + # (an explicit False outranks the environment in live_from_environment). + if request.rewrite_strength is not None: + request = replace(request, **overrides, rewrite_allow_remote=allow_remote) + else: + request = replace(request, rewrite_allow_remote=None) + + command = request.command_string() + return Composition(request=replace(request, json=True, quiet=True), command=command) + + +def _endpoint_values(endpoint: Mapping[str, object], where: str) -> dict[str, str | None]: + """Validate an endpoint object and map it onto request/settings fields. + + ``null`` and ``""`` both mean "not stated". Remote egress is deliberately + not a field: it is a per-run confirmation. + """ + unknown = sorted(str(key) for key in endpoint if key not in ENDPOINT_FIELDS) + if unknown: + raise BadRequest(f"unknown {where} field(s): {', '.join(unknown)}") + values: dict[str, str | None] = {} + for key, target in ENDPOINT_FIELDS.items(): + value = endpoint.get(key) + if value is not None and not isinstance(value, str): + raise BadRequest(f"{where}.{key} must be a string or null") + values[target] = value or None + backend = values["rewrite_backend"] + if backend is not None and backend not in LIVE_REWRITE_BACKENDS: + raise InvalidOptions( + f"unknown rewrite backend: {backend} (choose from {', '.join(LIVE_REWRITE_BACKENDS)})" + ) + return values + + +# --- selection --------------------------------------------------------------- + + +def preflight_work( + request: CleanRequest, selection: InputSelection +) -> list[tuple[InputItem, Path | None, CleanPlan]]: + """Every destination and plan, validated before the first write. + + ``clean_request.plan_work`` with its per-file error folded into one + ``ValueError`` that names the offending file. + """ + try: + return plan_work(selection.items, request, selection.batch) + except CleanPlanPreflightError as error: + raise ValueError(f"{error.path}: {error}") from error + + +def display_path(path: Path) -> str: + """A path as the operator would type it: relative to cwd when under it.""" + try: + return str(path.resolve().relative_to(Path.cwd().resolve())) + except (ValueError, OSError): + return str(path) + + +def describe_output(request: CleanRequest, file_count: int) -> str: + """Where a run will write, in one line, before anything is written.""" + if request.dry_run: + return "Dry run: describes the plan and writes nothing." + if request.in_place: + return "Overwrites each file in place and keeps a .bak copy. Asks first." + if request.output is not None: + if file_count > 1: + return f"Writes into {request.output}/, keeping each file's relative path." + return f"Writes to {request.output}." + return "Writes NAME.cleaned.EXT next to each file. Originals are never touched." + + +# --- confirmation gates ------------------------------------------------------ + + +def _endpoint_policy(request: CleanRequest) -> EndpointPolicy | None: + """How the endpoint a rewrite would really contact classifies; None without one. + + Mirrors ``RewritePlan.live_from_environment``'s precedence (explicit value, + then ``WATERMARKS_REWRITE_BASE_URL``, then default) so the verdict is about + the endpoint the run will contact, not the one the form happens to show. + Classified with remote allowed, so an off-machine host reads as a question + to ask rather than a refusal. + """ + if request.rewrite_strength is None: + return None + return classify_endpoint(resolve_base_url(request.rewrite_base_url), allow_remote=True) + + +def confirm_gates(request: CleanRequest, file_count: int, text_count: int) -> list[dict[str, str]]: + """Every gate a clean stops at before the first write. + + One function feeds both ``plan`` (the preview) and ``clean`` (the gate), so + the two can never disagree about what needs a yes. *text_count* is how + many of the *file_count* files read as text: only those are rewritten, so + only those can leave the machine or cost model time. + + The remote gate fires for any off-machine endpoint, whatever + ``--rewrite-allow-remote`` or ``WATERMARKS_REWRITE_ALLOW_REMOTE`` say: in + the TUI, sending a document away is agreed to per run, never standing. + """ + gates: list[dict[str, str]] = [] + policy = _endpoint_policy(request) if text_count else None + if policy is not None and policy.allowed and not policy.loopback: + # The rewrite path's own warning: the operator agrees to that + # statement, not to a paraphrase of it. + gates.append( + { + "kind": "remote", + "message": f"{_sentence(str(policy.warning).removeprefix('warning: '))}. " + f"The full text of {_plural(text_count, 'file')} will be sent to {policy.host}.", + } + ) + if request.in_place: + gates.append( + { + "kind": "in_place", + "message": f"{_plural(file_count, 'file')} will be overwritten in place. " + "A .bak backup is kept for each.", + } + ) + if request.strip_semantic_format: + gates.append( + { + "kind": "semantic", + "message": "Contextual ZWJ, variation selectors and balanced bidi controls " + "are preserved by default because removing them can change how text " + "renders or what it means.", + } + ) + estimate = estimate_rewrite_seconds(request, text_count) + if estimate > COST_CONFIRM_SECONDS: + gates.append( + { + "kind": "cost", + "message": f"Rewriting {_plural(text_count, 'file')} one after another can " + f"take up to {format_duration(estimate)} (files x calls x " + f"{_rewrite_timeout(request):.0f}s timeout). " + "Cancelling stops after the current file; an in-flight model call " + "cannot be interrupted.", + } + ) + return gates + + +#: How a non-text kind is named to someone who picked the files. +_NOT_TEXT_LABELS = {"container": "Markdown and other documents", "image": "images"} + + +def _joined(items: Sequence[str]) -> str: + return items[0] if len(items) == 1 else f"{', '.join(items[:-1])} and {items[-1]}" + + +def plan_warnings(request: CleanRequest, kinds: Sequence[str] | None) -> list[str]: + """What the preview should say that no confirmation can change. + + * A malformed or non-http(s) base URL is not a gate (there is nothing to + agree to) but every rewritten file would fail on it. An off-machine + host is not listed here: that is the ``remote`` gate. + * Text-body transforms are no-ops on files that do not read as text + (``clean_request.dropped_text_transforms``); without a word here, a + Markdown file would look rewritten. + + *kinds* are the selected files' resolved kinds, or None when unknown. + """ + warnings: list[str] = [] + rewrites = kinds is None or "text" in kinds + policy = _endpoint_policy(request) if rewrites else None + if policy is not None and not policy.allowed: + warnings.append(f"Layer B endpoint refused: {policy.reason}.") + skipping = [kind for kind in kinds or () if dropped_text_transforms(request, kind)] + if skipping: + names = dropped_text_transforms(request, skipping[0]) + which = " and ".join(label for kind, label in _NOT_TEXT_LABELS.items() if kind in skipping) + warnings.append( + f"{_sentence(_joined(names))} will skip {_plural(len(skipping), 'file')} " + f"that {'is' if len(skipping) == 1 else 'are'} not plain text ({which}). " + f"Add --as text to force {'it' if len(names) == 1 else 'them'}." + ) + return warnings + + +# --- reading files for display ----------------------------------------------- + + +def read_text(path: Path, max_bytes: int) -> tuple[str, bool] | None: + """Up to *max_bytes* of *path* decoded as UTF-8, and whether that was all of it. + + None when the file cannot be read or looks binary: offsets or a diff into + a decoded ZIP or PNG would point at noise. + """ + try: + with path.open("rb") as source: + data = source.read(max_bytes + 1) + except OSError: + return None + head = data[:max_bytes] + if looks_binary(head) is not None: + return None + return head.decode("utf-8", errors="replace"), len(data) <= max_bytes + + +#: Leading words the pipeline writes in lower case that are names, not words. +_ACRONYMS = {"svg": "SVG", "json-ld": "JSON-LD", "xmp": "XMP", "exif": "EXIF", "opf": "OPF"} + + +def _sentence(text: str) -> str: + """*text* opening with a capital, for a line shown on its own. + + Only a leading plain word is capitalised: pipeline lines often open with a + path or a key (``docProps/core.xml: ...``, ``ai:Claude``), and changing + its case would name something that does not exist. + """ + first, space, rest = text.partition(" ") + word = first.removesuffix(":") + if word in _ACRONYMS: + return _ACRONYMS[word] + first[len(word) :] + space + rest + if word.isalpha() and word.islower(): + return text[:1].upper() + text[1:] + return text + + +def _capped(lines: list[str], limit: int, tail: str) -> list[str]: + """At most *limit* lines; the last says how many were cut (``tail`` gets the count).""" + if len(lines) <= limit: + return lines + return [*lines[: limit - 1], tail.format(len(lines) - (limit - 1))] + + +# --- inspection findings ----------------------------------------------------- + +#: Most lines a finding or a result carries. A file with thousands of distinct +#: carriers would otherwise flood the panel; the counts still hold the totals. +MAX_FINDING_LINES = 12 +_MORE_LINES = "{} more not shown." + +#: ``text_unicode`` hit kinds, in the operator's words. +HIT_NOUNS: dict[str, str] = { + "zwj_family": "zero-width carrier", + "bidi": "bidi control", + "tag_chars": "tag character", + "variation_selector": "variation selector", + "private_use": "private-use character", + "space": "space homoglyph", + "confusable": "confusable character", + "strip": "invisible character", + "other_cf": "format character", +} + + +def _plural(count: int, noun: str) -> str: + return f"{count} {noun}" if count == 1 else f"{count} {noun}s" + + +def _hits(report: Mapping[str, Any]) -> list[Mapping[str, Any]]: + """Layer A hits: ``hits`` for a text report, ``layer_a_hits`` for a container.""" + hits = report.get("hits") or report.get("layer_a_hits") or [] + return [hit for hit in hits if isinstance(hit, Mapping)] + + +def _metadata_findings(report: Mapping[str, Any]) -> list[str]: + """Metadata findings, without the Layer A lines a container also lists there. + + Reports keep informational context in ``notes``, so every finding here is + a mark (or a scan problem) and may be counted. ``inspect_container`` also + lists its Layer A hits among the findings, prefixed ``layer-a``, because the + audits read them from there; they are counted from ``layer_a_hits`` instead. + """ + findings = (str(item) for item in report.get("findings") or []) + return [finding for finding in findings if not finding.startswith("layer-a")] + + +def _soft_binding(report: Mapping[str, Any]) -> Mapping[str, Any] | None: + soft = (report.get("soft_binding") or {}).get("soft_binding") + return soft if isinstance(soft, Mapping) else None + + +def finding_counts(report: Mapping[str, Any]) -> dict[str, int]: + """Counts by class from an ``inspect_asset`` report. + + ``hidden`` is invisible-Unicode carriers (Layer A); ``metadata`` is + metadata findings (Layer M). + """ + hidden = sum(int(hit.get("count", 0)) for hit in _hits(report)) + if not hidden: + hidden = int(report.get("suspicious_total") or 0) + counts = {"hidden": hidden, "metadata": len(_metadata_findings(report))} + if report.get("has_c2pa"): + counts["c2pa"] = 1 + soft = _soft_binding(report) + if soft is not None and soft.get("found"): + counts["soft_binding"] = 1 + return counts + + +def _hit_line(hit: Mapping[str, Any]) -> str: + noun = HIT_NOUNS.get(str(hit.get("kind")), "hidden character") + return f"{_plural(int(hit.get('count', 0)), noun)} ({hit.get('codepoint', '?')})" + + +def finding_lines(report: Mapping[str, Any]) -> list[str]: + """Short human labels for one report, at most ``MAX_FINDING_LINES``.""" + lines = [_hit_line(hit) for hit in _hits(report)] + if report.get("has_c2pa"): + lines.append("C2PA manifest present") + lines.extend(_sentence(finding) for finding in _metadata_findings(report)) + soft = _soft_binding(report) + if soft is not None: + lines.append("Soft binding found" if soft.get("found") else "No soft binding") + if report.get("note"): + lines.append(_sentence(str(report["note"]))) + return _capped(lines, MAX_FINDING_LINES, _MORE_LINES) + + +def finding_from_report(report: Mapping[str, Any], path: Path, display: str) -> dict[str, Any]: + """One PROTOCOL ``Finding`` from an ``inspect_asset`` report.""" + return { + "path": str(path.resolve()), + "display": display, + "kind": str(report.get("kind", "unknown")), + "suspicious": bool(report.get("suspicious")), + "counts": finding_counts(report), + "lines": finding_lines(report), + "reveal": reveal_excerpts(path, report), + "error": None, + } + + +def failed_finding(path: Path, display: str, error: BaseException) -> dict[str, Any]: + """The ``Finding`` for a file ``inspect_asset`` could not read.""" + return { + "path": str(path.resolve()), + "display": display, + "kind": "unknown", + "suspicious": False, + "counts": {}, + "lines": [], + "reveal": [], + "error": f"{type(error).__name__}: {error}", + } + + +# --- reveal: hidden characters shown in place -------------------------------- + +#: Most excerpts a finding carries, and the widest one. +MAX_REVEAL_EXCERPTS = 6 +REVEAL_WIDTH = 100 +#: Largest prefix read to build excerpts. Past this the counts still hold. +REVEAL_MAX_BYTES = 2 * 1024 * 1024 +#: One column per hidden codepoint, so a mark's columns are exact. +REVEAL_GLYPH = "◆" +ELLIPSIS = "…" + +#: Short labels for the hidden codepoints an operator is likely to meet. +SHORT_NAMES: dict[int, str] = { + 0x00AD: "SHY", + 0x061C: "ALM", + 0x180E: "MVS", + 0x200B: "ZWSP", + 0x200C: "ZWNJ", + 0x200D: "ZWJ", + 0x200E: "LRM", + 0x200F: "RLM", + 0x202A: "LRE", + 0x202B: "RLE", + 0x202C: "PDF", + 0x202D: "LRO", + 0x202E: "RLO", + 0x2060: "WJ", + 0x2061: "FA", + 0x2062: "IT", + 0x2063: "IS", + 0x2064: "IP", + 0x2066: "LRI", + 0x2067: "RLI", + 0x2068: "FSI", + 0x2069: "PDI", + 0xFE0E: "VS15", + 0xFE0F: "VS16", + 0xFEFF: "BOM", +} + + +def short_name(codepoint: int) -> str: + """A short label for a hidden codepoint, or its ``U+XXXX`` hex.""" + if codepoint in SHORT_NAMES: + return SHORT_NAMES[codepoint] + if 0xE0000 <= codepoint <= 0xE007F: + return "TAG" + if 0xFE00 <= codepoint <= 0xFE0D: + return f"VS{codepoint - 0xFE00 + 1}" + if 0xE0100 <= codepoint <= 0xE01EF: + return f"VS{codepoint - 0xE0100 + 17}" + return f"U+{codepoint:04X}" + + +def _is_hidden(char: str, flagged: frozenset[int] = frozenset()) -> bool: + """Whether a character renders as nothing (or as a blank) where it sits. + + Format, separator and private-use characters are invisible by category, + and so are variation selectors, whose category says "mark". ``flagged`` + adds whatever the inspector itself reported, such as Hangul fillers or + space look-alikes, whose categories say "letter" or "space". + """ + codepoint = ord(char) + if codepoint in flagged: + return True + if unicodedata.category(char) in ("Cf", "Zl", "Zp", "Co"): + return True + return 0xFE00 <= codepoint <= 0xFE0F or 0xE0100 <= codepoint <= 0xE01EF + + +def _hit_offsets(text: str, report: Mapping[str, Any]) -> tuple[list[int], frozenset[int]]: + """Offsets of reported hits, each checked against the text it points into. + + The inspector decodes with ``surrogateescape`` and this reads with + ``replace``; on a file with bad bytes the two can disagree about + positions, so an offset is used only when the character there is the one + reported, and a codepoint whose offsets all miss is found by search. + """ + offsets: set[int] = set() + flagged: set[int] = set() + for hit in _hits(report): + if hit.get("kind") == "confusable": + continue + try: + codepoint = int(str(hit.get("codepoint", "")).removeprefix("U+"), 16) + except ValueError: + continue + flagged.add(codepoint) + char = chr(codepoint) + valid = [ + offset + for offset in hit.get("sample_offsets") or [] + if isinstance(offset, int) and 0 <= offset < len(text) and text[offset] == char + ] + if not valid: + start = text.find(char) + while start != -1 and len(valid) < 10: + valid.append(start) + start = text.find(char, start + 1) + offsets.update(valid) + return sorted(offsets), frozenset(flagged) + + +def _excerpt(line: str, first_mark: int, flagged: frozenset[int]) -> dict[str, Any]: + """One line with hidden characters as glyphs, trimmed around its first mark.""" + columns: list[str] = [] + marks: list[list[Any]] = [] + for column, char in enumerate(line): + if _is_hidden(char, flagged): + columns.append(REVEAL_GLYPH) + marks.append([column, column + 1, short_name(ord(char))]) + elif char == "\t" or unicodedata.category(char) == "Cc": + columns.append(" ") + else: + columns.append(char) + start, end = 0, len(columns) + if len(columns) > REVEAL_WIDTH: + window = REVEAL_WIDTH - 2 # room for an ellipsis at each cut edge + start = min(max(0, first_mark - window // 2), len(columns) - window) + end = start + window + left = ELLIPSIS if start > 0 else "" + right = ELLIPSIS if end < len(columns) else "" + shift = len(left) - start + return { + "text": left + "".join(columns[start:end]) + right, + "marks": [ + [begin + shift, finish + shift, name] + for begin, finish, name in marks + if start <= begin and finish <= end + ], + } + + +def reveal_excerpts(path: Path, report: Mapping[str, Any]) -> list[dict[str, Any]]: + """Up to six excerpts showing where the hidden characters sit. + + A count says "3 zero-width characters"; an excerpt says *which words* they + are glued to, which is what an operator needs to judge whether a mark is a + watermark or a legitimate joiner. Only assets that read as text get + excerpts. + """ + read = read_text(path, REVEAL_MAX_BYTES) + if read is None: + return [] + text = read[0] + offsets, flagged = _hit_offsets(text, report) + excerpts: list[dict[str, Any]] = [] + seen_lines: set[int] = set() + for offset in offsets: + line_start = text.rfind("\n", 0, offset) + 1 + if line_start in seen_lines: + continue + seen_lines.add(line_start) + line_end = text.find("\n", offset) + line = text[line_start : len(text) if line_end == -1 else line_end].removesuffix("\r") + excerpt = _excerpt(line, offset - line_start, flagged) + excerpts.append({"line": text.count("\n", 0, line_start) + 1, **excerpt}) + if len(excerpts) == MAX_REVEAL_EXCERPTS: + break + return excerpts + + +# --- what a clean did -------------------------------------------------------- + +#: Diff budget per ``file_done``. +DIFF_MAX_LINES = 200 +#: Files above this are not diffed: the diff would be truncated to nothing +#: useful, and reading them twice costs more than it tells. +DIFF_MAX_BYTES = 1 << 20 + + +def text_for_diff(path: Path) -> str | None: + """The whole file as text, or None when it is binary, unreadable or too big.""" + read = read_text(path, DIFF_MAX_BYTES) + return read[0] if read is not None and read[1] else None + + +def _visible(line: str) -> str: + """Show invisible characters in a diff line. + + The whole point of a Layer A diff is a character nobody can see; a diff + that renders it as nothing shows two identical lines. + """ + return "".join( + f"" + if _is_hidden(char) or (char != "\t" and unicodedata.category(char) == "Cc") + else char + for char in line + ) + + +def unified_diff(before: str | None, after: str | None, display: str) -> str | None: + """A text-only diff, invisible characters made visible, at most 200 lines.""" + if before is None or after is None or before == after: + return None + lines = [ + _visible(line) + for line in difflib.unified_diff( + before.splitlines(), + after.splitlines(), + fromfile=f"{display} (before)", + tofile=f"{display} (after)", + lineterm="", + n=2, + ) + ] + lines = _capped(lines, DIFF_MAX_LINES, "... diff truncated ({} more lines)") + return "\n".join(lines) if lines else None + + +def _count(count: int, singular: str, plural: str | None = None) -> str: + return f"{count} {singular if count == 1 else plural or singular + 's'}" + + +def _layer_a_phrases(removed: int, replaced: int) -> list[str]: + phrases = [] + if removed: + phrases.append(f"removed {_plural(removed, 'hidden character')}") + if replaced: + phrases.append(f"replaced {_plural(replaced, 'look-alike character')}") + return phrases + + +def _because(params: Mapping[str, Any]) -> str: + return f" ({params['reason']})" if params.get("reason") else "" + + +def _failure(params: Mapping[str, Any]) -> str: + """`` (exit code 1): detail`` for a tool failure, from whichever parts it has.""" + code = f" (exit code {params['returncode']})" if params.get("returncode") is not None else "" + detail = f": {params['detail']}" if params.get("detail") else "" + return code + detail + + +def _nested(params: Mapping[str, Any]) -> str: + """``: dropped X; dropped Y`` for the changing steps of a nested clean.""" + inner = [ + phrase + for step in params.get("actions") or () + if step.get("effect") == Effect.CHANGE.value + for phrase in action_phrases(step) + ] + return f": {'; '.join(inner)}" if inner else "" + + +def _jpeg_segment(p: Mapping[str, Any]) -> str: + if p["segment"] == "COM": + return "dropped the JPEG comment" + return f"dropped the JPEG {p['segment']} segment{_because(p)}" + + +def _tiff_tag(p: Mapping[str, Any]) -> str: + if p.get("name"): + return f"dropped TIFF tag {p['tag']} ({p['name']})" + return f"dropped TIFF tag {p['tag']}, which carried AI markers" + + +def _pdf_metadata(p: Mapping[str, Any]) -> str: + if p.get("pages"): + return f"dropped page metadata from {_count(p['pages'], 'page')}" + return f"dropped the PDF {p['target']}" + + +def _pdf_rewritten(p: Mapping[str, Any]) -> str: + if p["tool"] == "qpdf": + return "rebuilt the PDF structure with qpdf, so old metadata bytes are gone" + return "rebuilt the PDF with pypdf, without its info dictionary or XMP" + + +def _gif_extension(p: Mapping[str, Any]) -> str: + extension = p["extension"] + if extension == "extension": + return "dropped a GIF extension block" + return f"dropped the GIF {extension} extension" + + +#: One entry per ``pipeline_actions.ActionCode``: the phrases a step reads as, +#: each opening with a lower-case verb written here, so capitalising the first +#: letter never re-cases a path, key or acronym. An empty result means the +#: step says nothing worth a line (nothing removed, bookkeeping). +_ACTION_PHRASES: dict[str, Callable[[Mapping[str, Any]], Sequence[str]]] = { + ActionCode.NOTHING_REMOVED: lambda p: (), + ActionCode.DROP_PNG_CHUNK: lambda p: (f"dropped the PNG {p['chunk']} chunk{_because(p)}",), + ActionCode.DROP_JPEG_SEGMENT: lambda p: (_jpeg_segment(p),), + ActionCode.PRESERVE_JPEG_SCAN: lambda p: (), + ActionCode.DROP_WEBP_CHUNK: lambda p: (f"dropped the WebP {p['chunk'].strip()} chunk",), + ActionCode.DROP_GIF_EXTENSION: lambda p: (_gif_extension(p),), + ActionCode.DROP_TIFF_TAG: lambda p: (_tiff_tag(p),), + ActionCode.DROP_BMP_TRAILER: lambda p: ( + f"dropped {_count(p['bytes'], 'trailing byte')} after the BMP pixels" + + (f" ({', '.join(p['markers'])})" if p.get("markers") else ""), + ), + ActionCode.KEEP_BMP_TRAILER: lambda p: (), + ActionCode.BMP_UNPARSED: lambda p: ( + "left the BMP unchanged: its header could not be fully parsed", + ), + ActionCode.NEUTRALIZE_BOX: lambda p: ( + f"neutralized the '{p['box']}' C2PA box ({_count(p['bytes'], 'byte')} zeroed)", + ), + ActionCode.ZERO_PAYLOAD: lambda p: ( + f"zeroed the {p['target']} payload ({_count(p['bytes'], 'byte')})", + ), + ActionCode.NEUTRALIZE_TOKENS: lambda p: ( + f"neutralized AI tokens in the {p['target']} ({', '.join(p['tokens'])})", + ), + ActionCode.EXIFTOOL_STRIP: lambda p: ("stripped remaining metadata with exiftool",), + ActionCode.SYNTHID_BAND_REMOVAL: lambda p: ( + f"suppressed the SynthID frequency band at strength {p['strength']}", + ), + ActionCode.WMCT_MARKER_WRITTEN: lambda p: ( + "wrote a wmCt marker recording that wm cleaned the file", + ), + ActionCode.WMCT_MARKER_SKIPPED: lambda p: (f"skipped the wmCt marker: {p['reason']}",), + ActionCode.DROP_FRONTMATTER_KEY: lambda p: ( + f"dropped frontmatter key {p['key']}" + + (" (its value names an AI tool)" if p.get("value_hit") else ""), + ), + ActionCode.DROP_EMPTY_FRONTMATTER: lambda p: ("removed the frontmatter block, now empty",), + ActionCode.CLEAN_DATA_URI: lambda p: (f"cleaned an embedded {p['mime']} image{_nested(p)}",), + ActionCode.DROP_HTML_META: lambda p: (f"dropped meta tag {p['tag']}",), + ActionCode.DROP_JSON_LD: lambda p: ("dropped a JSON-LD provenance script",), + ActionCode.DROP_DATA_AI_ATTRIBUTES: lambda p: ( + f"dropped {_count(p['count'], 'data-ai attribute')}", + ), + ActionCode.DROP_SVG_METADATA: lambda p: ( + f"dropped {_count(p['count'], 'SVG metadata block')}", + ), + ActionCode.DROP_SVG_XMP: lambda p: (f"dropped {_count(p['count'], 'XMP packet')}",), + ActionCode.DROP_SVG_COMMENT: lambda p: ("dropped an SVG comment with AI markers",), + ActionCode.DROP_SVG_GENERATOR_ATTRIBUTES: lambda p: ( + f"dropped {_count(p['count'], 'generator attribute')}", + ), + ActionCode.CLEAN_EMBEDDED_MEDIA: lambda p: (f"cleaned embedded image {p['part']}{_nested(p)}",), + ActionCode.CLEAN_PART: lambda p: tuple( + f"{phrase} in {p['part']}" + for step in p.get("actions") or () + for phrase in action_phrases(step) + ), + ActionCode.DROP_PART: lambda p: (f"dropped part {p['part']}{_because(p)}",), + ActionCode.SCRUB_FIELD: lambda p: ( + f"cleared {p['field']} in {p['part']}" + if p.get("part") + else f"cleared {p['field']}{_because(p)}", + ), + ActionCode.DROP_CONTENT_TYPE_OVERRIDES: lambda p: ( + f"dropped {_count(p['count'], p['target'] + ' content-type override')}", + ), + ActionCode.PRUNE_RELATIONSHIPS: lambda p: ( + f"pruned {_count(p['count'], 'dangling relationship')} in {p['part']}", + ), + ActionCode.PRUNE_MANIFEST: lambda p: ( + "pruned " + + _count( + p["count"], + f"{p['manifest']} manifest entry", + f"{p['manifest']} manifest entries", + ), + ), + ActionCode.DROP_GENERATOR_META: lambda p: (f"dropped the {p['field']} field",), + ActionCode.SCRUB_CREATOR: lambda p: (f"removed an AI {p['field']} field",), + ActionCode.DROP_OPF_META: lambda p: ("dropped an AI-related OPF meta tag",), + ActionCode.LAYER_A_TEXT: lambda p: _layer_a_phrases(p["removed"], p["replaced"]), + ActionCode.EXIFTOOL_RUN: lambda p: ( + "stripped metadata with exiftool" + if p["returncode"] == 0 + else f"ran exiftool, which exited with code {p['returncode']}", + ), + ActionCode.TOOL_FAILED: lambda p: ( + f"could not use {p['tool']}{_failure(p)}" + + (f"; trying {p['fallback']} instead" if p.get("fallback") else ""), + ), + ActionCode.TOOL_MISSING: lambda p: (f"skipped {p['tool']}: not installed",), + ActionCode.TRY_FALLBACK: lambda p: (f"fell back to {p['tool']}",), + ActionCode.PDF_ENCRYPTED: lambda p: ( + "copied the PDF unchanged: it is encrypted and needs a password", + ), + ActionCode.PDF_DECRYPTED: lambda p: ("opened the encrypted PDF with an empty password",), + ActionCode.DROP_PDF_METADATA: lambda p: (_pdf_metadata(p),), + ActionCode.PDF_REWRITTEN: lambda p: (_pdf_rewritten(p),), + ActionCode.PDF_REWRITE_FAILED: lambda p: ( + f"could not rebuild the PDF with qpdf{_failure(p)}; " + "old metadata bytes may still be recoverable", + ), + ActionCode.PDF_COPIED_UNCHANGED: lambda p: ( + "copied the PDF unchanged: no structural cleaner succeeded", + ), + ActionCode.C2PATOOL_HINT: lambda p: (), + ActionCode.VISIBLE_NEEDS_SOURCE: lambda p: ( + "found no visible mark to remove: give a mask, a box or a detector command", + ), + ActionCode.VISIBLE_PLAN: lambda p: (), + ActionCode.MASK_SOURCE: lambda p: (), + ActionCode.REFINE_MASK: lambda p: ( + f"filled holes and dilated the mask by {p['dilation_radius']} px " + f"({p['pixels_before']} to {p['pixels_after']} pixels)", + ), + ActionCode.EFFECTIVE_MASK: lambda p: ( + (f"wrote the mask to {p['published']}",) if p.get("published") else () + ), + ActionCode.INPAINT_SKIPPED: lambda p: ( + "ran no inpainting: the print-plan backend only builds the mask", + ), + ActionCode.INPAINT: lambda p: (f"inpainted the visible mark with the {p['backend']} backend",), + ActionCode.PLAN_LOCALIZE: lambda p: (f"would find the visible mark from {p['source']}",), + ActionCode.PLAN_REFINE_MASK: lambda p: ( + f"would fill holes and dilate the mask by {p['dilation_radius']} px", + ), + ActionCode.PLAN_INPAINT: lambda p: (f"would inpaint with the {p['backend']} backend",), + ActionCode.PLAN_STRIP_METADATA: lambda p: ("would strip the requested metadata",), + ActionCode.PLAN_DEGRADE: lambda p: (f"would apply {p['strategy']} degradation",), + ActionCode.PLAN_PUBLISH: lambda p: ( + f"would write the mask to {p['mask']} and the image to {p['image']}", + ), + ActionCode.FAILED: lambda p: (f"failed: {p['error']}",), +} + + +def action_phrases(detail: Mapping[str, Any]) -> Sequence[str]: + """The phrases one ``action_details`` entry reads as, verb first, no full stop.""" + return _ACTION_PHRASES[detail["code"]](detail.get("params") or {}) + + +def _as_sentence(phrase: str) -> str: + """A phrase from ``action_phrases`` as a line: its own verb capitalised, a full stop.""" + text = phrase[:1].upper() + phrase[1:] + return text if text.endswith((".", "!", "?")) else f"{text}." + + +def _rewrite_line(rewrite: Mapping[str, Any]) -> str | None: + """The Layer B sentence from ``stats["tsapa"]``. + + That key holds ``rewrite()``'s info, except after a TSAPA run, where it + holds only the search record (``generations``, ``population``, ...). + """ + if "generations" in rewrite: + how = f"a Layer B TSAPA search over {_plural(int(rewrite['generations']), 'generation')}" + elif rewrite.get("mode") == "rewritten": + model = rewrite.get("model") + how = f"a Layer B {rewrite.get('strength')} rewrite" + (f" by {model}" if model else "") + else: + return None + return f"Rewrote the text with {how}. Best-effort, no detector guarantee." + + +def result_lines(payload: Mapping[str, Any]) -> list[str]: + """What a clean did to one file, in short sentences, from its payload. + + Built from the payload's numbers and its ``action_details`` codes, never + from the human ``actions`` lines. Steps that changed nothing say nothing. + """ + if payload.get("error"): + return [f"Failed: {payload['error']}"] + phrases: list[str] = [] + stats = payload.get("stats") + if isinstance(stats, Mapping): + phrases.extend( + _layer_a_phrases( + int(stats.get("removed_count") or 0), int(stats.get("replaced_count") or 0) + ) + ) + lines = [_as_sentence(phrase) for phrase in phrases] + if isinstance(stats, Mapping): + if stats.get("nfkc_changed"): + lines.append("NFKC normalisation changed the text.") + rewrite = stats.get("tsapa") + if isinstance(rewrite, Mapping) and (line := _rewrite_line(rewrite)): + lines.append(line) + # A visible-mark clean runs first, then the metadata strip. + visible = payload.get("visible") + visible_steps = visible.get("action_details") or [] if isinstance(visible, Mapping) else [] + for detail in [*visible_steps, *(payload.get("action_details") or [])]: + lines.extend(_as_sentence(phrase) for phrase in action_phrases(detail)) + skipped = payload.get("skipped_text_transforms") + if skipped: + lines.append(f"Skipped {', '.join(str(item) for item in skipped)}: not a text asset.") + if payload.get("residual"): + lines.append("Residual C2PA or AI signals may remain.") + return _capped(lines, MAX_FINDING_LINES, _MORE_LINES) + + +def history_summary(total: int, errors: int, layer: str, *, cancelled: bool = False) -> str: + """One history line: counts, whether it was cancelled, and the result class.""" + counts = f"{_plural(total, 'file')}, {_plural(errors, 'error')}" + if cancelled: + counts += ", cancelled" + return f"{counts}. {result_class_for(layer)}." + + +# --- persisted setup --------------------------------------------------------- + +#: Environment override for the settings file, so a test never touches the +#: real one and an operator can keep per-project setups side by side. +SETTINGS_ENV = "WATERMARKS_TUI_SETTINGS" + + +@dataclass(frozen=True) +class TuiSettings: + """The setup wm-tui remembers between runs: a preset and an endpoint. + + Deliberately not routed through ``configuration``: that module is the + shared CLI/server config seam with its own precedence rules, and this is + one UI's memory of which endpoint you last pointed it at. The generated + command still carries every value explicitly, so a command copied out of + the TUI runs the same way on a machine that has no settings file. + + There is no API key field, and there never will be one. Nor is there a + remote opt-in: sending text off-machine is agreed to per run, so a saved + "yes" cannot silently apply to a later document. Keys a file carries + beyond these fields (``rewrite_reasoning_effort`` and + ``rewrite_allow_remote`` in files written by earlier releases) are ignored + on read and dropped on the next save. + """ + + preset: str | None = None + rewrite_backend: str | None = None + rewrite_base_url: str | None = None + rewrite_model: str | None = None + + def to_wire(self) -> dict[str, Any]: + """The PROTOCOL shape: ``{"preset", "endpoint": {backend, base_url, model}}``.""" + return { + "preset": self.preset, + "endpoint": {key: getattr(self, target) for key, target in ENDPOINT_FIELDS.items()}, + } + + +#: Every top-level key ``save_settings`` accepts. An allow-list, so nothing a +#: client sends can widen what is written to disk. +SETTINGS_KEYS: tuple[str, ...] = ("preset", "endpoint") + + +def settings_path(environ: Mapping[str, str] | None = None) -> Path: + """Where the setup file lives, honouring the usual per-platform roots.""" + env = os.environ if environ is None else environ + override = env.get(SETTINGS_ENV) + if override: + return Path(override) + base = env.get("XDG_CONFIG_HOME") or env.get("APPDATA") + root = Path(base) if base else Path.home() / ".config" + return root / "watermark-remover" / "tui.json" + + +def load_settings(path: Path | None = None) -> TuiSettings: + """Read the setup file. Anything unreadable means "no saved setup". + + Fail-soft on purpose: a corrupt or hand-edited settings file must not stop + the operator from starting the TUI, and every value in it is a convenience + with a visible control behind it. Every field is a string, and a value of + any other type is dropped here rather than failing late, inside + ``classify_endpoint``. + """ + target = path or settings_path() + try: + raw = json.loads(target.read_text(encoding="utf-8")) + except (OSError, ValueError): + return TuiSettings() + if not isinstance(raw, dict): + return TuiSettings() + known = {item.name for item in fields(TuiSettings)} + loaded = {key: value for key, value in raw.items() if key in known and isinstance(value, str)} + # hello echoes these into every state; a backend or preset this release + # does not know would otherwise make every plan fail validation. + if loaded.get("rewrite_backend") not in (None, *LIVE_REWRITE_BACKENDS): + del loaded["rewrite_backend"] + if loaded.get("preset") is not None and preset_for(loaded["preset"]) is None: + del loaded["preset"] + return TuiSettings(**loaded) + + +def apply_settings_update(current: TuiSettings, update: object) -> TuiSettings: + """Validate a ``save_settings`` payload strictly and apply it to *current*. + + Unlike ``load_settings`` this refuses rather than drops: a client that + sends an unknown key (anything named like a key, token or secret included) + is wrong, and saying so beats silently writing less than it asked for. A + top-level key that is absent keeps its saved value; ``endpoint: null`` + clears the endpoint, and an empty string clears one field. + """ + if not isinstance(update, Mapping): + raise BadRequest("settings must be an object") + unknown = sorted(str(key) for key in update if key not in SETTINGS_KEYS) + if unknown: + raise BadRequest(f"unknown settings key(s): {', '.join(unknown)}") + changes: dict[str, str | None] = {} + if "preset" in update: + preset = update["preset"] + if preset is not None and (not isinstance(preset, str) or preset_for(preset) is None): + raise BadRequest(f"unknown preset: {preset!r}") + changes["preset"] = preset + if "endpoint" in update: + endpoint = update["endpoint"] + if endpoint is not None and not isinstance(endpoint, Mapping): + raise BadRequest("settings.endpoint must be an object or null") + try: + changes.update(_endpoint_values(endpoint or {}, "settings.endpoint")) + except InvalidOptions as error: + raise BadRequest(str(error)) from error + return replace(current, **changes) + + +def save_settings(settings: TuiSettings, path: Path | None = None) -> Path: + """Write the setup file atomically and return where it went.""" + target = path or settings_path() + target.parent.mkdir(parents=True, exist_ok=True) + atomic_write_text(target, json.dumps(asdict(settings), indent=2, sort_keys=True) + "\n") + return target + + +# --- first-run setup --------------------------------------------------------- + + +def should_onboard(*, settings_exist: bool, force: bool = False, skip: bool = False) -> bool: + """Whether wm-tui opens on the setup screen. + + Only a first run does: finishing *or* skipping the setup writes the + settings file, and its existence is the "already set up" marker. A file + that exists but will not parse still counts as "set up": the fix for a + corrupt file is the setup screen on demand (``--setup``), not a wizard that + ambushes every start. + """ + if force: + return True + if skip: + return False + return not settings_exist + + +#: Loopback servers worth asking, in the order a person would guess them. +#: Every one is loopback on purpose: detection runs before the operator has +#: agreed to anything, so it must never be what reaches another machine. +#: Base URLs carry no ``/v1``; ``rewrite_text`` and discovery both append +#: their own routes. +LOCAL_ENDPOINT_CANDIDATES: tuple[tuple[str, str, str], ...] = ( + ("ollama", DEFAULT_BASE_URL, "Ollama"), + ("openai-compatible", "http://127.0.0.1:1234", "LM Studio"), + ("openai-compatible", "http://127.0.0.1:8080", "llama.cpp server"), + ("openai-compatible", "http://127.0.0.1:8000", "vLLM / other OpenAI-compatible"), +) + +#: Per-candidate timeout for detection. A closed port refuses instantly; this +#: bounds the port that accepts and then never answers. Probes run in +#: parallel, so the whole scan is one timeout, not one per candidate. +DETECT_TIMEOUT = 2.0 + + +def endpoint_candidates(environ: Mapping[str, str] | None = None) -> list[tuple[str, str, str]]: + """The candidates to probe, the environment's own endpoint first. + + An endpoint already configured through ``WATERMARKS_REWRITE_*`` is the + likeliest answer, so it leads, but it goes through the same loopback-only + probe as the rest, so a remote URL in the environment is reported as + blocked rather than contacted. + """ + env = os.environ if environ is None else environ + candidates: list[tuple[str, str, str]] = [] + backend = env.get("WATERMARKS_REWRITE_BACKEND") + base_url = env.get("WATERMARKS_REWRITE_BASE_URL") + if backend in MODEL_ROUTES and base_url: + candidates.append((backend, base_url.rstrip("/"), "from WATERMARKS_REWRITE_*")) + for candidate in LOCAL_ENDPOINT_CANDIDATES: + if all(candidate[:2] != known[:2] for known in candidates): + candidates.append(candidate) + return candidates + + +def detect_local_endpoints( + candidates: Sequence[tuple[str, str, str]] | None = None, +) -> list[dict[str, Any]]: + """Probe every candidate at once and report each, in candidate order. + + Never sends document text: ``probe_backend`` reads the model list only, + refuses non-loopback hosts before any request (``allow_remote=False``), + and turns every failure into an unreachable probe. Parallel because four + hung ports probed one after another is a frozen-looking screen for four + timeouts. + """ + targets = list(endpoint_candidates() if candidates is None else candidates) + if not targets: + return [] + + def ask(target: tuple[str, str, str]) -> dict[str, Any]: + backend, base_url, label = target + probe = probe_backend(backend, base_url, allow_remote=False, timeout=DETECT_TIMEOUT) + return { + "label": label, + "backend": probe.backend, + "base_url": probe.base_url, + "reachable": probe.reachable, + "models": list(probe.models), + "error": probe.error, + } + + with ThreadPoolExecutor(max_workers=len(targets)) as pool: + return list(pool.map(ask, targets)) + + +@dataclass(frozen=True) +class EnvironmentCheck: + """One line of the setup screen's "what do I have" table.""" + + name: str + state: str + detail: str + #: The shell command that fixes it, or "" when there is nothing to do. + fix: str = "" + #: False for things a first run needs; True for opt-in capabilities. + optional: bool = True + #: Whether this line is in the state an operator wants. Stated rather + #: than inferred from ``state`` so the frontend never parses prose. + good: bool = True + + +#: What each extra adds, in the operator's terms rather than the import's. +EXTRA_PURPOSE: dict[str, str] = { + "visible": "Images: visible-mark cleaning and pixel work (numpy, Pillow, OpenCV).", + "quality": "Image quality scoring (scikit-image).", + "ai": "Torch-backed adapters such as CtrlRegen and the text detectors (torch).", + "provenance": "C2PA manifest reading (c2pa-python).", +} + + +def environment_checks( + environ: Mapping[str, str] | None = None, + settings_file: Path | None = None, +) -> list[EnvironmentCheck]: + """What this install can do, and the command for each thing it cannot. + + The core clean needs nothing beyond the standard library, so every extra + is reported as optional. The table's job on a first run is to say "you + are ready" first and "here is what more you could add" second. The API + key line says whether a key is set, never what it is. + """ + env = os.environ if environ is None else environ + checks = [ + EnvironmentCheck( + "Python", + "OK", + f"Python {sys.version_info.major}.{sys.version_info.minor}.{sys.version_info.micro}. " + "Hidden-mark and metadata cleaning work with nothing else installed.", + optional=False, + ) + ] + for extra in KNOWN_EXTRAS: + availability = check_optional(extra) + checks.append( + EnvironmentCheck( + f"Extra: {extra}", + "Installed" if availability.available else "Not installed", + EXTRA_PURPOSE.get(extra, availability.hint), + "" if availability.available else f'pip install "watermark-remover[{extra}]"', + good=availability.available, + ) + ) + key_set = bool(env.get(API_KEY_ENV)) + checks.append( + EnvironmentCheck( + "Layer B API key", + "Set" if key_set else "Not set", + f"Read from {API_KEY_ENV} at run time, never shown or saved. " + "Local servers do not need one.", + "" if key_set else f"export {API_KEY_ENV}=... # only for a keyed server", + good=key_set, + ) + ) + target = settings_file or settings_path(env) + checks.append( + EnvironmentCheck( + "Settings file", + "Exists" if target.exists() else "Will be created", + str(target), + ) + ) + if env.get("TERM_PROGRAM") == "Apple_Terminal": + # OSC 52 is ignored by Terminal.app, and nothing tells the app so. + checks.append( + EnvironmentCheck( + "Clipboard", + "Limited", + "Terminal.app ignores OSC 52 copy. Select commands from the boxes instead.", + good=False, + ) + ) + return checks diff --git a/skills/remove-ai-marks/tui/.gitignore b/skills/remove-ai-marks/tui/.gitignore new file mode 100644 index 0000000..c2658d7 --- /dev/null +++ b/skills/remove-ai-marks/tui/.gitignore @@ -0,0 +1 @@ +node_modules/ diff --git a/skills/remove-ai-marks/tui/PROTOCOL.md b/skills/remove-ai-marks/tui/PROTOCOL.md new file mode 100644 index 0000000..d91a01c --- /dev/null +++ b/skills/remove-ai-marks/tui/PROTOCOL.md @@ -0,0 +1,287 @@ +# wm-tui bridge protocol + +`wm-tui` is two processes: + +- **The frontend** is TypeScript on Bun with OpenTUI and Solid, in this + directory. It owns the terminal. +- **The bridge** is Python, in `scripts/tui_bridge.py`. It owns every decision + about a clean. + +The frontend spawns the bridge and talks to it in JSON Lines over the bridge's +stdin and stdout. The bridge is the only place a `CleanRequest` is built. +Everything else in the frontend is presentation. + +``` +wm-tui (python, scripts/tui.py) ── validates argv, finds bun, execs ──▶ bun src/index.tsx + │ spawn + ▼ + $WM_TUI_PYTHON $WM_TUI_BRIDGE (scripts/tui_bridge.py) +``` + +## Environment the launcher sets + +| var | meaning | +| --- | --- | +| `WM_TUI_PYTHON` | absolute path of the interpreter `wm-tui` itself runs on; the frontend spawns the bridge with it, never `python` from PATH | +| `WM_TUI_BRIDGE` | absolute path of `tui_bridge.py` | +| `WM_TUI_ARGV` | JSON list: the raw `wm-tui` arguments, already validated by the launcher | +| `WM_TUI_LOG` | file the frontend appends the bridge's stderr to | +| `WM_TUI_CWD` | the user's working directory; the launcher runs Bun from the frontend directory, so the bridge `chdir`s back here at start and relative paths mean what the user typed | + +## Framing + +One JSON object per line, UTF-8, `\n` terminated, in both directions. + +- request, frontend → bridge: `{"id": 7, "method": "clean", "params": {...}}` +- response, bridge → frontend: `{"id": 7, "result": {...}}`, or + `{"id": 7, "error": {"code": "needs_confirm", "message": "...", "data": {...}}}` +- event, bridge → frontend, zero or more before the response to the same id: + `{"id": 7, "event": "file_done", "data": {...}}` +- unsolicited event, bridge → frontend, sent once at start: + `{"event": "ready", "data": {"version": "0.5.0"}}` + +Error codes are `bad_request`, `invalid_options`, `needs_confirm`, `busy`, +`cancelled` and `internal`. `invalid_options` covers anything argparse or +`CleanRequest` refuses; its `message` is argparse's own text. + +**stdout discipline.** At startup the bridge `os.dup`s fd 1 for protocol +frames only, then points fd 1 and `sys.stdout` at stderr. The pipeline prints, +and a stray `print` must never corrupt a frame. The frontend writes the +bridge's stderr to `WM_TUI_LOG` and never inherits it. + +Requests are handled on worker threads, so `detect_endpoints` or `cancel` +stays responsive during a clean. Only one `clean` or `inspect` runs at a time. +A second one returns `busy`. + +## State: the one input shape + +Every method that plans work takes the same `state` object. The bridge composes +the `CleanRequest` from it through `clean_file`'s own argparse parser, so any +CLI flag is valid here and the command preview is exact. + +```json +{ + "paths": ["drafts/", "notes.md"], + "preset": "hidden", + "flags": ["--recursive", "--glob", "*.md"], + "endpoint": {"backend": "ollama", "base_url": "http://127.0.0.1:11434", "model": "qwen3:14b"} +} +``` + +`endpoint` has the same three keys everywhere: in `state`, in `hello.settings`, +and in `save_settings`. Reasoning effort is set with +`--rewrite-reasoning-effort`, the same way as any other flag, so there is +only one way to set it. + +The bridge composes the request in four steps: + +1. It parses `[*preset.flags, *flags, "--", *paths]` with + `clean_file._build_parser()`, where a later flag wins. The `--` keeps a file + named `-x.md` a path. +2. It builds the request with `CleanRequest.from_args`. +3. It applies the `endpoint` fields, but only when the request has a + `rewrite_strength`. This keeps a preview command free of endpoint flags + that do nothing. `rewrite_allow_remote` is true only for a `clean` whose + `confirmed` list includes `"remote"`; a typed `--rewrite-allow-remote` and + `WATERMARKS_REWRITE_ALLOW_REMOTE` are both overridden. It is never + persisted. A confirmed remote clean adds `--rewrite-allow-remote` to its + `command` and history entry. +4. It forces `json=True` and `quiet=True`. + +The bridge never handles the API key. The pipeline reads +`WATERMARKS_REWRITE_API_KEY` at run time, the same way it does for the CLI. + +The key never appears in any frame, command string, history entry or settings +file. As a last line of defence the frame writer also redacts the key's value +(when it is at least 8 characters) from every `result`, `error` and `data` +payload. The composition function has unit tests. + +## Methods + +### `hello` `{}` + +Returns: + +```json +{"version": "0.5.0", + "presets": [{"key": "hidden", "label": "Hidden marks", "description": "...", + "layer": "A", "result_class": "Verifiable", "flags": [], + "requires_endpoint": false}], + "settings": {"preset": "hidden", "endpoint": {"backend": null, "base_url": null, "model": null}}, + "settings_path": "/Users/x/.config/watermark-remover/tui.json", + "onboard": true, + "initial": {"paths": ["tests/fixtures"], "flags": ["--recursive"]}, + "backends": ["ollama", "openai-compatible"]} +``` + +`onboard` is true exactly when no settings file exists. `--setup` forces it +and `--no-setup` suppresses it. + +### `plan` `{state}` + +This is the cheap preview. The frontend calls it, debounced, on every change. +It never writes. + +Returns: + +```json +{"ok": true, "error": null, + "command": "wm drafts/ notes.md --recursive --glob '*.md'", + "files": [{"path": "/abs/drafts/a.md", "display": "drafts/a.md", "kind": "text", "size": 1234}], + "files_total": 1, + "discover_error": null, + "layer": "A", "result_class": "Verifiable", + "output": "writes NAME.cleaned.EXT next to each file; originals are never touched", + "confirm": [{"kind": "remote", "message": "..."}], + "estimate_seconds": 0, + "warnings": [], + "preflight_error": null} +``` + +- **`ok` and `error`.** `ok: false` with `error` set means argparse or + `CleanRequest` refused the flags. The flags are still echoed in `command` + as far as possible. A refused plan selects no files, and its `layer`, + `result_class` and `output` are null. +- **`files` and `files_total`.** Selection comes from + `batch_inputs.select_inputs`, the same way `clean_file.main` does it. + `files` is capped at 2000 entries; `files_total` is the full count. A + file's `size` is null when it cannot be read. +- **`warnings`.** Non-fatal problems with the options, such as a base URL that + is malformed or not http(s). +- **`preflight_error`.** What `clean_request.plan_work` would refuse for the + whole selection, such as an output collision or a binary file forced to + text. +- **`layer` and `result_class`.** The weakest layer the selected files will + actually get: B if a text file is rewritten, else V if pixels change or text + is perturbed, else A. Text transforms skip files that are not plain text + (Markdown, Office, images), so a rewrite over Markdown alone stays A and + `warnings` says so, suggesting `--as text`. A file's own `file_done.layer` can + be finer (`M` for container metadata, `perturb`, `synthid`); its + `result_class` is the label to show. +- **`confirm`.** Every gate a clean would stop at before the first write: + - `remote`: Layer B endpoint not on loopback, and at least one text file + to send; + - `in_place`; + - `semantic`: `--strip-semantic-format`; + - `cost`: a Layer B batch estimated over 300 s, counting text files only. + + `estimate_seconds` counts text files only as well. + +### `inspect` `{state, soft?: bool}` + +Returns `{"files": [Finding], "cancelled": false}` for the selected files, and emits +`{"event": "inspected", "data": Finding}` as each one finishes. + +```json +{"path": "/abs/a.md", "display": "a.md", "kind": "text", "suspicious": true, + "counts": {"hidden": 4, "metadata": 1}, + "lines": ["3 zero-width carriers (U+200B)", "1 bidi control (U+202E)", "Frontmatter key: generator"], + "reveal": [{"line": 1, "text": "Hello◆ world", "marks": [[5, 6, "ZWSP"]]}], + "error": null} +``` + +`lines` are short, human sentences built bridge-side from `inspect_asset`, +at most 12 per file. `counts.metadata` and `lines` cover the report's +`findings` only. Its `notes` (HEIF brands, a CMS generator, marker-free Exif) +say nothing about AI provenance and are left out. + +`reveal` shows the hidden characters in place. It holds at most 6 excerpts, +one per affected line, and is empty for anything that is not read as text: + +```json +"reveal": [{"line": 1, "text": "Hello◆ world, this is◆ a test.", + "marks": [[5, 6, "ZWSP"], [21, 22, "RLO"]]}] +``` + +- In `text`, each invisible or format codepoint is replaced by `◆`, which takes + one column. An excerpt is at most 100 columns and is centred on its first + mark. `…` marks the edge where it was cut. +- Each entry in `marks` is `[start, end, short_name]`. The columns count into + `text`, and `end` is exclusive. +- Offsets come from `inspect_asset`'s `layer_a_hits[*].sample_offsets`. +- `suspicious` is true when the raw report says so, or when + `counts.hidden > 0`. + +### `clean` `{state, confirmed?: ["remote", ...]}` + +If `plan` reports a `confirm` kind that is not in `confirmed`, the bridge +returns error `needs_confirm` with `data.confirm` set, and nothing is written. +Otherwise it runs `plan_work`, then `run_clean_item` per file, then +re-inspects each output. Events: + +- `{"event": "file_start", "data": {"index": 0, "total": 3, "display": "a.md"}}` +- `{"event": "token", "data": {"index": 0, "text": "..."}}` + - This is the Layer B stream, coalesced to at most one frame per 50 ms. +- `{"event": "file_done", "data": {"index": 0, "display": "a.md", "output": "/abs/a.cleaned.md", "layer": "A", "result_class": "Verifiable", "exit_code": 0, "before": {"hidden": 4}, "after": {"hidden": 0}, "lines": ["Removed 4 hidden characters.", "Dropped frontmatter key generator."], "diff": "unified diff, text only, <= 200 lines", "error": null}}` + - `lines` has one sentence per step that changed something, visible + steps included. The bridge builds each one from the payload's + `action_details` code and params, never by parsing the `actions` text. + - With `--dry-run` (images with `--visible-mask`, `--visible-box` or + `--detect-command` only) nothing is written, `output` is + null, and `lines` starts with "Dry run: would write PATH.". + - A file whose `run_clean_item` raises becomes a `file_done` with `error` + set. The batch continues. + - `diff` starts with the two `---`/`+++` file headers. Each invisible + character is written as ``, so the diff is safe to print and the + frontend can mark it. + +A failed selection or a `plan_work` refusal returns `invalid_options`, and +nothing is written. + +The response is `{"total": 3, "errors": 0, "cancelled": false, "command": "wm ...", "history": HistoryEntry}`. +`errors` counts every file with a nonzero exit code, as the CLI does, so a file +that still carries C2PA or AI signals after the clean counts as an error. + +### `cancel` `{}` + +Stops after the current file. Returns `{"ok": true}`. + +### `detect_endpoints` `{}` + +Returns `{"endpoints": [{"label": "Ollama", "backend": "ollama", "base_url": "http://127.0.0.1:11434", "reachable": true, "models": ["qwen3:14b"], "error": null}]}`. + +The candidates are the `WATERMARKS_REWRITE_*` endpoint first, then Ollama +:11434, LM Studio :1234, llama.cpp :8080 and OpenAI-compatible :8000. All +probes run in parallel through `layer_b_discovery.probe_backend` with +`allow_remote=False`, so detection is loopback only. + +### `checks` `{}` + +Returns `{"checks": [{"name": "Python", "state": "OK", "good": true, "detail": "...", "fix": "", "optional": false}]}`. `state` is display text; `good` is what to test. + +### `save_settings` `{settings: {preset, endpoint: {backend, base_url, model}}}` + +Any other key is `bad_request`, and that includes anything named like a +key, token or secret. + +The write is atomic and returns `{"path": "..."}`. It merges: a missing +top-level key keeps its saved value, and `endpoint: null` clears it. When a +saved file is read, an unknown preset or backend is dropped. On disk, the file keeps +the key names the old `tui.json` used (`preset`, `rewrite_backend`, +`rewrite_base_url`, `rewrite_model`), so an existing file still loads. When +an old file is read, its `rewrite_reasoning_effort` and +`rewrite_allow_remote` keys are ignored and are not written back. + +### `history` `{}` + +Returns `{"entries": [{"time": "14:02:11", "command": "wm ...", "summary": "3 files, 0 errors. Verifiable."}]}`. + +This covers the current session only, newest first. + +### `shutdown` `{}` + +Returns `{"ok": true}`, then the bridge exits 0. EOF on stdin also exits. +Either way the bridge cancels any running work and waits up to 5 s for +in-flight requests to answer first. + +## Invariants (enforced by tests) + +- The bridge builds requests only through `clean_file._build_parser` and + `CleanRequest.from_args`. It never constructs a `CleanPlan`: plans come from + `plan_work`, and runs from `run_clean_item`. +- No `urllib` import in `tui_bridge.py`, `tui_core.py` or `tui.py`. HTTP goes + through `layer_b_discovery` and `rewrite_text` only. +- Detection is loopback only. +- No API key in any output frame, including under a hostile + `WATERMARKS_REWRITE_API_KEY`. +- Results are labelled by layer, never by outcome. diff --git a/skills/remove-ai-marks/tui/bun.lock b/skills/remove-ai-marks/tui/bun.lock new file mode 100644 index 0000000..6f2d1e9 --- /dev/null +++ b/skills/remove-ai-marks/tui/bun.lock @@ -0,0 +1,245 @@ +{ + "lockfileVersion": 2, + "configVersion": 1, + "workspaces": { + "": { + "name": "wm-tui", + "dependencies": { + "@opentui/core": "0.5.12", + "@opentui/solid": "0.5.12", + "solid-js": "1.9.12", + }, + "devDependencies": { + "@types/bun": "latest", + "typescript": "^5.9.0", + }, + }, + }, + "packages": { + "@ampproject/remapping": ["@ampproject/remapping@2.3.0", "", { "dependencies": { "@jridgewell/gen-mapping": "^0.3.5", "@jridgewell/trace-mapping": "^0.3.24" } }, "sha512-30iZtAPgz+LTIYoeivqYo853f02jBYSd5uGnGpkFV0M3xOt9aN73erkgYAmZU43x4VfqcnLxW9Kpg3R5LC4YYw=="], + + "@babel/code-frame": ["@babel/code-frame@7.29.7", "", { "dependencies": { "@babel/helper-validator-identifier": "^7.29.7", "js-tokens": "^4.0.0", "picocolors": "^1.1.1" } }, "sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw=="], + + "@babel/compat-data": ["@babel/compat-data@7.29.7", "", {}, "sha512-locTkQyKvwIEgBzVrn8693ebc97F2U8ZHjbXwDXJ5Fn2TCpNwTlKcaKLkdHop5c/icOFE7qt7Q9JC5hnKNa6Gg=="], + + "@babel/core": ["@babel/core@7.28.0", "", { "dependencies": { "@ampproject/remapping": "^2.2.0", "@babel/code-frame": "^7.27.1", "@babel/generator": "^7.28.0", "@babel/helper-compilation-targets": "^7.27.2", "@babel/helper-module-transforms": "^7.27.3", "@babel/helpers": "^7.27.6", "@babel/parser": "^7.28.0", "@babel/template": "^7.27.2", "@babel/traverse": "^7.28.0", "@babel/types": "^7.28.0", "convert-source-map": "^2.0.0", "debug": "^4.1.0", "gensync": "^1.0.0-beta.2", "json5": "^2.2.3", "semver": "^6.3.1" } }, "sha512-UlLAnTPrFdNGoFtbSXwcGFQBtQZJCNjaN6hQNP3UPvuNXT1i82N26KL3dZeIpNalWywr9IuQuncaAfUaS1g6sQ=="], + + "@babel/generator": ["@babel/generator@7.29.8", "", { "dependencies": { "@babel/parser": "^7.29.8", "@babel/types": "^7.29.8", "@jridgewell/gen-mapping": "^0.3.12", "@jridgewell/trace-mapping": "^0.3.28", "jsesc": "^3.0.2" } }, "sha512-gZbepsdh3WDtgZKWL+vTPh71LSBrm/Y4/QDZBVCcYfmeTEEuoOYwlSy+G1StfJg+/Zy550u/3TATbm7qDbbMtg=="], + + "@babel/helper-annotate-as-pure": ["@babel/helper-annotate-as-pure@7.29.7", "", { "dependencies": { "@babel/types": "^7.29.7" } }, "sha512-OoK6239jHPuSQOoS0kfTVKn0b/rVTk0seKq4Gd2UMLtmOVLjDC0ki3e+c90Trqv2gMfvJFqkiljrr568+qddiw=="], + + "@babel/helper-compilation-targets": ["@babel/helper-compilation-targets@7.29.7", "", { "dependencies": { "@babel/compat-data": "^7.29.7", "@babel/helper-validator-option": "^7.29.7", "browserslist": "^4.24.0", "lru-cache": "^5.1.1", "semver": "^6.3.1" } }, "sha512-wem6WaBj4NaVYVdNhLPPVacES6ZJ+KBBfSkTMD3YZxbP3rm3Di85tJU5ljaUNhaOynt+Aj0xruhYuzQBt8n71g=="], + + "@babel/helper-create-class-features-plugin": ["@babel/helper-create-class-features-plugin@7.29.7", "", { "dependencies": { "@babel/helper-annotate-as-pure": "^7.29.7", "@babel/helper-member-expression-to-functions": "^7.29.7", "@babel/helper-optimise-call-expression": "^7.29.7", "@babel/helper-replace-supers": "^7.29.7", "@babel/helper-skip-transparent-expression-wrappers": "^7.29.7", "@babel/traverse": "^7.29.7", "semver": "^6.3.1" }, "peerDependencies": { "@babel/core": "^7.0.0" } }, "sha512-IY3ZD9Tmooqr3TUhc3DUWxiuo8xx1DWLhd5M7hQ+ZWJamqM2BbalrBJb2MisSLoYorOj75U03qULCxQTY9r3hg=="], + + "@babel/helper-globals": ["@babel/helper-globals@7.29.7", "", {}, "sha512-3nQVUAtvkKH9zahfWgw96Jc/uFOmjACE1kQz82E2lqWmHBgjzbNlsC22nuQTfahmWeQtTq5nQ/4Nnd2A1wj4zA=="], + + "@babel/helper-member-expression-to-functions": ["@babel/helper-member-expression-to-functions@7.29.7", "", { "dependencies": { "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7" } }, "sha512-j+7JYmk1JYDtACIGj0QJqqWZjoUpMoEikQGADMaHgCMCSDqd2+P32rfcibUNrGOMWrlzK1WJBdxrB3JJQZwWtg=="], + + "@babel/helper-module-imports": ["@babel/helper-module-imports@7.29.7", "", { "dependencies": { "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7" } }, "sha512-ejHwrQQYcm9xnTivShn2IDOlIzInN34AXskvq9QicvCtEzq1Vzclu/tKF8Jq1Cg8JG2GL6/EmjgsCT7lXepE3g=="], + + "@babel/helper-module-transforms": ["@babel/helper-module-transforms@7.29.7", "", { "dependencies": { "@babel/helper-module-imports": "^7.29.7", "@babel/helper-validator-identifier": "^7.29.7", "@babel/traverse": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0" } }, "sha512-UPUVSyXbOh627KiCIGQSgwWzGeBKLkaJ9PJEdrngIwMSzxLR4jS4+f1f1jb7VzBbg8nFLaYotvVPFCTqdrmTAg=="], + + "@babel/helper-optimise-call-expression": ["@babel/helper-optimise-call-expression@7.29.7", "", { "dependencies": { "@babel/types": "^7.29.7" } }, "sha512-+kmGVjcT9RGYzoDwdwEqEvGgKe3BYq+O1iGzjFubaNgZHwYHP6lsF2Yghf4kEuv9BV7tYDZ913aBW9am6YKong=="], + + "@babel/helper-plugin-utils": ["@babel/helper-plugin-utils@7.29.7", "", {}, "sha512-G7sHYigPY17oO5SYWnfD/0MTBwVR781S/JI643e/JhUYgVgWE/61SoW3NH9KWUKyKq5LVh3npif99Wkt6j86Jw=="], + + "@babel/helper-replace-supers": ["@babel/helper-replace-supers@7.29.7", "", { "dependencies": { "@babel/helper-member-expression-to-functions": "^7.29.7", "@babel/helper-optimise-call-expression": "^7.29.7", "@babel/traverse": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0" } }, "sha512-atfGXWSeCiF4DnKZIfmJfQRkSw9b9gNNXR1kqKjbhG4pGYCOnkp8OcTB8E3NXjBu8NpheSnOeNKz8KT7UNFTmQ=="], + + "@babel/helper-skip-transparent-expression-wrappers": ["@babel/helper-skip-transparent-expression-wrappers@7.29.7", "", { "dependencies": { "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7" } }, "sha512-brcMGQaVzIeUb+6/bs1Av0f8YuNNjKY2JyvfRCsFuFsdKccEQ5Ges2y74D74NZ1Rz8lKJ9ksJkfqwQFJ/iNEyQ=="], + + "@babel/helper-string-parser": ["@babel/helper-string-parser@7.29.7", "", {}, "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw=="], + + "@babel/helper-validator-identifier": ["@babel/helper-validator-identifier@7.29.7", "", {}, "sha512-qehxGkRj55h/ff8EMaJ+cYhyaKlHIxqYDn682wQD7RNp9UujOQsHog2uS0r2vzr4pW+sXf90NeeayjcNaX3fFg=="], + + "@babel/helper-validator-option": ["@babel/helper-validator-option@7.29.7", "", {}, "sha512-N9ZErrD+yW5geCDtBqnOoxmR8+tNKiGuxKlDpuJxfsqpa2dFcexaziGAE/qoHLiDDreVNMupxGmSoNlyvsA3gw=="], + + "@babel/helpers": ["@babel/helpers@7.29.7", "", { "dependencies": { "@babel/template": "^7.29.7", "@babel/types": "^7.29.7" } }, "sha512-1k2lAGRMfHTcwuNYcCNUmaUffmQv8KWMfh2iJUUeRlwlwH4FdNG7mfPI10NPfLHJFThE4Tyr4mv7kTNZOiPuBg=="], + + "@babel/parser": ["@babel/parser@7.29.9", "", { "dependencies": { "@babel/types": "^7.29.8" }, "bin": "./bin/babel-parser.js" }, "sha512-CjXrNHTnvqBVqHgdBysY3vk2T8tpJHb5/RMeHJBTyVa9xgugCB0CJTx/3oO8RV2QRQP391RWpB7D6hLjm8V9uA=="], + + "@babel/plugin-syntax-jsx": ["@babel/plugin-syntax-jsx@7.29.7", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-TSu8+mHCoEaaCDEZ0I3+6mvTBYR4PCxQwf2z9/r5Tbztv6NaLR3B9thGTTxX2WGuGHJqRiAbKPeGTJ5XWXVg6A=="], + + "@babel/plugin-syntax-typescript": ["@babel/plugin-syntax-typescript@7.29.7", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-ngr+82Sh0xMz25TPCZi+nC2iTzjfCdWS2ONXTp/PtSCHCgaCNBpdMqgvJ2ccdLlClVZ7sisIgB914j/JFe+RZA=="], + + "@babel/plugin-transform-modules-commonjs": ["@babel/plugin-transform-modules-commonjs@7.29.7", "", { "dependencies": { "@babel/helper-module-transforms": "^7.29.7", "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-j0vCldybPC5b5dwCQOJ21uKtHzt7hxLygJTg9eF1ScfaikEDNfzn94XoW5Fi+seBR0nCyL23xaBFFkq7dTM8XQ=="], + + "@babel/plugin-transform-typescript": ["@babel/plugin-transform-typescript@7.29.9", "", { "dependencies": { "@babel/helper-annotate-as-pure": "^7.29.7", "@babel/helper-create-class-features-plugin": "^7.29.7", "@babel/helper-plugin-utils": "^7.29.7", "@babel/helper-skip-transparent-expression-wrappers": "^7.29.7", "@babel/plugin-syntax-typescript": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-FFwIwzU+7SCOuxxV4YtJql6T9981ZVTm+FHO5GhVsRqCTdy0WwrEZh3l42ARpXruJUDGOptdduHW6Zr7pNPLNg=="], + + "@babel/preset-typescript": ["@babel/preset-typescript@7.27.1", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.27.1", "@babel/helper-validator-option": "^7.27.1", "@babel/plugin-syntax-jsx": "^7.27.1", "@babel/plugin-transform-modules-commonjs": "^7.27.1", "@babel/plugin-transform-typescript": "^7.27.1" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-l7WfQfX0WK4M0v2RudjuQK4u99BS6yLHYEmdtVPP7lKV013zr9DygFuWNlnbvQ9LR+LS0Egz/XAvGx5U9MX0fQ=="], + + "@babel/template": ["@babel/template@7.29.7", "", { "dependencies": { "@babel/code-frame": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/types": "^7.29.7" } }, "sha512-puq+Gf35oI24FeN11LkoUQFqv9uwNeWpxXZi/Ji3rRIoKAzKnxRaZ+Gkj0vKS9ZCiTESfng1N9LyOyXvo+m+Gg=="], + + "@babel/traverse": ["@babel/traverse@7.29.8", "", { "dependencies": { "@babel/code-frame": "^7.29.7", "@babel/generator": "^7.29.8", "@babel/helper-globals": "^7.29.7", "@babel/parser": "^7.29.8", "@babel/template": "^7.29.7", "@babel/types": "^7.29.8", "debug": "^4.3.1" } }, "sha512-I5z7H3bf/41ktsNVLtpN0wAa336HkqIHQ5BuPLEhTkt1jVSyZpeNKIzTgEWmlxjdg81R0IgUCcaE+Ok3NvrfZg=="], + + "@babel/types": ["@babel/types@7.29.8", "", { "dependencies": { "@babel/helper-string-parser": "^7.29.7", "@babel/helper-validator-identifier": "^7.29.7" } }, "sha512-Vj1jF3cPfxg7OAfoI7QnVKLoILlm2JF9pnVHrX8qx7AHMiYWT+NDAA7jChlNgRS4WTLc/fD1lXLmPixluj+3Gg=="], + + "@jridgewell/gen-mapping": ["@jridgewell/gen-mapping@0.3.13", "", { "dependencies": { "@jridgewell/sourcemap-codec": "^1.5.0", "@jridgewell/trace-mapping": "^0.3.24" } }, "sha512-2kkt/7niJ6MgEPxF0bYdQ6etZaA+fQvDcLKckhy1yIQOzaoKjBBjSj63/aLVjYE3qhRt5dvM+uUyfCg6UKCBbA=="], + + "@jridgewell/resolve-uri": ["@jridgewell/resolve-uri@3.1.2", "", {}, "sha512-bRISgCIjP20/tbWSPWMEi54QVPRZExkuD9lJL+UIxUKtwVJA8wW1Trb1jMs1RFXo1CBTNZ/5hpC9QvmKWdopKw=="], + + "@jridgewell/sourcemap-codec": ["@jridgewell/sourcemap-codec@1.6.0", "", {}, "sha512-T7jf+5zgsZHwNJ4lvQ7/aezbyk0nNX+zJVWpmHA7VYsEx7a7qr5Rg5IbtJFqkgze5Y2sruq1RUY8Q837Od7iFw=="], + + "@jridgewell/trace-mapping": ["@jridgewell/trace-mapping@0.3.31", "", { "dependencies": { "@jridgewell/resolve-uri": "^3.1.0", "@jridgewell/sourcemap-codec": "^1.4.14" } }, "sha512-zzNR+SdQSDJzc8joaeP8QQoCQr8NuYx2dIIytl1QeBEZHJ9uW6hebsrYgbz8hJwUQao3TWCMtmfV8Nu1twOLAw=="], + + "@opentui/core": ["@opentui/core@0.5.12", "", { "dependencies": { "bun-ffi-structs": "0.3.1", "diff": "9.0.0", "marked": "17.0.1", "string-width": "7.2.0", "strip-ansi": "7.1.2" }, "optionalDependencies": { "@opentui/core-darwin-arm64": "0.5.12", "@opentui/core-darwin-x64": "0.5.12", "@opentui/core-linux-arm64": "0.5.12", "@opentui/core-linux-arm64-musl": "0.5.12", "@opentui/core-linux-x64": "0.5.12", "@opentui/core-linux-x64-musl": "0.5.12", "@opentui/core-win32-arm64": "0.5.12", "@opentui/core-win32-x64": "0.5.12" }, "peerDependencies": { "web-tree-sitter": "0.25.10" } }, "sha512-ZXBE5gmvdovmV8zJQrOQf6E44v1tJRDEgrM2MYhEglzgXZ+smIUp95O8zeRYGsuIzQIiMPMgQqKtTJuzvAb7BQ=="], + + "@opentui/core-darwin-arm64": ["@opentui/core-darwin-arm64@0.5.12", "", { "os": "darwin", "cpu": "arm64" }, "sha512-YdVnP0tAyerBNl0mIcmQEOotPeZzW1VnSXKBl5cyZ5e6nDd2Y+ui/8eRPpn1oqcamf1NCnzS4ohMgejOvna8Zg=="], + + "@opentui/core-darwin-x64": ["@opentui/core-darwin-x64@0.5.12", "", { "os": "darwin", "cpu": "x64" }, "sha512-uRrQJdHmLUSj3PV23QPi3WSimYTTxcXnVouxF6U4xMXlOv4N3SxnHfVwMRQkPqbGOfvVWHeLE6FdK4C+ubU0sQ=="], + + "@opentui/core-linux-arm64": ["@opentui/core-linux-arm64@0.5.12", "", { "os": "linux", "cpu": "arm64" }, "sha512-XeKhuIaEtgipvuPHbl4qPOBj+Ut+2zObmsxMVM1jDcjz/FatG9PGeGQPx1G1SnvH2AgpT4K+eCu7DUF0+yIqoQ=="], + + "@opentui/core-linux-arm64-musl": ["@opentui/core-linux-arm64-musl@0.5.12", "", { "os": "linux", "cpu": "arm64" }, "sha512-VZ2sNMw1d/r1SLPjUbOP9LKscKz1CQjID8adTL6gG8Lrrq+mYcIUxutyB+P/eG0J/7oRZLPR6OMt7dUOap6RTg=="], + + "@opentui/core-linux-x64": ["@opentui/core-linux-x64@0.5.12", "", { "os": "linux", "cpu": "x64" }, "sha512-eZiCjEzwbb6qClPPfk32Nha9xmr9obt69Xj0+9SKsXxWLBKkjQEGOMRoh/R9ObaQF4aq8If1xV3VEY0sD9W9vg=="], + + "@opentui/core-linux-x64-musl": ["@opentui/core-linux-x64-musl@0.5.12", "", { "os": "linux", "cpu": "x64" }, "sha512-WWW0hVBoSYZ3D6AgZ4u2Y5/u/IyIq2pDb+4yI3WgJ70Wyt6ofHy+6kRGRgbXFn1p+rPInAHjCXD2v6C7iEKSrA=="], + + "@opentui/core-win32-arm64": ["@opentui/core-win32-arm64@0.5.12", "", { "os": "win32", "cpu": "arm64" }, "sha512-aLbm6870Ybls6CYL4zMOCImTBPLZHZMUXJFGqMI44lIWxitkAtT6zg5lYA4oRqFRzzryDclxr29+hDgT3p3Blw=="], + + "@opentui/core-win32-x64": ["@opentui/core-win32-x64@0.5.12", "", { "os": "win32", "cpu": "x64" }, "sha512-KTwtwpfd2zF9opVh3SyRJYDd1o3Xv4XL8OZb8Zi+CqWUel6Y2IDCiVivCv8fGJt3J7wOIXXtuZI9ZUkLyKJCiQ=="], + + "@opentui/solid": ["@opentui/solid@0.5.12", "", { "dependencies": { "@babel/core": "7.28.0", "@babel/preset-typescript": "7.27.1", "@opentui/core": "0.5.12", "babel-plugin-module-resolver": "5.0.2", "babel-preset-solid": "1.9.12", "entities": "7.0.1", "s-js": "^0.4.9" }, "peerDependencies": { "solid-js": "1.9.12" } }, "sha512-hAiVlVMtT7AkHGblKwcW1YAuXtxkSy1XSf/RRc4j3IlG3mTNX0bhJdnGOo3Xw14EqeZMp41Mcp5WzHAzMm/DzA=="], + + "@types/bun": ["@types/bun@1.4.2", "", { "dependencies": { "bun-types": "1.4.2" } }, "sha512-GimotNn7+ZV0uVArItBbriZsR1oNf0+WTzPkdcFrzShI7k2norL0uzEaJT8T33dWr7O/c9ZDuAFQrctKCi72oQ=="], + + "@types/node": ["@types/node@26.6.2", "", { "dependencies": { "undici-types": "~8.9.0" } }, "sha512-X1P21scMv4zGKLYqjdGjaKa7COa0RKVYYZZN/NfvLQ1JegxFhdhpZG/Lyn8AXx6CDUavKAd11v6BvfpkDByK8g=="], + + "ansi-regex": ["ansi-regex@6.3.0", "", {}, "sha512-WpDfL7NO6j7tH88IDBNVdUJxDh9nmCteAVW9dsep846XdwF4naCBK+/tGLX3KJgcpgMRXCFlTM2hKGoK9FsdrQ=="], + + "babel-plugin-jsx-dom-expressions": ["babel-plugin-jsx-dom-expressions@0.40.10", "", { "dependencies": { "@babel/helper-module-imports": "7.18.6", "@babel/plugin-syntax-jsx": "^7.18.6", "@babel/types": "^7.20.7", "html-entities": "2.3.3", "parse5": "^7.1.2" }, "peerDependencies": { "@babel/core": "^7.20.12" } }, "sha512-lxve6Y02YiZTldB7efKpnbf1BH00XCFZNYYW235jSGsYaJNFtHrYlKV6/O+miHbjqpIr9FTe5+0no4hofAMbfA=="], + + "babel-plugin-module-resolver": ["babel-plugin-module-resolver@5.0.2", "", { "dependencies": { "find-babel-config": "^2.1.1", "glob": "^9.3.3", "pkg-up": "^3.1.0", "reselect": "^4.1.7", "resolve": "^1.22.8" } }, "sha512-9KtaCazHee2xc0ibfqsDeamwDps6FZNo5S0Q81dUqEuFzVwPhcT4J5jOqIVvgCA3Q/wO9hKYxN/Ds3tIsp5ygg=="], + + "babel-preset-solid": ["babel-preset-solid@1.9.12", "", { "dependencies": { "babel-plugin-jsx-dom-expressions": "^0.40.6" }, "peerDependencies": { "@babel/core": "^7.0.0", "solid-js": "^1.9.12" }, "optionalPeers": ["solid-js"] }, "sha512-LLqnuKVDlKpyBlMPcH6qEvs/wmS9a+NczppxJ3ryS/c0O5IiSFOIBQi9GzyiGDSbcJpx4Gr87jyFTos1MyEuWg=="], + + "balanced-match": ["balanced-match@1.0.2", "", {}, "sha512-3oSeUO0TMV67hN1AmbXsK4yaqU7tjiHlbxRDZOpH0KW9+CeX4bRAaX0Anxt0tx2MrpRpWwQaPwIlISEJhYU5Pw=="], + + "baseline-browser-mapping": ["baseline-browser-mapping@2.11.26", "", { "bin": { "baseline-browser-mapping": "dist/cli.cjs" } }, "sha512-GLQdD3y6UF8iVuMJl5fHgE4jdn/ua7n+toKfLgNlg3BqQtOZjpy68T8Tup8/wGWZCDlm7KMg7tPb4MPn7oN0TQ=="], + + "brace-expansion": ["brace-expansion@2.1.7", "", { "dependencies": { "balanced-match": "^1.0.0" } }, "sha512-uZbew1NqdmPDTMJ8ah1y+b+9QEJrfkXFk3RcTQw3X0jW/xRUvFKsg1CfQdSYGdTbXZWExtU3J3ccxtnfw1Fi0g=="], + + "browserslist": ["browserslist@4.29.1", "", { "dependencies": { "baseline-browser-mapping": "^2.11.25", "caniuse-lite": "^1.0.30001810", "electron-to-chromium": "^1.5.438", "node-releases": "^2.0.57", "update-browserslist-db": "^1.3.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-AUdjuRyCNGUYtqpqfTmWyM4fXay8yIQhmLnvYe/THMGfT9B/34X7xQd3ifKxwyNPPpowVBjLb+64BN9Rn1mizw=="], + + "bun-ffi-structs": ["bun-ffi-structs@0.3.1", "", { "peerDependencies": { "typescript": "^5" } }, "sha512-3gM7PpVWLyrwxWjcilSiGuhWanhZivvo6l0u573NziPH6f/gwk6McbaYgn7oJWov6pKGRTDbrg94W5DcJsKTtQ=="], + + "bun-types": ["bun-types@1.4.2", "", { "dependencies": { "@types/node": "*" } }, "sha512-bxV1FgK7yBIzjRe5zBozIM4Bem11ZJcCXSrjWRG3YWLt8yFDePu4cLjpebO8OvPeIE9trbyPF4fuj3Cia4Fj3w=="], + + "caniuse-lite": ["caniuse-lite@1.0.30001812", "", {}, "sha512-qN+QNNBr93TCmFrmte0bBCjSDMuRvt78VlHT99qIGPszm4QsqCX8lnyWUFkHi8B7ZBAPq8SH+UqyXNy/odMdng=="], + + "convert-source-map": ["convert-source-map@2.0.0", "", {}, "sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg=="], + + "csstype": ["csstype@3.2.3", "", {}, "sha512-z1HGKcYy2xA8AGQfwrn0PAy+PB7X/GSj3UVJW9qKyn43xWa+gl5nXmU4qqLMRzWVLFC8KusUX8T/0kCiOYpAIQ=="], + + "debug": ["debug@4.4.3", "", { "dependencies": { "ms": "^2.1.3" }, "peerDependencies": { "supports-color": "*" }, "optionalPeers": ["supports-color"] }, "sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA=="], + + "diff": ["diff@9.0.0", "", {}, "sha512-svtcdpS8CgJyqAjEQIXdb3OjhFVVYjzGAPO8WGCmRbrml64SPw/jJD4GoE98aR7r25A0XcgrK3F02yw9R/vhQw=="], + + "electron-to-chromium": ["electron-to-chromium@1.5.439", "", {}, "sha512-qu6QIPXhsb+CRcAiTMNjR4A1y/7tCYKkKjr5CZXRVih6qkDf79peZ2BpEU3qoDBNCRrcy3Mra3X9nG5oruuA7Q=="], + + "emoji-regex": ["emoji-regex@10.6.0", "", {}, "sha512-toUI84YS5YmxW219erniWD0CIVOo46xGKColeNQRgOzDorgBi1v4D71/OFzgD9GO2UGKIv1C3Sp8DAn0+j5w7A=="], + + "entities": ["entities@7.0.1", "", {}, "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA=="], + + "es-errors": ["es-errors@1.3.0", "", {}, "sha512-Zf5H2Kxt2xjTvbJvP2ZWLEICxA6j+hAmMzIlypy4xcBg1vKVnx89Wy0GbS+kf5cwCVFFzdCFh2XSCFNULS6csw=="], + + "escalade": ["escalade@3.2.0", "", {}, "sha512-WUj2qlxaQtO4g6Pq5c29GTcWGDyd8itL8zTlipgECz3JesAiiOKotd8JU6otB3PACgG6xkJUyVhboMS+bje/jA=="], + + "find-babel-config": ["find-babel-config@2.1.2", "", { "dependencies": { "json5": "^2.2.3" } }, "sha512-ZfZp1rQyp4gyuxqt1ZqjFGVeVBvmpURMqdIWXbPRfB97Bf6BzdK/xSIbylEINzQ0kB5tlDQfn9HkNXXWsqTqLg=="], + + "find-up": ["find-up@3.0.0", "", { "dependencies": { "locate-path": "^3.0.0" } }, "sha512-1yD6RmLI1XBfxugvORwlck6f75tYL+iR0jqwsOrOxMZyGYqUuDhJ0l4AXdO1iX/FTs9cBAMEk1gWSEx1kSbylg=="], + + "fs.realpath": ["fs.realpath@1.0.0", "", {}, "sha512-OO0pH2lK6a0hZnAdau5ItzHPI6pUlvI7jMVnxUQRtw4owF2wk8lOSabtGDCTP4Ggrg2MbGnWO9X8K1t4+fGMDw=="], + + "function-bind": ["function-bind@1.1.2", "", {}, "sha512-7XHNxH7qX9xG5mIwxkhumTox/MIRNcOgDrxWsMt2pAr23WHp6MrRlN7FBSFpCpr+oVO0F744iUgR82nJMfG2SA=="], + + "gensync": ["gensync@1.0.0-beta.2", "", {}, "sha512-3hN7NaskYvMDLQY55gnW3NQ+mesEAepTqlg+VEbj7zzqEMBVNhzcGYYeqFo/TlYz6eQiFcp1HcsCZO+nGgS8zg=="], + + "get-east-asian-width": ["get-east-asian-width@1.7.0", "", {}, "sha512-XjH1AECxf0giL2V1aU8vKyRR2ppRUb5c0EvT7zuJTokQ74bNo52zOtghqdWIqrhUD79fo3x0WfKZdOqxF6LG1Q=="], + + "glob": ["glob@9.3.5", "", { "dependencies": { "fs.realpath": "^1.0.0", "minimatch": "^8.0.2", "minipass": "^4.2.4", "path-scurry": "^1.6.1" } }, "sha512-e1LleDykUz2Iu+MTYdkSsuWX8lvAjAcs0Xef0lNIu0S2wOAzuTxCJtcd9S3cijlwYF18EsU3rzb8jPVobxDh9Q=="], + + "hasown": ["hasown@2.0.4", "", { "dependencies": { "function-bind": "^1.1.2" } }, "sha512-T2UbfbBEF32wiepXIsMlTW9+dDYC6wMh/t/vYA4tuOMKqWz/n3vr1NFSxQiyP+zk2mXsoMA/i/7qV6LKut1t1A=="], + + "html-entities": ["html-entities@2.3.3", "", {}, "sha512-DV5Ln36z34NNTDgnz0EWGBLZENelNAtkiFA4kyNOG2tDI6Mz1uSWiq1wAKdyjnJwyDiDO7Fa2SO1CTxPXL8VxA=="], + + "is-core-module": ["is-core-module@2.17.0", "", { "dependencies": { "hasown": "^2.0.4" } }, "sha512-J/vG0zBCbIKOQFfufSwyXdMrsohyJIUNkrnmo6WZGzoM7tr/lsbfW5b2BvisL6zsyMzK9UxV9L6c7AoFbyXHOA=="], + + "js-tokens": ["js-tokens@4.0.0", "", {}, "sha512-RdJUflcE3cUzKiMqQgsCu06FPu9UdIJO0beYbPhHN4k6apgJtifcoCtT9bcxOpYBtpD2kCM6Sbzg4CausW/PKQ=="], + + "jsesc": ["jsesc@3.1.0", "", { "bin": { "jsesc": "bin/jsesc" } }, "sha512-/sM3dO2FOzXjKQhJuo0Q173wf2KOo8t4I8vHy6lF9poUp7bKT0/NHE8fPX23PwfhnykfqnC2xRxOnVw5XuGIaA=="], + + "json5": ["json5@2.2.3", "", { "bin": { "json5": "lib/cli.js" } }, "sha512-XmOWe7eyHYH14cLdVPoyg+GOH3rYX++KpzrylJwSW98t3Nk+U8XOl8FWKOgwtzdb8lXGf6zYwDUzeHMWfxasyg=="], + + "locate-path": ["locate-path@3.0.0", "", { "dependencies": { "p-locate": "^3.0.0", "path-exists": "^3.0.0" } }, "sha512-7AO748wWnIhNqAuaty2ZWHkQHRSNfPVIsPIfwEOWO22AmaoVrWavlOcMR5nzTLNYvp36X220/maaRsrec1G65A=="], + + "lru-cache": ["lru-cache@5.1.1", "", { "dependencies": { "yallist": "^3.0.2" } }, "sha512-KpNARQA3Iwv+jTA0utUVVbrh+Jlrr1Fv0e56GGzAFOXN7dk/FviaDW8LHmK52DlcH4WP2n6gI8vN1aesBFgo9w=="], + + "marked": ["marked@17.0.1", "", { "bin": { "marked": "bin/marked.js" } }, "sha512-boeBdiS0ghpWcSwoNm/jJBwdpFaMnZWRzjA6SkUMYb40SVaN1x7mmfGKp0jvexGcx+7y2La5zRZsYFZI6Qpypg=="], + + "minimatch": ["minimatch@8.0.7", "", { "dependencies": { "brace-expansion": "^2.0.1" } }, "sha512-V+1uQNdzybxa14e/p00HZnQNNcTjnRJjDxg2V8wtkjFctq4M7hXFws4oekyTP0Jebeq7QYtpFyOeBAjc88zvYg=="], + + "minipass": ["minipass@4.2.8", "", {}, "sha512-fNzuVyifolSLFL4NzpF+wEF4qrgqaaKX0haXPQEdQ7NKAN+WecoKMHV09YcuL/DHxrUsYQOK3MiuDf7Ip2OXfQ=="], + + "ms": ["ms@2.1.3", "", {}, "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA=="], + + "node-releases": ["node-releases@2.0.57", "", {}, "sha512-kQK9LGGFiHtrWiNhZtA7Qbw17AQz+dmsEKODRIVTXA9+e5MS/2gZEBhYJt13GrAz5/IOZKddH/0Z3TP/Zgo+yw=="], + + "p-limit": ["p-limit@2.3.0", "", { "dependencies": { "p-try": "^2.0.0" } }, "sha512-//88mFWSJx8lxCzwdAABTJL2MyWB12+eIY7MDL2SqLmAkeKU9qxRvWuSyTjm3FUmpBEMuFfckAIqEaVGUDxb6w=="], + + "p-locate": ["p-locate@3.0.0", "", { "dependencies": { "p-limit": "^2.0.0" } }, "sha512-x+12w/To+4GFfgJhBEpiDcLozRJGegY+Ei7/z0tSLkMmxGZNybVMSfWj9aJn8Z5Fc7dBUNJOOVgPv2H7IwulSQ=="], + + "p-try": ["p-try@2.2.0", "", {}, "sha512-R4nPAVTAU0B9D35/Gk3uJf/7XYbQcyohSKdvAxIRSNghFl4e71hVoGnBNQz9cWaXxO2I10KTC+3jMdvvoKw6dQ=="], + + "parse5": ["parse5@7.3.0", "", { "dependencies": { "entities": "^6.0.0" } }, "sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw=="], + + "path-exists": ["path-exists@3.0.0", "", {}, "sha512-bpC7GYwiDYQ4wYLe+FA8lhRjhQCMcQGuSgGGqDkg/QerRWw9CmGRT0iSOVRSZJ29NMLZgIzqaljJ63oaL4NIJQ=="], + + "path-parse": ["path-parse@1.0.7", "", {}, "sha512-LDJzPVEEEPR+y48z93A0Ed0yXb8pAByGWo/k5YYdYgpY2/2EsOsksJrq7lOHxryrVOn1ejG6oAp8ahvOIQD8sw=="], + + "path-scurry": ["path-scurry@1.11.1", "", { "dependencies": { "lru-cache": "^10.2.0", "minipass": "^5.0.0 || ^6.0.2 || ^7.0.0" } }, "sha512-Xa4Nw17FS9ApQFJ9umLiJS4orGjm7ZzwUrwamcGQuHSzDyth9boKDaycYdDcZDuqYATXw4HFXgaqWTctW/v1HA=="], + + "picocolors": ["picocolors@1.1.1", "", {}, "sha512-xceH2snhtb5M9liqDsmEw56le376mTZkEX/jEb/RxNFyegNul7eNslCXP9FDj/Lcu0X8KEyMceP2ntpaHrDEVA=="], + + "pkg-up": ["pkg-up@3.1.0", "", { "dependencies": { "find-up": "^3.0.0" } }, "sha512-nDywThFk1i4BQK4twPQ6TA4RT8bDY96yeuCVBWL3ePARCiEKDRSrNGbFIgUJpLp+XeIR65v8ra7WuJOFUBtkMA=="], + + "reselect": ["reselect@4.1.8", "", {}, "sha512-ab9EmR80F/zQTMNeneUr4cv+jSwPJgIlvEmVwLerwrWVbpLlBuls9XHzIeTFy4cegU2NHBp3va0LKOzU5qFEYQ=="], + + "resolve": ["resolve@1.22.12", "", { "dependencies": { "es-errors": "^1.3.0", "is-core-module": "^2.16.1", "path-parse": "^1.0.7", "supports-preserve-symlinks-flag": "^1.0.0" }, "bin": { "resolve": "bin/resolve" } }, "sha512-TyeJ1zif53BPfHootBGwPRYT1RUt6oGWsaQr8UyZW/eAm9bKoijtvruSDEmZHm92CwS9nj7/fWttqPCgzep8CA=="], + + "s-js": ["s-js@0.4.9", "", {}, "sha512-RtpOm+cM6O0sHg6IA70wH+UC3FZcND+rccBZpBAHzlUgNO2Bm5BN+FnM8+OBxzXdwpKWFwX11JGF0MFRkhSoIQ=="], + + "semver": ["semver@6.3.1", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-BR7VvDCVHO+q2xBEWskxS6DJE1qRnb7DxzUrogb71CWoSficBxYsiAGd+Kl0mmq/MprG9yArRkyrQxTO6XjMzA=="], + + "seroval": ["seroval@1.5.6", "", {}, "sha512-rVQVWjjSvlINzaQPZH5JFqsqEsIWdTxY3iJZCnTL/5gQbXIRooVZKI60tVCkOVfzcRPejboxO2t0P89dg5mQaA=="], + + "seroval-plugins": ["seroval-plugins@1.5.6", "", { "peerDependencies": { "seroval": "^1.0" } }, "sha512-HXuLAX2pu/UByPpaeo/TaMfvMIi+1QqIoPJYCcAtU8QkVNwgR6MPlGuCQTErV1JwraaMbYaWVIBX7mppzGLATQ=="], + + "solid-js": ["solid-js@1.9.12", "", { "dependencies": { "csstype": "^3.1.0", "seroval": "~1.5.0", "seroval-plugins": "~1.5.0" } }, "sha512-QzKaSJq2/iDrWR1As6MHZQ8fQkdOBf8GReYb7L5iKwMGceg7HxDcaOHk0at66tNgn9U2U7dXo8ZZpLIAmGMzgw=="], + + "string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="], + + "strip-ansi": ["strip-ansi@7.1.2", "", { "dependencies": { "ansi-regex": "^6.0.1" } }, "sha512-gmBGslpoQJtgnMAvOVqGZpEz9dyoKTCzy2nfz/n8aIFhN/jCE/rCmcxabB6jOOHV+0WNnylOxaxBQPSvcWklhA=="], + + "supports-preserve-symlinks-flag": ["supports-preserve-symlinks-flag@1.0.0", "", {}, "sha512-ot0WnXS9fgdkgIcePe6RHNk1WA8+muPa6cSjeR3V8K27q9BB1rTE3R1p7Hv0z1ZyAc8s6Vvv8DIyWf681MAt0w=="], + + "typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="], + + "undici-types": ["undici-types@8.9.0", "", {}, "sha512-KTDyRTYX8sWmKXAikPHHSyc63CRPETMctyjKFupcC6OBLXT3xsN0e9aF7m+mIXutFWpUXuedtowG7iLOzp0kQg=="], + + "update-browserslist-db": ["update-browserslist-db@1.3.3", "", { "dependencies": { "escalade": "^3.2.0", "picocolors": "^1.1.1" }, "peerDependencies": { "browserslist": ">= 4.21.0" }, "bin": { "update-browserslist-db": "cli.js" } }, "sha512-pJ2sYawQS0R/WI928Gj5GlPhTGzbMelq0+4INtSYNDV9ErKJcX6xjGWkoG/VnB3dpUm00zALaqkrUD77pO5TDQ=="], + + "web-tree-sitter": ["web-tree-sitter@0.25.10", "", { "peerDependencies": { "@types/emscripten": "^1.40.0" }, "optionalPeers": ["@types/emscripten"] }, "sha512-Y09sF44/13XvgVKgO2cNDw5rGk6s26MgoZPXLESvMXeefBf7i6/73eFurre0IsTW6E14Y0ArIzhUMmjoc7xyzA=="], + + "yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="], + + "babel-plugin-jsx-dom-expressions/@babel/helper-module-imports": ["@babel/helper-module-imports@7.18.6", "", { "dependencies": { "@babel/types": "^7.18.6" } }, "sha512-0NFvs3VkuSYbFi1x2Vd6tKrywq+z/cLeYC/RJNFrIX/30Bf5aiGYbtvGXolEktzJH8o5E5KJ3tT+nkxuuZFVlA=="], + + "parse5/entities": ["entities@6.0.1", "", {}, "sha512-aN97NXWF6AWBTahfVOIrB/NShkzi5H7F9r1s9mD3cDj4Ko5f2qhhVoYMibXF7GlLveb/D2ioWay8lxI97Ven3g=="], + + "path-scurry/lru-cache": ["lru-cache@10.4.3", "", {}, "sha512-JNAzZcXrCt42VGLuYz0zfAzDfAvJWW6AfYlDBQyDV5DClI2m5sAmK+OIO7s59XfsRsWHp02jAJrRadPRGTt6SQ=="], + + "path-scurry/minipass": ["minipass@7.1.3", "", {}, "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A=="], + } +} diff --git a/skills/remove-ai-marks/tui/bunfig.toml b/skills/remove-ai-marks/tui/bunfig.toml new file mode 100644 index 0000000..154f6c6 --- /dev/null +++ b/skills/remove-ai-marks/tui/bunfig.toml @@ -0,0 +1,6 @@ +# The Solid JSX transform, for `bun run` and `bun test` alike: bun reads +# the top-level preload for run only, and [test] preload for test only. +preload = ["@opentui/solid/preload"] + +[test] +preload = ["@opentui/solid/preload"] diff --git a/skills/remove-ai-marks/tui/package.json b/skills/remove-ai-marks/tui/package.json new file mode 100644 index 0000000..13742cc --- /dev/null +++ b/skills/remove-ai-marks/tui/package.json @@ -0,0 +1,19 @@ +{ + "name": "wm-tui", + "private": true, + "type": "module", + "scripts": { + "start": "bun run src/index.tsx", + "test": "bun test", + "typecheck": "tsc --noEmit" + }, + "dependencies": { + "@opentui/core": "0.5.12", + "@opentui/solid": "0.5.12", + "solid-js": "1.9.12" + }, + "devDependencies": { + "@types/bun": "latest", + "typescript": "^5.9.0" + } +} diff --git a/skills/remove-ai-marks/tui/src/app.tsx b/skills/remove-ai-marks/tui/src/app.tsx new file mode 100644 index 0000000..cf2fbf6 --- /dev/null +++ b/skills/remove-ai-marks/tui/src/app.tsx @@ -0,0 +1,615 @@ +import { type InputRenderable, TextAttributes } from "@opentui/core" +import { useKeyboard, useTerminalDimensions } from "@opentui/solid" +import { createMemo, createSignal, For, onCleanup, onMount, Show } from "solid-js" +import { runCommand } from "./commands" +import { outputDisplay, parsePrompt, shortPath } from "./prompt" +import type { FileDone, FileEntry, Finding, Reveal } from "./protocol" +import { type AppController, type Dialog, endpointLabel } from "./store" +import { resultClassColor, theme } from "./theme" +import { Keys } from "./ui/dialog" +import { + ConfirmDialog, + DoctorDialog, + HelpDialog, + HistoryDialog, + ModelDialog, + Palette, + PresetDialog, + WelcomeDialog, +} from "./ui/dialogs" + +const SPINNER = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"] + +/** + * One screen, no tabs. Files on the left, the selected file on the right, a + * prompt at the bottom that takes a path, a `--flag` or a `/command`. The + * file list is also the progress view and the result table. + */ +export function App(props: { ctl: AppController }) { + const ctl = props.ctl + const app = ctl.app + const dims = useTerminalDimensions() + const [frame, setFrame] = createSignal(0) + let input: InputRenderable | undefined + + const timer = setInterval(() => { + if (app.run) setFrame((f) => (f + 1) % SPINNER.length) + }, 80) + onCleanup(() => clearInterval(timer)) + onMount(() => void ctl.start()) + + const dialogOpen = () => app.dialogs.length > 0 + const files = () => app.plan?.files ?? [] + const selectedFile = () => files()[app.selected] + + const submit = (value: string) => { + const parsed = parsePrompt(value) + if (input) input.value = "" + if (parsed.kind === "empty") return + if (parsed.kind === "paths") return ctl.addPaths(parsed.tokens) + if (parsed.kind === "flags") return ctl.applyFlags(parsed.tokens) + runCommand(ctl, parsed.name, parsed.arg) + } + + useKeyboard((key) => { + if (key.ctrl && (key.name === "c" || key.name === "d")) { + key.preventDefault() + return ctl.quit() + } + if (dialogOpen()) return + const name = key.name + if (key.ctrl && name === "p") { + key.preventDefault() + ctl.openDialog({ type: "palette" }) + } else if (key.ctrl && name === "r") { + key.preventDefault() + void ctl.clean() + } else if (key.ctrl && name === "e") { + key.preventDefault() + void ctl.inspect() + } else if (name === "f1") { + ctl.openDialog({ type: "help" }) + } else if (name === "tab") { + key.preventDefault() + ctl.cyclePreset(key.shift ? -1 : 1) + } else if (name === "up" || (key.ctrl && name === "k")) { + key.preventDefault() + ctl.select(app.selected - 1) + } else if (name === "down" || (key.ctrl && name === "j")) { + key.preventDefault() + ctl.select(app.selected + 1) + } else if (name === "pageup") { + ctl.select(app.selected - 10) + } else if (name === "pagedown") { + ctl.select(app.selected + 10) + } else if (name === "escape") { + if (app.run) void ctl.cancel() + else if (input) input.value = "" + } + }) + + const listWidth = () => { + const w = dims().width - 4 + return w < 70 ? w : Math.min(56, Math.max(30, Math.floor(w * 0.42))) + } + const showDetail = () => dims().width - 4 >= 70 + + return ( + +
+ }> + + + + } + > + + + + + + + {(file: () => FileEntry) => } + + + + + (input = r)} /> +