diff --git a/lessismore.py b/lessismore.py index d8c2c72..84f5d7e 100644 --- a/lessismore.py +++ b/lessismore.py @@ -64,8 +64,19 @@ def dedupe_lines(text: str) -> str: return "\n".join(out) -# no '/' in the class: URL paths are ≥64-char alnum+slash runs and they ARE the content -_BLOB = re.compile(r"\b(?:[A-Za-z0-9+]{64,}={0,2}|[0-9a-fA-F]{48,})\b") +# no '/' in the class: URL paths are ≥64-char alnum+slash runs and they ARE the content. +# '-' and '_' admitted for base64url (real JWTs), guarded below so slugs survive. +_BLOB = re.compile(r"\b(?:[A-Za-z0-9+_-]{64,}={0,2}|[0-9a-fA-F]{48,})\b") + + +def _squash_blob(m: "re.Match") -> str: + s = m.group() + # a run with - or _ might be a kebab/snake identifier, not a blob: demand + # blob-typical entropy (digits + both cases) before eating it + if ("-" in s or "_" in s) and not (any(c.isdigit() for c in s) + and s.lower() != s and s.upper() != s): + return s + return f"{s[:12]}[+{len(s) - 16} chars]{s[-4:]}" def squash_blobs(text: str) -> str: @@ -74,26 +85,50 @@ def squash_blobs(text: str) -> str: The tail suffix keeps two different blobs from squashing to the same stub and then being falsely merged as "repeated" by dedupe_lines. """ - return _BLOB.sub(lambda m: f"{m.group()[:12]}[+{len(m.group()) - 16} chars]{m.group()[-4:]}", text) + return _BLOB.sub(_squash_blob, text) +_MON = r"(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)" _TIMESTAMP = re.compile( - r"\b\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}(?:[.,]\d+)?(?:Z|[+-]\d{2}:?\d{2})?\b ?" + r"\b\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}(?:[.,]\d+)?(?:Z|[+-]\d{2}:?\d{2})?\b ?" # ISO-8601 + r"|\b" + _MON + r" {1,2}\d{1,2} \d{2}:\d{2}:\d{2}\b ?" # syslog + r"|\b\d{2}/" + _MON + r"/\d{4}:\d{2}:\d{2}:\d{2}(?: [+-]\d{4})?\b ?" # nginx CLF ) def strip_timestamps(text: str) -> str: - """Line order already encodes sequence; per-line ISO timestamps are ~8 tokens each.""" - # ponytail: ISO-8601 only; add syslog/other formats when a real log needs them + """Line order already encodes sequence; per-line timestamps are ~8 tokens each.""" + # ponytail: epoch timestamps skipped on purpose — any 10-digit number matches return _TIMESTAMP.sub("", text) def minify_json(text: str) -> str: - """Whole-input pretty JSON → minified. Lossless when it fires, untouched when not.""" + """Pretty JSON → minified, lossless when it fires, untouched when not. + + Fires on the whole input AND on multi-line JSON embedded in other text + ('response body:\\n{...}' — the common shape of captured tool output). + Single-line JSON (JSONL) is left alone: ', ' and ',' are 1 token either way; + only the newline+indent of pretty-printing costs real tokens. + """ try: return json.dumps(json.loads(text), separators=(",", ":"), ensure_ascii=False) except ValueError: - return text + pass + dec = json.JSONDecoder() + out, last = [], 0 + for m in re.finditer(r"[{\[](?=[ \t]*\n)", text): # pretty JSON opens then breaks the line + if m.start() < last: + continue # inside a block we already minified + try: + obj, end = dec.raw_decode(text, m.start()) + except ValueError: + continue + mini = json.dumps(obj, separators=(",", ":"), ensure_ascii=False) + if len(mini) < end - m.start(): + out += [text[last:m.start()], mini] + last = end + out.append(text[last:]) + return "".join(out) def collapse_cr(text: str) -> str: @@ -159,12 +194,30 @@ def two_sticks(text: str) -> str: _SKEL = re.compile(r"[^A-Za-z]+") +_DIGITS = re.compile(r"\d+") + + +def _variant_summary(lines: list) -> str: + """Digits are often THE signal (status codes, ports, counts) — summarize the + values a similar-line collapse would otherwise silently eat.""" + runs = [_DIGITS.findall(l) for l in lines] + if not runs[0] or any(len(r) != len(runs[0]) for r in runs): + return "" + parts = [] + for col in zip(*runs): + vals = sorted({int(v) for v in col}) + if len(vals) == 1: + continue + parts.append("/".join(map(str, vals)) if len(vals) <= 4 + else f"{vals[0]}–{vals[-1]}") + return "; values " + ", ".join(parts) if parts else "" def dedupe_similar(text: str) -> str: """Collapse runs of lines identical after masking non-letters — 'same words, different numbers' (progress lines, per-item CI steps). Keeps first and last - so progression endpoints survive.""" + so progression endpoints survive, and summarizes the varying values in the + marker so distinct facts (status codes, ports) aren't lost.""" lines = text.split("\n") out, i = [], 0 while i < len(lines): @@ -172,7 +225,9 @@ def dedupe_similar(text: str) -> str: while j < len(lines) and _SKEL.sub(" ", lines[j]).strip() == k: j += 1 if j - i >= 4 and k: - out += [lines[i], f"[{j - i - 2} similar lines omitted]", lines[j - 1]] + out += [lines[i], + f"[{j - i - 2} similar lines omitted{_variant_summary(lines[i:j])}]", + lines[j - 1]] else: out.extend(lines[i:j]) i = j @@ -216,7 +271,7 @@ LEVELS = { } -def compress(text: str, level: int = 1) -> str: +def compress(text: str, level: int = 2) -> str: for f in LEVELS[level]: text = f(text) return text @@ -323,27 +378,70 @@ def compress_ml(text: str, rate: float = 0.5) -> str: # ---------------------------------------------------------------- CLI +def run_command(cmd: str) -> "tuple[str, int]": + """Run CMD in a shell, return (merged raw output, exit code) — exit code + survives, unlike piping through a filter.""" + import subprocess + r = subprocess.run(cmd, shell=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) + return r.stdout.decode("utf-8", "replace"), r.returncode + + +def hook(level: int = 2, min_chars: int = 2000) -> None: + """Claude Code PostToolUse hook mode: read the hook JSON on stdin, emit + updatedToolOutput JSON with the tool output compressed. Zero deps, no jq. + + Small outputs pass through untouched (marker lines aren't worth the churn). + """ + try: + d = json.load(sys.stdin) + except ValueError: + return # not hook JSON — emit nothing, Claude Code keeps the original + out = d.get("tool_response") or d.get("tool_output") or {} + if isinstance(out, dict): + raw = out.get("stdout") or out.get("output") or "" + else: + raw = str(out) + if len(raw) < min_chars: + return + small = compress(raw, level) + if len(small) >= len(raw): + return + json.dump({"hookSpecificOutput": {"hookEventName": "PostToolUse", + "updatedToolOutput": {"stdout": small}}}, + sys.stdout) + + def main(): import argparse p = argparse.ArgumentParser(prog="lessismore", description=__doc__.splitlines()[0]) p.add_argument("file", nargs="?", help="input file (default: stdin)") - p.add_argument("-l", "--level", type=int, default=1, choices=sorted(LEVELS), - help="1=code-safe 2=logs/dumps 3=prose 4=gist-only caveman (default 1)") + p.add_argument("-l", "--level", type=int, default=2, choices=sorted(LEVELS), + help="1=code-safe 2=logs/dumps 3=prose 4=gist-only caveman (default 2)") p.add_argument("--ml", type=float, metavar="RATE", help="also run LLMLingua-2 keeping RATE of tokens (needs: pip install llmlingua)") p.add_argument("--budget", type=int, metavar="N", help="hard cap output at ~N tokens: keep head+tail, drop the middle") + p.add_argument("--run", metavar="CMD", + help="run CMD in a shell, print its output compressed, exit with its status") + p.add_argument("--hook", action="store_true", + help="Claude Code PostToolUse hook mode (reads hook JSON on stdin)") p.add_argument("--serve", nargs="?", const=7777, type=int, metavar="PORT", help="serve a paste-in demo page on localhost (default port 7777)") a = p.parse_args() if a.serve: return serve(a.serve) + if a.hook: + return hook(a.level) - # newline="" / buffer.read(): keep \r intact for collapse_cr - raw = (open(a.file, encoding="utf-8", errors="replace", newline="").read() if a.file - else sys.stdin.buffer.read().decode("utf-8", "replace")) + if a.run: + raw, code = run_command(a.run) + else: + # newline="" / buffer.read(): keep \r intact for collapse_cr + raw = (open(a.file, encoding="utf-8", errors="replace", newline="").read() if a.file + else sys.stdin.buffer.read().decode("utf-8", "replace")) + code = 0 out = compress(raw, a.level) if a.ml: try: @@ -356,6 +454,7 @@ def main(): before, after = count_tokens(raw), count_tokens(out) print(f"lessismore: {before} → {after} tokens ({1 - after / max(before, 1):.0%} saved)", file=sys.stderr) + sys.exit(code) if __name__ == "__main__": diff --git a/test_lessismore.py b/test_lessismore.py index f0fafd3..f8d4520 100644 --- a/test_lessismore.py +++ b/test_lessismore.py @@ -65,12 +65,18 @@ def test(): # uuids shorten to 8-hex prefix assert compress("id=550e8400-e29b-41d4-a716-446655440000 done\n", 2) == "id=550e8400… done\n" - # same-words-different-numbers runs collapse, endpoints kept + # same-words-different-numbers runs collapse, endpoints kept, ranges summarized sim = "\n".join(f"Downloading chunk {i} of 50 at {i * 3}kbps" for i in range(50)) out = compress(sim, 2) - assert "[48 similar lines omitted]" in out + assert "48 similar lines omitted; values 0–49, 0–147" in out assert "chunk 0 of" in out and "chunk 49 of" in out + # digits are often THE signal: distinct status codes survive the collapse + codes = "\n".join(f"ERROR upstream returned status {c} for /checkout" + for c in (500, 404, 503, 429, 500)) + out = compress(codes, 2) + assert "404/429/500/503" in out + # whole-input pretty JSON minifies losslessly import json pretty = json.dumps({"items": [{"id": i, "name": f"x{i}"} for i in range(10)]}, indent=2) @@ -78,6 +84,29 @@ def test(): assert json.loads(out) == json.loads(pretty) assert len(out) < len(pretty) / 1.5 + # pretty JSON EMBEDDED in other text minifies too — the captured-tool-output shape + emb = "response body:\n" + pretty + "\ndone in 0.31s" + out = compress(emb, 2) + assert out.startswith("response body:\n{") and out.endswith("done in 0.31s") + assert json.loads(out.split("\n")[1]) == json.loads(pretty) + + # real-world JWTs are base64url (- and _) and get squashed... + jwt = "eyJhbGciOiJSUzI1NiIsImtpZCI6Il9abDVGdS0zOSJ9" * 3 + assert "chars]" in squash_blobs("Authorization: Bearer " + jwt) + # ...but a long kebab-case slug is content, not a blob + slug = "-".join(["very", "long", "kebab", "case", "identifier"] * 4) + assert len(slug) >= 64 and squash_blobs(slug) == slug + + # syslog and nginx CLF timestamps strip like ISO ones + assert strip_timestamps("Jul 7 10:00:00 host sshd[1]: boom") == "host sshd[1]: boom" + assert strip_timestamps('[07/Jul/2026:10:00:00 +0000] "GET /"') == '[] "GET /"' + + # level 2 is idempotent: same input, same output, byte for byte — the cache claim + gnarly = ("2026-07-07T10:00:01Z ERROR conn refused db=orders retry\n" * 40 + + sim + "\n" + inter + "\x1b[32mok\x1b[0m\n") + once = compress(gnarly, 2) + assert compress(once, 2) == once + # budget: guaranteed cap, head and tail kept, middle dropped text = "\n".join(f"line {i} of the log with some words" for i in range(200)) out = budget(text, 100) @@ -93,6 +122,26 @@ def test(): input=stdin, capture_output=True) assert r.returncode == 0, r.stderr + # --run compresses the command's output and passes its exit code through + r = subprocess.run([sys.executable, "lessismore.py", "--run", + "printf 'a very long unique diagnostic line here\\n%.0s' 1 2 3 4 5; exit 7"], + capture_output=True, text=True) + assert r.returncode == 7 + assert "repeated 4 more times" in r.stdout + + # --hook: small outputs pass through untouched, big ones come back compressed + import json as _json + payload = _json.dumps({"tool_response": {"stdout": "tiny"}}) + r = subprocess.run([sys.executable, "lessismore.py", "--hook"], + input=payload, capture_output=True, text=True) + assert r.returncode == 0 and r.stdout == "" + payload = _json.dumps({"tool_response": {"stdout": + "ERROR connection refused by upstream database\n" * 200}}) + r = subprocess.run([sys.executable, "lessismore.py", "--hook"], + input=payload, capture_output=True, text=True) + hook_out = _json.loads(r.stdout)["hookSpecificOutput"] + assert "repeated 199 more times" in hook_out["updatedToolOutput"]["stdout"] + print("ok")