lessismore/lessismore.py
jing 0d4c5464b9 Two-tier template mining + faithful collapse + equal-budget receipt; 0.4.0
alias_templates is now two tiers with exact char accounting:
- tier A masks hex/digit runs INSIDE tokens (run-level constants — dates,
  IP prefixes, the 2 in ssh2 — inline into the template); groups are
  re-costed under whole-token value encoding and take the cheaper
- tier B masks whole digit-bearing tokens on what tier A left, so paths/ids
  varying in letters still group (compile/build logs)
- dedupe_similar skeleton is hex-aware but keeps a leading @tN verbatim:
  refs from different templates can no longer merge as "similar" — an
  earlier build scored 8.9x on LogHub from exactly that silent value
  destruction, and v0.3.0's HDFS number had the same phantom; both gone

Measured (bench_real.py): LogHub 3.0x -> 3.9x, LogDx-CI 2.2x -> 2.3x at
unchanged 95.2% critical-signal retention, LogChunks 1.9x -> 2.1x.
Equal-budget receipt added to bench_real logdx: at a 2k cap,
compress-then-truncate retains 46.6% of critical signals vs 24.7% for
plain truncation. Synthetic eval re-run on shipped code: 8/8 at 6.4x.
Graveyard: Drain-style letter-merging measured flat (3.0x -> 3.0x, Linux
regressed) and documented dead.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-07 18:27:39 +10:00

602 lines
24 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""lessismore — squeeze text before it hits an LLM.
Deterministic passes first (free, safe, cacheable). Optional ML pruning
(LLMLingua-2) only when installed and only worth it on big inputs.
from lessismore import compress
small = compress(big_log, level=2)
$ python3 lessismore.py dump.log -l 2 > small.log
$ tail -5000 app.log | python3 lessismore.py -l 2 | llm ...
"""
import json
import re
import sys
from collections import Counter
from functools import lru_cache
# ---------------------------------------------------------------- tokens
@lru_cache(maxsize=1)
def _encoder():
try:
import tiktoken
return tiktoken.get_encoding("o200k_base")
except ImportError:
return None
def count_tokens(text: str) -> int:
enc = _encoder()
if enc:
return len(enc.encode(text))
return max(1, len(text) // 4) # ponytail: chars/4 heuristic; pip install tiktoken for real counts
# ---------------------------------------------------------------- passes
# Each pass is (str) -> str. Order matters: whitespace before dedupe.
def collapse_whitespace(text: str) -> str:
text = re.sub(r"[ \t]+$", "", text, flags=re.M) # trailing whitespace
text = re.sub(r"(?<=\S)[ \t]{2,}", " ", text) # interior runs (leading indent kept: code-safe)
text = re.sub(r"\n{3,}", "\n\n", text) # blank-line runs
return text
def dedupe_lines(text: str) -> str:
"""Collapse runs of identical lines — the classic log killer."""
lines = text.split("\n")
out, i = [], 0
while i < len(lines):
j = i
while j < len(lines) and lines[j] == lines[i]:
j += 1
run = j - i
# marker must actually be shorter than the lines it replaces
if run >= 4 and lines[i].strip() and (run - 1) * (len(lines[i]) + 1) > 45:
out.append(lines[i])
out.append(f"[previous line repeated {run - 1} more times]")
else:
out.extend(lines[i:j])
i = j
# ponytail: only consecutive repeats; add block-level dedupe (repeated stack traces) if logs demand it
return "\n".join(out)
# no '/' in the class: URL paths are ≥64-char alnum+slash runs and they ARE the content.
# '-' and '_' admitted for base64url (real JWTs), guarded below so slugs survive.
_BLOB = re.compile(r"\b(?:[A-Za-z0-9+_-]{64,}={0,2}|[0-9a-fA-F]{48,})\b")
def _squash_blob(m: "re.Match") -> str:
s = m.group()
# a run with - or _ might be a kebab/snake/camelCase identifier, not a blob:
# demand blob-typical entropy (digits + both cases, no long alpha runs —
# 16+ consecutive letters ~never happens in base64, always in identifiers;
# the identifier case was caught by the LogDx-CI benchmark, not imagination)
if ("-" in s or "_" in s) and (not any(c.isdigit() for c in s)
or s.lower() == s or s.upper() == s
or re.search(r"[A-Za-z]{16,}", s)):
return s
return f"{s[:12]}[+{len(s) - 16} chars]{s[-4:]}"
def squash_blobs(text: str) -> str:
"""Base64/hex runs are token-dense and semantically opaque — keep head and tail.
The tail suffix keeps two different blobs from squashing to the same stub
and then being falsely merged as "repeated" by dedupe_lines.
"""
return _BLOB.sub(_squash_blob, text)
_MON = r"(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)"
# the last four formats were found by benchmarking on real LogHub logs
# (HDFS/BGL/Thunderbird/HealthApp); each is anchored enough not to eat data
_TIMESTAMP = re.compile(
r"\b\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}(?:[.,]\d+)?(?:Z|[+-]\d{2}:?\d{2})?\b ?" # ISO-8601
r"|\b" + _MON + r" {1,2}\d{1,2} \d{2}:\d{2}:\d{2}\b ?" # syslog
r"|\b\d{2}/" + _MON + r"/\d{4}:\d{2}:\d{2}:\d{2}(?: [+-]\d{4})?\b ?" # nginx CLF
r"|\b\d{9,10} \d{4}\.\d{2}\.\d{2}\b ?" # epoch+date pair (BGL/Thunderbird)
r"|\b\d{4}-\d{2}-\d{2}-\d{2}\.\d{2}\.\d{2}\.\d{6}\b ?" # BGL RAS event stamp
r"|(?m:^)\d{6} \d{6} (?=\d+ )" # HDFS 'yymmdd hhmmss pid'
r"|\b\d{8}-\d{2}:\d{2}:\d{2}:\d{3}\b ?" # HealthApp 'yyyymmdd-hh:mm:ss:ms'
)
def strip_timestamps(text: str) -> str:
"""Line order already encodes sequence; per-line timestamps are ~8 tokens each."""
# ponytail: epoch timestamps skipped on purpose — any 10-digit number matches
return _TIMESTAMP.sub("", text)
def minify_json(text: str) -> str:
"""Pretty JSON → minified, lossless when it fires, untouched when not.
Fires on the whole input AND on multi-line JSON embedded in other text
('response body:\\n{...}' — the common shape of captured tool output).
Single-line JSON (JSONL) is left alone: ', ' and ',' are 1 token either way;
only the newline+indent of pretty-printing costs real tokens.
"""
try:
return json.dumps(json.loads(text), separators=(",", ":"), ensure_ascii=False)
except ValueError:
pass
dec = json.JSONDecoder()
out, last = [], 0
for m in re.finditer(r"[{\[](?=[ \t]*\n)", text): # pretty JSON opens then breaks the line
if m.start() < last:
continue # inside a block we already minified
try:
obj, end = dec.raw_decode(text, m.start())
except ValueError:
continue
mini = json.dumps(obj, separators=(",", ":"), ensure_ascii=False)
if len(mini) < end - m.start():
out += [text[last:m.start()], mini]
last = end
out.append(text[last:])
return "".join(out)
def collapse_cr(text: str) -> str:
"""Keep only the final state of \\r-overwritten progress lines (pip/tqdm/wget)."""
text = text.replace("\r\n", "\n")
return "\n".join(l.rsplit("\r", 1)[-1] for l in text.split("\n"))
_ANSI = re.compile(r"\x1b\[[0-9;?]*[A-Za-z]") # ponytail: CSI only; add OSC if titles show up
def strip_ansi(text: str) -> str:
"""Color codes carry nothing for an LLM, and they make identical lines differ."""
return _ANSI.sub("", text)
_UUID = re.compile(r"\b[0-9a-fA-F]{8}-(?:[0-9a-fA-F]{4}-){3}[0-9a-fA-F]{12}\b")
def squash_uuids(text: str) -> str:
"""36 chars → 8-hex prefix, git-short-hash style; cross-references still resolve."""
return _UUID.sub(lambda m: m.group()[:8] + "", text)
_PKGPATH = re.compile(r'[^\s"\']+/(?:site-packages|dist-packages|lib/python3\.\d+)/')
def squash_pkgpaths(text: str) -> str:
"""Traceback path spam: …/httpx/_client.py:1054 is still unique without the venv prefix."""
return _PKGPATH.sub("…/", text)
# the lookbehinds keep "not just X" from becoming "not X" — meaning inversion
_FILLER = re.compile(
r"(?<!\bnot )(?<!n't )"
r"\b(?:could you please|can you please|i would like you to|i want you to|"
r"go ahead and|hey there,?|please|kindly|basically|actually|currently|"
r"really|simply|just|very|quite)\b ?",
re.I,
)
def strip_filler(text: str) -> str:
"""Aggressive: eats words like 'just' anywhere, including inside strings. Prose only."""
return _FILLER.sub("", text)
# caveman-style word dropping: every function word is a whole token.
# NEVER add negations (not/no/never), modals (must/should), or order words
# (before/after) — dropping those changes meaning, not just style.
_STICKS = re.compile(
r"\b(?:the|a|an|is|are|was|were|be|been|being|am|i|we|you|they|it|"
r"that|which|who|have|has|had|do|does|did|there)\b ?",
re.I,
)
def two_sticks(text: str) -> str:
"""Gist-only: strips text to caveman. Fine for articles/transcripts, never instructions."""
# keep IT/US-style acronyms that case-insensitively collide with function words
return _STICKS.sub(lambda m: m.group() if m.group().strip().isupper()
and len(m.group().strip()) > 1 else "", text)
_SKEL = re.compile(r"[^A-Za-z]+")
_DIGITS = re.compile(r"\d+")
def _variant_summary(lines: list) -> str:
"""Digits are often THE signal (status codes, ports, counts) — summarize the
values a similar-line collapse would otherwise silently eat."""
runs = [_DIGITS.findall(l) for l in lines]
if not runs[0] or any(len(r) != len(runs[0]) for r in runs):
return ""
parts = []
for col in zip(*runs):
vals = sorted({int(v) for v in col})
if len(vals) == 1:
continue
parts.append("/".join(map(str, vals)) if len(vals) <= 4
else f"{vals[0]}{vals[-1]}")
return "; values " + ", ".join(parts) if parts else ""
def _skel(l: str) -> str:
# hex runs masked first so goroutine addresses / template-ref hex values
# don't make otherwise-identical lines look distinct; a leading @tN ref
# name stays verbatim so different templates never merge as "similar"
m = _TREF.match(l)
head, tail = (l[:m.end()].strip() + " ", l[m.end():]) if m else ("", l)
return (head + _SKEL.sub(" ", _HEXRUN.sub(" ", tail)).strip()).strip()
def dedupe_similar(text: str) -> str:
"""Collapse runs of lines identical after masking hex runs and non-letters —
'same words, different numbers' (progress lines, per-item CI steps). Keeps
first and last so progression endpoints survive, and summarizes the varying
values in the marker so distinct facts (status codes, ports) aren't lost."""
lines = text.split("\n")
out, i = [], 0
while i < len(lines):
k, j = _skel(lines[i]), i
while j < len(lines) and _skel(lines[j]) == k:
j += 1
if j - i >= 4 and k:
out += [lines[i],
f"[{j - i - 2} similar lines omitted{_variant_summary(lines[i:j])}]",
lines[j - 1]]
else:
out.extend(lines[i:j])
i = j
return "\n".join(out)
_TREF = re.compile(r"@t\d+(?:\s|$)")
# a maximal hex-char run containing at least one digit: catches decimal runs,
# hex ids, and addresses, but never plain words ("added" is hex chars, no digit)
_HEXRUN = re.compile(r"[0-9a-fA-F]*[0-9][0-9a-fA-F]*")
_W = "\x00"
def _mine(lines, live, key_fn, val_fn, min_count):
"""One template-mining tier: group `live` line indexes by key_fn, keep
groups whose EXACT char accounting (original lines vs refs+legend) pays."""
groups = {}
for idx in live:
toks = lines[idx].split()
k, v = key_fn(toks), val_fn(toks)
if v:
g = groups.setdefault(k, [[], [], 0])
g[0].append(idx)
g[1].append(v)
g[2] += len(lines[idx])
worth = {} # insertion order = first appearance: numbering is deterministic
for k, (idxs, vs, orig_chars) in groups.items():
if len(vs) < min_count or len({len(v) for v in vs}) != 1:
continue
if sum(1 for t in k for c in t if c != _W) < 12: # junk guard
continue
# a value column that never varies (ssh2, a constant IP prefix)
# belongs in the template, not repeated in every ref line
const = [all(v[i] == vs[0][i] for v in vs) for i in range(len(vs[0]))]
it, tpl_toks = iter(range(len(vs[0]))), []
for t in k:
if _W not in t:
tpl_toks.append(t)
continue
parts = t.split(_W)
filled = parts[0]
for p in parts[1:]:
i = next(it)
filled += (vs[0][i] if const[i] else "<*>") + p
tpl_toks.append(filled)
tpl = " ".join(tpl_toks)
if "<*>" not in tpl: # all-constant = identical lines = alias_repeats territory
continue
ref_chars = sum(5 + sum(len(x) + 1 for x, c in zip(v, const) if not c)
for v in vs)
if orig_chars - ref_chars - (len(tpl) + 8) > 100: # exact payoff, not a guess
worth[k] = (tpl, const, idxs)
return worth
def _cheaper_whole(lines, tpl_c_idxs, key):
"""A mostly-varying token is cheaper carried whole (`10.251.91.84:52063`)
than as separate runs (`10 251 91 84 52063`). Re-cost the group with
whole-token values and switch encodings if that wins."""
tpl, const, idxs = tpl_c_idxs
wild_tok = [_W in t for t in key]
wvs = [tuple(t for t, w in zip(lines[i].split(), wild_tok) if w) for i in idxs]
wconst = [all(v[i] == wvs[0][i] for v in wvs) for i in range(len(wvs[0]))]
it = iter(range(len(wvs[0])))
wtpl = " ".join((wvs[0][i] if wconst[(i := next(it))] else "<*>") if w else t
for t, w in zip(key, wild_tok))
sub_cost = sum(sum(len(x) + 1 for x, c in zip(_subtok_vals(lines[i].split()), const)
if not c) for i in idxs) + len(tpl)
whole_cost = sum(sum(len(x) + 1 for x, c in zip(v, wconst) if not c)
for v in wvs) + len(wtpl)
if whole_cost < sub_cost and "<*>" in wtpl:
return (wtpl, wconst, idxs), "b"
return tpl_c_idxs, "a"
def _subtok_vals(toks):
return tuple(m for t in toks for m in _HEXRUN.findall(t))
def alias_templates(text: str, min_count: int = 4) -> str:
"""Two-tier Drain-lite template mining. Lossless modulo whitespace runs —
values keep their order, so a line reconstructs by filling <*> left to right.
Tier A masks hex/digit runs INSIDE tokens: `worker-3` and `10.2.3.44`
group as `worker-<*>` and `10.2.3.<*>`, and run-level constants (dates,
IP prefixes, the 2 in ssh2) inline into the template.
Tier B masks WHOLE digit-bearing tokens on whatever tier A left behind:
paths and ids that vary in letters (`.../hashtable.o` vs `.../sampler.o`)
still group, which is what crushes compile/build logs.
Both tiers were forced by public benchmarks (LogHub, LogDx-CI), not
imagination — each exists because a real corpus regressed without it.
"""
lines = text.split("\n")
if any(_TREF.match(l) for l in lines): # already templated, or real @tN content — bail
return text
live = range(len(lines))
worth_a = _mine(lines, live, lambda toks: tuple(_HEXRUN.sub(_W, t) for t in toks),
_subtok_vals, min_count)
taken = {i for _, _, idxs in worth_a.values() for i in idxs}
worth_b = _mine(lines, [i for i in live if i not in taken],
lambda toks: tuple(_W if any(c.isdigit() for c in t) else t
for t in toks),
lambda toks: tuple(t for t in toks if any(c.isdigit() for c in t)),
min_count)
if not (worth_a or worth_b):
return text
entries = [_cheaper_whole(lines, w, k) for k, w in worth_a.items()]
entries += [(w, "b") for w in worth_b.values()]
legend, ref_of = [], {}
for n, ((tpl, const, idxs), tier) in enumerate(entries, 1):
legend.append(f"@t{n} = {tpl}")
for i in idxs:
ref_of[i] = (f"@t{n}", const, tier)
out = []
for i, l in enumerate(lines):
r = ref_of.get(i)
if not r:
out.append(l)
continue
name, const, tier = r
toks = l.split()
v = _subtok_vals(toks) if tier == "a" else \
tuple(t for t in toks if any(c.isdigit() for c in t))
out.append(" ".join([name] + [x for x, c in zip(v, const) if not c]))
return "\n".join(["[templated lines, <*> = per-line values:]"] + legend + [""] + out)
_REF = re.compile(r"@\d+")
def alias_repeats(text: str, min_count: int = 4, min_len: int = 30) -> str:
"""Dictionary-code scattered duplicate lines that consecutive dedupe can't reach.
Lossless — the legend keeps every line verbatim. A code only pays when it
replaces a repeated multi-token sequence; single words are already 1 BPE
token each, so word-level codebooks lose (measured, see README).
"""
lines = text.split("\n")
if any(_REF.fullmatch(l) for l in lines): # already aliased, or real @N content — bail
return text
counts = Counter(l for l in lines if len(l) >= min_len)
# ponytail: c*len>200 chars is the payoff heuristic; tune if legends ever dominate
worth = [l for l, c in counts.items() if c >= min_count and c * len(l) > 200]
if not worth:
return text
ref = {l: f"@{i}" for i, l in enumerate(worth, 1)}
legend = [f"@{i} = {l}" for i, l in enumerate(worth, 1)]
return "\n".join(["[repeated lines aliased:]"] + legend + [""] +
[ref.get(l, l) for l in lines])
# order: strippers leave doubled spaces, so collapse_whitespace runs after them;
# normalized lines then match better in alias_repeats/dedupe_lines/dedupe_similar
_STRIP2 = [minify_json, collapse_cr, strip_ansi, strip_timestamps,
squash_blobs, squash_uuids, squash_pkgpaths]
_MERGE = [collapse_whitespace, alias_repeats, alias_templates, dedupe_lines, dedupe_similar]
LEVELS = {
1: [collapse_whitespace, dedupe_lines], # code-safe
2: _STRIP2 + _MERGE, # logs/dumps/tool output
3: _STRIP2 + [strip_filler] + _MERGE, # prose
4: _STRIP2 + [strip_filler, two_sticks] + _MERGE, # gist-only
}
def compress(text: str, level: int = 2) -> str:
for f in LEVELS[level]:
text = f(text)
return text
def budget(text: str, max_tokens: int) -> str:
"""Hard cap: keep head and tail lines, drop the middle with a marker.
The only pass with GUARANTEED bounded output — run it after compression,
as the backstop, never instead of it.
"""
total = count_tokens(text)
if total <= max_tokens:
return text
lines = text.split("\n")
limit = max(0, max_tokens - 8) # reserve for the marker line
head, tail = [], []
h_tok = t_tok = 0
hi, ti = 0, len(lines) - 1
while hi <= ti:
take_head = h_tok <= t_tok
line = lines[hi] if take_head else lines[ti]
tok = count_tokens(line) + 1
if h_tok + t_tok + tok > limit:
break
if take_head:
head.append(line); h_tok += tok; hi += 1
else:
tail.append(line); t_tok += tok; ti -= 1
if not head and not tail: # one giant line (e.g. minified JSON): slice by chars
keep = limit * 2 # ~4 chars/token, half per side
head, tail = [text[:keep]], [text[-keep:]]
h_tok = count_tokens(head[0]); t_tok = count_tokens(tail[0])
omitted = max(0, total - h_tok - t_tok)
return "\n".join(head + ["[~%d tokens omitted]" % omitted] + list(reversed(tail)))
def serve(port=7777):
"""Paste-in demo page on localhost. Zero deps, binds 127.0.0.1 only."""
import html
import http.server
import urllib.parse
page = ("<!doctype html><meta charset=utf-8><title>lessismore</title>"
"<style>body{font-family:system-ui;max-width:900px;margin:2rem auto;padding:0 1rem;"
"background:#10141a;color:#e6edf3}textarea{width:100%;height:35vh;font-family:monospace;"
"background:#1a2028;color:#e6edf3;border:1px solid #2d3743;border-radius:8px;padding:8px}"
"select,button{font-size:16px;padding:8px 14px;border-radius:8px;border:1px solid #2d3743;"
"background:#1f6feb;color:#fff;cursor:pointer}select{background:#1a2028}</style>"
"<h1>lessismore \U0001F4C9</h1>"
"<form method=post><textarea name=text placeholder='paste your ugliest log here'></textarea>"
"<p>level <select name=level><option>1<option selected>2<option>3<option>4</select> "
"<button>squeeze</button></p></form>{result}")
class Handler(http.server.BaseHTTPRequestHandler):
def log_message(self, *args):
pass
def _page(self, result=""):
body = page.replace("{result}", result).encode("utf-8")
self.send_response(200)
self.send_header("Content-Type", "text/html; charset=utf-8")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def do_GET(self):
self._page()
def do_POST(self):
n = int(self.headers.get("Content-Length", 0))
q = urllib.parse.parse_qs(self.rfile.read(n).decode("utf-8", "replace"))
text = q.get("text", [""])[0]
level = min(4, max(1, int(q.get("level", ["2"])[0])))
out = compress(text, level)
b, a = count_tokens(text), count_tokens(out)
self._page("<p><b>{:,}{:,} tokens ({}% saved)</b></p>"
"<textarea readonly>{}</textarea>".format(
b, a, round(100 * (1 - a / max(b, 1))), html.escape(out)))
print("lessismore UI on http://localhost:%d (Ctrl+C to stop)" % port)
http.server.ThreadingHTTPServer(("127.0.0.1", port), Handler).serve_forever()
# ---------------------------------------------------------------- optional ML pass
@lru_cache(maxsize=1)
def _llmlingua():
from llmlingua import PromptCompressor # pip install llmlingua
return PromptCompressor(
model_name="microsoft/llmlingua-2-xlm-roberta-large-meetingbank",
use_llmlingua2=True,
)
def compress_ml(text: str, rate: float = 0.5) -> str:
"""Perplexity-based token pruning (LLMLingua-2).
Runs a local classifier model — only pays for itself on multi-KB inputs.
Run the deterministic passes first; never feed it code you need verbatim.
"""
return _llmlingua().compress_prompt(text, rate=rate)["compressed_prompt"]
# ---------------------------------------------------------------- CLI
def run_command(cmd: str) -> "tuple[str, int]":
"""Run CMD in a shell, return (merged raw output, exit code) — exit code
survives, unlike piping through a filter."""
import subprocess
r = subprocess.run(cmd, shell=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT)
return r.stdout.decode("utf-8", "replace"), r.returncode
def hook(level: int = 2, min_chars: int = 2000) -> None:
"""Claude Code PostToolUse hook mode: read the hook JSON on stdin, emit
updatedToolOutput JSON with the tool output compressed. Zero deps, no jq.
Small outputs pass through untouched (marker lines aren't worth the churn).
"""
try:
d = json.load(sys.stdin)
except ValueError:
return # not hook JSON — emit nothing, Claude Code keeps the original
out = d.get("tool_response") or d.get("tool_output") or {}
if isinstance(out, dict):
raw = out.get("stdout") or out.get("output") or ""
else:
raw = str(out)
if len(raw) < min_chars:
return
small = compress(raw, level)
if len(small) >= len(raw):
return
json.dump({"hookSpecificOutput": {"hookEventName": "PostToolUse",
"updatedToolOutput": {"stdout": small}}},
sys.stdout)
def main():
import argparse
p = argparse.ArgumentParser(prog="lessismore", description=__doc__.splitlines()[0])
p.add_argument("file", nargs="?", help="input file (default: stdin)")
p.add_argument("-l", "--level", type=int, default=2, choices=sorted(LEVELS),
help="1=code-safe 2=logs/dumps 3=prose 4=gist-only caveman (default 2)")
p.add_argument("--ml", type=float, metavar="RATE",
help="also run LLMLingua-2 keeping RATE of tokens (needs: pip install llmlingua)")
p.add_argument("--budget", type=int, metavar="N",
help="hard cap output at ~N tokens: keep head+tail, drop the middle")
p.add_argument("--run", metavar="CMD",
help="run CMD in a shell, print its output compressed, exit with its status")
p.add_argument("--hook", action="store_true",
help="Claude Code PostToolUse hook mode (reads hook JSON on stdin)")
p.add_argument("--serve", nargs="?", const=7777, type=int, metavar="PORT",
help="serve a paste-in demo page on localhost (default port 7777)")
a = p.parse_args()
if a.serve:
return serve(a.serve)
if a.hook:
return hook(a.level)
if a.run:
raw, code = run_command(a.run)
else:
# newline="" / buffer.read(): keep \r intact for collapse_cr
raw = (open(a.file, encoding="utf-8", errors="replace", newline="").read() if a.file
else sys.stdin.buffer.read().decode("utf-8", "replace"))
code = 0
out = compress(raw, a.level)
if a.ml:
try:
out = compress_ml(out, a.ml)
except ImportError:
sys.exit("--ml needs: pip install llmlingua")
if a.budget:
out = budget(out, a.budget)
sys.stdout.write(out)
before, after = count_tokens(raw), count_tokens(out)
print(f"lessismore: {before}{after} tokens ({1 - after / max(before, 1):.0%} saved)",
file=sys.stderr)
sys.exit(code)
if __name__ == "__main__":
main()