lessismore/test_lessismore.py
jing 0d4c5464b9 Two-tier template mining + faithful collapse + equal-budget receipt; 0.4.0
alias_templates is now two tiers with exact char accounting:
- tier A masks hex/digit runs INSIDE tokens (run-level constants — dates,
  IP prefixes, the 2 in ssh2 — inline into the template); groups are
  re-costed under whole-token value encoding and take the cheaper
- tier B masks whole digit-bearing tokens on what tier A left, so paths/ids
  varying in letters still group (compile/build logs)
- dedupe_similar skeleton is hex-aware but keeps a leading @tN verbatim:
  refs from different templates can no longer merge as "similar" — an
  earlier build scored 8.9x on LogHub from exactly that silent value
  destruction, and v0.3.0's HDFS number had the same phantom; both gone

Measured (bench_real.py): LogHub 3.0x -> 3.9x, LogDx-CI 2.2x -> 2.3x at
unchanged 95.2% critical-signal retention, LogChunks 1.9x -> 2.1x.
Equal-budget receipt added to bench_real logdx: at a 2k cap,
compress-then-truncate retains 46.6% of critical signals vs 24.7% for
plain truncation. Synthetic eval re-run on shipped code: 8/8 at 6.4x.
Graveyard: Drain-style letter-merging measured flat (3.0x -> 3.0x, Linux
regressed) and documented dead.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-07 18:27:39 +10:00

170 lines
8.0 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import subprocess
import sys
from lessismore import (budget, compress, count_tokens, dedupe_lines,
squash_blobs, strip_filler, strip_timestamps, two_sticks)
def test():
# dedupe: long-line runs collapse, short runs where the marker wouldn't pay stay
line = "ERROR connection refused by upstream database"
assert dedupe_lines((line + "\n") * 5 + "b") == \
line + "\n[previous line repeated 4 more times]\nb"
assert dedupe_lines("a\n" * 5 + "b") == "a\n" * 5 + "b"
assert dedupe_lines(line + "\n" + line + "\nb") == line + "\n" + line + "\nb"
# blobs squash with head+tail stub; URLs are not blobs
assert "[+84 chars]" in squash_blobs("token=" + "A" * 100)
url = ("GET https://api.example.com/api/v1/users/12345/orders/67890/"
"items/status/pending/verbose HTTP/1.1")
assert squash_blobs(url) == url
# distinct blobs keep distinct stubs — no false "repeated" merge at level 2
blobs = "\n".join("stored blob deadbeefcafe" + f"{i:052x}" for i in range(4))
assert "repeated" not in compress(blobs, 2)
# timestamps
assert strip_timestamps("2026-07-07T10:00:00.123Z boom") == "boom"
# filler stripping never inverts negation
assert strip_filler("This is not just a warning.") == "This is not just a warning."
assert strip_filler("just do it") == "do it"
# level 1 is code-safe: indentation and content untouched
code = "def f(x):\n return x + 1\n"
assert compress(code, 1) == code
# two_sticks drops function words, never negations, and spares acronyms
assert two_sticks("The server is down") == "server down"
assert "not" in two_sticks("do not delete the backup")
assert two_sticks("The IT team said it is a US issue") == "IT team said US issue"
# interleaved repeats defeat consecutive dedupe but get dictionary-aliased
inter = ("db connection refused on host alpha-primary\n"
"redis queue depth exceeded on host beta-cache\n") * 30
out = compress(inter, 2)
assert "[repeated lines aliased:]" in out and "@1" in out
assert count_tokens(out) < count_tokens(inter) / 2
# a repetitive log shrinks hard, and stripping leaves no doubled spaces
log = "2026-07-07T10:00:01Z ERROR conn refused db=orders retry\n" * 500
out = compress(log, 2)
assert count_tokens(out) < count_tokens(log) / 50
assert " " not in out
# \r progress redraws keep only the final state
assert compress("downloading 1%\rdownloading 99%\rdone\n", 2) == "done\n"
# ANSI escapes stripped
assert compress("\x1b[32mok\x1b[0m\n", 2) == "ok\n"
# venv path spam squashed, file:line still unique
tb = 'File "/Users/j/venv/lib/python3.14/site-packages/httpx/_client.py", line 5\n'
assert compress(tb, 2) == 'File "…/httpx/_client.py", line 5\n'
# uuids shorten to 8-hex prefix
assert compress("id=550e8400-e29b-41d4-a716-446655440000 done\n", 2) == "id=550e8400… done\n"
# same-words-different-numbers: template mined, endpoints kept, ranges summarized
sim = "\n".join(f"Downloading chunk {i} of 50 at {i * 3}kbps" for i in range(50))
out = compress(sim, 2)
# "of 50", "kbps", and "ssh2"-style run-level constants inline into the template
assert "@t1 = Downloading chunk <*> of 50 at <*>kbps" in out
assert "48 similar lines omitted; values 049, 0147" in out
assert "@t1 0 0" in out and "@t1 49 147" in out
# scattered template repeats (the real-log shape): same message, different
# IPs/ports, interleaved with a second message — dictionary-coded as
# template + per-line values, and every varying value survives
ssh = "\n".join(f"Failed password for {['root', 'admin', 'guest'][i % 3]} "
f"from 10.2.3.{i} port {2200 + i} ssh2\n"
f"pam_unix(sshd:auth): authentication failure; rhost=10.2.3.{i}"
for i in range(24))
out = compress(ssh, 2)
# constant IP prefix and ssh2 inline; interleaved refs never merge as "similar"
assert "@t1 = Failed password for root from 10.2.3.<*> port <*> ssh2" in out
assert "\n@t1 0 2200\n" in out
for v in ("@t4 17 2217", "@t2 23"):
assert v in out, v
assert count_tokens(out) < count_tokens(ssh) / 1.5
# digits are often THE signal: distinct status codes survive the collapse
codes = "\n".join(f"ERROR upstream returned status {c} for /checkout"
for c in (500, 404, 503, 429, 500))
out = compress(codes, 2)
assert "404/429/500/503" in out
# whole-input pretty JSON minifies losslessly
import json
pretty = json.dumps({"items": [{"id": i, "name": f"x{i}"} for i in range(10)]}, indent=2)
out = compress(pretty, 2)
assert json.loads(out) == json.loads(pretty)
assert len(out) < len(pretty) / 1.5
# pretty JSON EMBEDDED in other text minifies too — the captured-tool-output shape
emb = "response body:\n" + pretty + "\ndone in 0.31s"
out = compress(emb, 2)
assert out.startswith("response body:\n{") and out.endswith("done in 0.31s")
assert json.loads(out.split("\n")[1]) == json.loads(pretty)
# real-world JWTs are base64url (- and _) and get squashed...
jwt = "eyJhbGciOiJSUzI1NiIsImtpZCI6Il9abDVGdS0zOSJ9" * 3
assert "chars]" in squash_blobs("Authorization: Bearer " + jwt)
# ...but a long kebab-case slug is content, not a blob
slug = "-".join(["very", "long", "kebab", "case", "identifier"] * 4)
assert len(slug) >= 64 and squash_blobs(slug) == slug
# ...and so is a camelCase test name with digits (LogDx-CI regression)
ident = "codeFixMissingTypeAnnotationOnExports52-generics-oversimplifiedTypes"
assert squash_blobs(ident) == ident
# syslog and nginx CLF timestamps strip like ISO ones
assert strip_timestamps("Jul 7 10:00:00 host sshd[1]: boom") == "host sshd[1]: boom"
assert strip_timestamps('[07/Jul/2026:10:00:00 +0000] "GET /"') == '[] "GET /"'
# level 2 is idempotent: same input, same output, byte for byte — the cache claim
gnarly = ("2026-07-07T10:00:01Z ERROR conn refused db=orders retry\n" * 40
+ sim + "\n" + inter + "\x1b[32mok\x1b[0m\n")
once = compress(gnarly, 2)
assert compress(once, 2) == once
# budget: guaranteed cap, head and tail kept, middle dropped
text = "\n".join(f"line {i} of the log with some words" for i in range(200))
out = budget(text, 100)
assert count_tokens(out) <= 110
assert "line 0 " in out and "line 199" in out and "tokens omitted]" in out
assert budget("tiny", 100) == "tiny"
giant = ("word " * 3000).strip() # single huge line falls back to char slicing
assert count_tokens(budget(giant, 50)) <= 80
# CLI survives empty and non-UTF-8 stdin
for stdin in (b"", b"ok\n\xff\xfe\n"):
r = subprocess.run([sys.executable, "lessismore.py"],
input=stdin, capture_output=True)
assert r.returncode == 0, r.stderr
# --run compresses the command's output and passes its exit code through
r = subprocess.run([sys.executable, "lessismore.py", "--run",
"printf 'a very long unique diagnostic line here\\n%.0s' 1 2 3 4 5; exit 7"],
capture_output=True, text=True)
assert r.returncode == 7
assert "repeated 4 more times" in r.stdout
# --hook: small outputs pass through untouched, big ones come back compressed
import json as _json
payload = _json.dumps({"tool_response": {"stdout": "tiny"}})
r = subprocess.run([sys.executable, "lessismore.py", "--hook"],
input=payload, capture_output=True, text=True)
assert r.returncode == 0 and r.stdout == ""
payload = _json.dumps({"tool_response": {"stdout":
"ERROR connection refused by upstream database\n" * 200}})
r = subprocess.run([sys.executable, "lessismore.py", "--hook"],
input=payload, capture_output=True, text=True)
hook_out = _json.loads(r.stdout)["hookSpecificOutput"]
assert "repeated 199 more times" in hook_out["updatedToolOutput"]["stdout"]
print("ok")
if __name__ == "__main__":
test()