alias_templates is now two tiers with exact char accounting: - tier A masks hex/digit runs INSIDE tokens (run-level constants — dates, IP prefixes, the 2 in ssh2 — inline into the template); groups are re-costed under whole-token value encoding and take the cheaper - tier B masks whole digit-bearing tokens on what tier A left, so paths/ids varying in letters still group (compile/build logs) - dedupe_similar skeleton is hex-aware but keeps a leading @tN verbatim: refs from different templates can no longer merge as "similar" — an earlier build scored 8.9x on LogHub from exactly that silent value destruction, and v0.3.0's HDFS number had the same phantom; both gone Measured (bench_real.py): LogHub 3.0x -> 3.9x, LogDx-CI 2.2x -> 2.3x at unchanged 95.2% critical-signal retention, LogChunks 1.9x -> 2.1x. Equal-budget receipt added to bench_real logdx: at a 2k cap, compress-then-truncate retains 46.6% of critical signals vs 24.7% for plain truncation. Synthetic eval re-run on shipped code: 8/8 at 6.4x. Graveyard: Drain-style letter-merging measured flat (3.0x -> 3.0x, Linux regressed) and documented dead. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
170 lines
8.0 KiB
Python
170 lines
8.0 KiB
Python
import subprocess
|
||
import sys
|
||
|
||
from lessismore import (budget, compress, count_tokens, dedupe_lines,
|
||
squash_blobs, strip_filler, strip_timestamps, two_sticks)
|
||
|
||
|
||
def test():
|
||
# dedupe: long-line runs collapse, short runs where the marker wouldn't pay stay
|
||
line = "ERROR connection refused by upstream database"
|
||
assert dedupe_lines((line + "\n") * 5 + "b") == \
|
||
line + "\n[previous line repeated 4 more times]\nb"
|
||
assert dedupe_lines("a\n" * 5 + "b") == "a\n" * 5 + "b"
|
||
assert dedupe_lines(line + "\n" + line + "\nb") == line + "\n" + line + "\nb"
|
||
|
||
# blobs squash with head+tail stub; URLs are not blobs
|
||
assert "[+84 chars]" in squash_blobs("token=" + "A" * 100)
|
||
url = ("GET https://api.example.com/api/v1/users/12345/orders/67890/"
|
||
"items/status/pending/verbose HTTP/1.1")
|
||
assert squash_blobs(url) == url
|
||
|
||
# distinct blobs keep distinct stubs — no false "repeated" merge at level 2
|
||
blobs = "\n".join("stored blob deadbeefcafe" + f"{i:052x}" for i in range(4))
|
||
assert "repeated" not in compress(blobs, 2)
|
||
|
||
# timestamps
|
||
assert strip_timestamps("2026-07-07T10:00:00.123Z boom") == "boom"
|
||
|
||
# filler stripping never inverts negation
|
||
assert strip_filler("This is not just a warning.") == "This is not just a warning."
|
||
assert strip_filler("just do it") == "do it"
|
||
|
||
# level 1 is code-safe: indentation and content untouched
|
||
code = "def f(x):\n return x + 1\n"
|
||
assert compress(code, 1) == code
|
||
|
||
# two_sticks drops function words, never negations, and spares acronyms
|
||
assert two_sticks("The server is down") == "server down"
|
||
assert "not" in two_sticks("do not delete the backup")
|
||
assert two_sticks("The IT team said it is a US issue") == "IT team said US issue"
|
||
|
||
# interleaved repeats defeat consecutive dedupe but get dictionary-aliased
|
||
inter = ("db connection refused on host alpha-primary\n"
|
||
"redis queue depth exceeded on host beta-cache\n") * 30
|
||
out = compress(inter, 2)
|
||
assert "[repeated lines aliased:]" in out and "@1" in out
|
||
assert count_tokens(out) < count_tokens(inter) / 2
|
||
|
||
# a repetitive log shrinks hard, and stripping leaves no doubled spaces
|
||
log = "2026-07-07T10:00:01Z ERROR conn refused db=orders retry\n" * 500
|
||
out = compress(log, 2)
|
||
assert count_tokens(out) < count_tokens(log) / 50
|
||
assert " " not in out
|
||
|
||
# \r progress redraws keep only the final state
|
||
assert compress("downloading 1%\rdownloading 99%\rdone\n", 2) == "done\n"
|
||
|
||
# ANSI escapes stripped
|
||
assert compress("\x1b[32mok\x1b[0m\n", 2) == "ok\n"
|
||
|
||
# venv path spam squashed, file:line still unique
|
||
tb = 'File "/Users/j/venv/lib/python3.14/site-packages/httpx/_client.py", line 5\n'
|
||
assert compress(tb, 2) == 'File "…/httpx/_client.py", line 5\n'
|
||
|
||
# uuids shorten to 8-hex prefix
|
||
assert compress("id=550e8400-e29b-41d4-a716-446655440000 done\n", 2) == "id=550e8400… done\n"
|
||
|
||
# same-words-different-numbers: template mined, endpoints kept, ranges summarized
|
||
sim = "\n".join(f"Downloading chunk {i} of 50 at {i * 3}kbps" for i in range(50))
|
||
out = compress(sim, 2)
|
||
# "of 50", "kbps", and "ssh2"-style run-level constants inline into the template
|
||
assert "@t1 = Downloading chunk <*> of 50 at <*>kbps" in out
|
||
assert "48 similar lines omitted; values 0–49, 0–147" in out
|
||
assert "@t1 0 0" in out and "@t1 49 147" in out
|
||
|
||
# scattered template repeats (the real-log shape): same message, different
|
||
# IPs/ports, interleaved with a second message — dictionary-coded as
|
||
# template + per-line values, and every varying value survives
|
||
ssh = "\n".join(f"Failed password for {['root', 'admin', 'guest'][i % 3]} "
|
||
f"from 10.2.3.{i} port {2200 + i} ssh2\n"
|
||
f"pam_unix(sshd:auth): authentication failure; rhost=10.2.3.{i}"
|
||
for i in range(24))
|
||
out = compress(ssh, 2)
|
||
# constant IP prefix and ssh2 inline; interleaved refs never merge as "similar"
|
||
assert "@t1 = Failed password for root from 10.2.3.<*> port <*> ssh2" in out
|
||
assert "\n@t1 0 2200\n" in out
|
||
for v in ("@t4 17 2217", "@t2 23"):
|
||
assert v in out, v
|
||
assert count_tokens(out) < count_tokens(ssh) / 1.5
|
||
|
||
# digits are often THE signal: distinct status codes survive the collapse
|
||
codes = "\n".join(f"ERROR upstream returned status {c} for /checkout"
|
||
for c in (500, 404, 503, 429, 500))
|
||
out = compress(codes, 2)
|
||
assert "404/429/500/503" in out
|
||
|
||
# whole-input pretty JSON minifies losslessly
|
||
import json
|
||
pretty = json.dumps({"items": [{"id": i, "name": f"x{i}"} for i in range(10)]}, indent=2)
|
||
out = compress(pretty, 2)
|
||
assert json.loads(out) == json.loads(pretty)
|
||
assert len(out) < len(pretty) / 1.5
|
||
|
||
# pretty JSON EMBEDDED in other text minifies too — the captured-tool-output shape
|
||
emb = "response body:\n" + pretty + "\ndone in 0.31s"
|
||
out = compress(emb, 2)
|
||
assert out.startswith("response body:\n{") and out.endswith("done in 0.31s")
|
||
assert json.loads(out.split("\n")[1]) == json.loads(pretty)
|
||
|
||
# real-world JWTs are base64url (- and _) and get squashed...
|
||
jwt = "eyJhbGciOiJSUzI1NiIsImtpZCI6Il9abDVGdS0zOSJ9" * 3
|
||
assert "chars]" in squash_blobs("Authorization: Bearer " + jwt)
|
||
# ...but a long kebab-case slug is content, not a blob
|
||
slug = "-".join(["very", "long", "kebab", "case", "identifier"] * 4)
|
||
assert len(slug) >= 64 and squash_blobs(slug) == slug
|
||
# ...and so is a camelCase test name with digits (LogDx-CI regression)
|
||
ident = "codeFixMissingTypeAnnotationOnExports52-generics-oversimplifiedTypes"
|
||
assert squash_blobs(ident) == ident
|
||
|
||
# syslog and nginx CLF timestamps strip like ISO ones
|
||
assert strip_timestamps("Jul 7 10:00:00 host sshd[1]: boom") == "host sshd[1]: boom"
|
||
assert strip_timestamps('[07/Jul/2026:10:00:00 +0000] "GET /"') == '[] "GET /"'
|
||
|
||
# level 2 is idempotent: same input, same output, byte for byte — the cache claim
|
||
gnarly = ("2026-07-07T10:00:01Z ERROR conn refused db=orders retry\n" * 40
|
||
+ sim + "\n" + inter + "\x1b[32mok\x1b[0m\n")
|
||
once = compress(gnarly, 2)
|
||
assert compress(once, 2) == once
|
||
|
||
# budget: guaranteed cap, head and tail kept, middle dropped
|
||
text = "\n".join(f"line {i} of the log with some words" for i in range(200))
|
||
out = budget(text, 100)
|
||
assert count_tokens(out) <= 110
|
||
assert "line 0 " in out and "line 199" in out and "tokens omitted]" in out
|
||
assert budget("tiny", 100) == "tiny"
|
||
giant = ("word " * 3000).strip() # single huge line falls back to char slicing
|
||
assert count_tokens(budget(giant, 50)) <= 80
|
||
|
||
# CLI survives empty and non-UTF-8 stdin
|
||
for stdin in (b"", b"ok\n\xff\xfe\n"):
|
||
r = subprocess.run([sys.executable, "lessismore.py"],
|
||
input=stdin, capture_output=True)
|
||
assert r.returncode == 0, r.stderr
|
||
|
||
# --run compresses the command's output and passes its exit code through
|
||
r = subprocess.run([sys.executable, "lessismore.py", "--run",
|
||
"printf 'a very long unique diagnostic line here\\n%.0s' 1 2 3 4 5; exit 7"],
|
||
capture_output=True, text=True)
|
||
assert r.returncode == 7
|
||
assert "repeated 4 more times" in r.stdout
|
||
|
||
# --hook: small outputs pass through untouched, big ones come back compressed
|
||
import json as _json
|
||
payload = _json.dumps({"tool_response": {"stdout": "tiny"}})
|
||
r = subprocess.run([sys.executable, "lessismore.py", "--hook"],
|
||
input=payload, capture_output=True, text=True)
|
||
assert r.returncode == 0 and r.stdout == ""
|
||
payload = _json.dumps({"tool_response": {"stdout":
|
||
"ERROR connection refused by upstream database\n" * 200}})
|
||
r = subprocess.run([sys.executable, "lessismore.py", "--hook"],
|
||
input=payload, capture_output=True, text=True)
|
||
hook_out = _json.loads(r.stdout)["hookSpecificOutput"]
|
||
assert "repeated 199 more times" in hook_out["updatedToolOutput"]["stdout"]
|
||
|
||
print("ok")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
test()
|