@
Add star-map Angular app, ETL pipeline, and caveman plugin Angular 3D star map (galaxy/system/body views, Three.js rendering, navigation store) plus the NASA ETL tooling that builds the star, exoplanet and solar-system datasets, Playwright e2e suite, and the cs:caveman Claude Code plugin (command, agent, skill). Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> @
This commit is contained in:
@@ -0,0 +1,145 @@
|
||||
#!/usr/bin/env python3
|
||||
"""caveman_compressor.py
|
||||
|
||||
Compress text into "caveman mode" style per the `caveman` skill rules:
|
||||
drop articles, filler, pleasantries, and hedging; abbreviate common
|
||||
technical terms; turn simple causal phrases into `X -> Y` arrows.
|
||||
|
||||
Code blocks (``` ... ```) and inline code (`...`) are left untouched.
|
||||
|
||||
Usage:
|
||||
python caveman_compressor.py "text to compress"
|
||||
python caveman_compressor.py --file some.md
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import re
|
||||
import sys
|
||||
|
||||
# Words/phrases dropped entirely (case-insensitive, whole-word match).
|
||||
ARTICLES = ["a", "an", "the"]
|
||||
FILLER = ["just", "really", "basically", "actually", "simply"]
|
||||
HEDGING = ["might", "maybe", "perhaps", "likely"]
|
||||
|
||||
# Multi-word pleasantries dropped entirely (checked as phrases, longest first).
|
||||
PLEASANTRIES = [
|
||||
"of course",
|
||||
"happy to",
|
||||
"sure thing",
|
||||
"certainly",
|
||||
"sure",
|
||||
]
|
||||
|
||||
DROP_WORDS = ARTICLES + FILLER + HEDGING
|
||||
|
||||
# Common abbreviations. Keys are matched case-insensitively as whole words;
|
||||
# the replacement preserves the target casing shown here.
|
||||
ABBREVIATIONS = {
|
||||
"database": "DB",
|
||||
"databases": "DBs",
|
||||
"authentication": "auth",
|
||||
"configuration": "config",
|
||||
"configurations": "configs",
|
||||
"request": "req",
|
||||
"requests": "reqs",
|
||||
"response": "res",
|
||||
"responses": "res",
|
||||
"function": "fn",
|
||||
"functions": "fns",
|
||||
"implementation": "impl",
|
||||
"implementations": "impls",
|
||||
"environment": "env",
|
||||
"environments": "envs",
|
||||
"dependency": "dep",
|
||||
"dependencies": "deps",
|
||||
"repository": "repo",
|
||||
"repositories": "repos",
|
||||
"documentation": "docs",
|
||||
"application": "app",
|
||||
"applications": "apps",
|
||||
}
|
||||
|
||||
# Causal phrases turned into `X -> Y` arrows.
|
||||
CAUSAL_PHRASES = [
|
||||
"leads to",
|
||||
"results in",
|
||||
"causes",
|
||||
"will cause",
|
||||
]
|
||||
|
||||
# Splits text into segments, tagging fenced code blocks / inline code so
|
||||
# they can be skipped during compression.
|
||||
_CODE_SPLIT_RE = re.compile(r"(```.*?```|`[^`\n]*`)", re.DOTALL)
|
||||
|
||||
|
||||
def _drop_words(text: str) -> str:
|
||||
for phrase in PLEASANTRIES:
|
||||
text = re.sub(
|
||||
r"(?i)\b" + re.escape(phrase) + r"\b[,!]?\s*", "", text
|
||||
)
|
||||
for word in DROP_WORDS:
|
||||
text = re.sub(r"(?i)\b" + re.escape(word) + r"\b\s*", "", text)
|
||||
return text
|
||||
|
||||
|
||||
def _abbreviate(text: str) -> str:
|
||||
for long_form, short_form in ABBREVIATIONS.items():
|
||||
text = re.sub(
|
||||
r"(?i)\b" + re.escape(long_form) + r"\b",
|
||||
short_form,
|
||||
text,
|
||||
)
|
||||
return text
|
||||
|
||||
|
||||
def _arrows(text: str) -> str:
|
||||
for phrase in CAUSAL_PHRASES:
|
||||
text = re.sub(r"(?i)\s*\b" + re.escape(phrase) + r"\b\s*", " -> ", text)
|
||||
return text
|
||||
|
||||
|
||||
def _cleanup_whitespace(text: str) -> str:
|
||||
text = re.sub(r"[ \t]{2,}", " ", text)
|
||||
text = re.sub(r"[ \t]+([,.!?;:])", r"\1", text)
|
||||
text = re.sub(r"\n[ \t]+", "\n", text)
|
||||
text = re.sub(r"^[ \t]+", "", text, flags=re.MULTILINE)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def compress(text: str) -> str:
|
||||
"""Compress `text` into caveman style, preserving code spans."""
|
||||
segments = _CODE_SPLIT_RE.split(text)
|
||||
out = []
|
||||
for segment in segments:
|
||||
if segment.startswith("`"):
|
||||
out.append(segment)
|
||||
continue
|
||||
compressed = segment
|
||||
compressed = _arrows(compressed)
|
||||
compressed = _drop_words(compressed)
|
||||
compressed = _abbreviate(compressed)
|
||||
compressed = _cleanup_whitespace(compressed)
|
||||
out.append(compressed)
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("text", nargs="?", help="Text to compress")
|
||||
parser.add_argument("--file", help="Read text to compress from a file")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.file:
|
||||
with open(args.file, "r", encoding="utf-8") as fh:
|
||||
text = fh.read()
|
||||
elif args.text is not None:
|
||||
text = args.text
|
||||
else:
|
||||
text = sys.stdin.read()
|
||||
|
||||
print(compress(text))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,103 @@
|
||||
#!/usr/bin/env python3
|
||||
"""caveman_lint.py
|
||||
|
||||
Verify that a response follows the `caveman` skill rules: no articles,
|
||||
filler words, pleasantries, or hedging outside of code spans.
|
||||
|
||||
Code blocks (``` ... ```) and inline code (`...`) are ignored by the lint,
|
||||
since their contents are technical and must stay unchanged.
|
||||
|
||||
Exit code: 0 if no violations found, 1 otherwise.
|
||||
|
||||
Usage:
|
||||
python caveman_lint.py "response text"
|
||||
python caveman_lint.py --file some.md
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import re
|
||||
import sys
|
||||
|
||||
ARTICLES = ["a", "an", "the"]
|
||||
FILLER = ["just", "really", "basically", "actually", "simply"]
|
||||
HEDGING = ["might", "maybe", "perhaps", "likely"]
|
||||
PLEASANTRIES = ["sure", "certainly", "of course", "happy to", "sure thing"]
|
||||
|
||||
RULES = {
|
||||
"article": ARTICLES,
|
||||
"filler": FILLER,
|
||||
"hedging": HEDGING,
|
||||
"pleasantry": PLEASANTRIES,
|
||||
}
|
||||
|
||||
_CODE_SPLIT_RE = re.compile(r"(```.*?```|`[^`\n]*`)", re.DOTALL)
|
||||
|
||||
|
||||
def _non_code_segments(text: str):
|
||||
"""Yield (segment_text, start_offset_in_original_text) for every
|
||||
segment of `text` that is NOT inside a fenced/inline code span."""
|
||||
offset = 0
|
||||
for segment in _CODE_SPLIT_RE.split(text):
|
||||
if not segment.startswith("`"):
|
||||
yield segment, offset
|
||||
offset += len(segment)
|
||||
|
||||
|
||||
def find_violations(text: str):
|
||||
"""Return a list of violation dicts: category, word, position, line,
|
||||
context (a short snippet around the match)."""
|
||||
violations = []
|
||||
for category, words in RULES.items():
|
||||
for word in words:
|
||||
pattern = re.compile(r"(?i)\b" + re.escape(word) + r"\b")
|
||||
for segment, offset in _non_code_segments(text):
|
||||
for match in pattern.finditer(segment):
|
||||
pos = offset + match.start()
|
||||
line = text.count("\n", 0, pos) + 1
|
||||
start = max(0, match.start() - 20)
|
||||
end = min(len(segment), match.end() + 20)
|
||||
context = segment[start:end].strip().replace("\n", " ")
|
||||
violations.append(
|
||||
{
|
||||
"category": category,
|
||||
"word": match.group(0),
|
||||
"position": pos,
|
||||
"line": line,
|
||||
"context": context,
|
||||
}
|
||||
)
|
||||
violations.sort(key=lambda v: v["position"])
|
||||
return violations
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("text", nargs="?", help="Response text to lint")
|
||||
parser.add_argument("--file", help="Read response text to lint from a file")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.file:
|
||||
with open(args.file, "r", encoding="utf-8") as fh:
|
||||
text = fh.read()
|
||||
elif args.text is not None:
|
||||
text = args.text
|
||||
else:
|
||||
text = sys.stdin.read()
|
||||
|
||||
violations = find_violations(text)
|
||||
|
||||
if not violations:
|
||||
print("OK: no caveman-rule violations found.")
|
||||
return 0
|
||||
|
||||
print(f"FAIL: {len(violations)} caveman-rule violation(s) found:\n")
|
||||
for v in violations:
|
||||
print(
|
||||
f" line {v['line']} [{v['category']}] '{v['word']}' "
|
||||
f"-> ...{v['context']}..."
|
||||
)
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,103 @@
|
||||
#!/usr/bin/env python3
|
||||
"""token_savings_estimator.py
|
||||
|
||||
Estimate token savings (and $ cost savings) achieved by compressing text
|
||||
into "caveman mode" style, using the `caveman_compressor` module.
|
||||
|
||||
Token counts are estimated with a simple heuristic (~4 chars/token) unless
|
||||
`tiktoken` is installed, in which case it is used for a more accurate count.
|
||||
|
||||
Usage:
|
||||
python token_savings_estimator.py "text" --price-per-mtok 3.00
|
||||
python token_savings_estimator.py --file some.md --price-per-mtok 3.00
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
|
||||
from caveman_compressor import compress # noqa: E402
|
||||
|
||||
CHARS_PER_TOKEN = 4.0
|
||||
|
||||
|
||||
def count_tokens(text: str) -> int:
|
||||
"""Count tokens in `text`, using tiktoken if available, else a
|
||||
character-based heuristic (~4 chars/token, roughly matching common
|
||||
English tokenizers)."""
|
||||
try:
|
||||
import tiktoken
|
||||
|
||||
encoding = tiktoken.get_encoding("cl100k_base")
|
||||
return len(encoding.encode(text))
|
||||
except ImportError:
|
||||
if not text:
|
||||
return 0
|
||||
return max(1, round(len(text) / CHARS_PER_TOKEN))
|
||||
|
||||
|
||||
def estimate_savings(text: str, price_per_mtok: float) -> dict:
|
||||
compressed = compress(text)
|
||||
|
||||
original_tokens = count_tokens(text)
|
||||
compressed_tokens = count_tokens(compressed)
|
||||
saved_tokens = max(0, original_tokens - compressed_tokens)
|
||||
pct_saved = (saved_tokens / original_tokens * 100) if original_tokens else 0.0
|
||||
|
||||
original_cost = original_tokens / 1_000_000 * price_per_mtok
|
||||
compressed_cost = compressed_tokens / 1_000_000 * price_per_mtok
|
||||
saved_cost = original_cost - compressed_cost
|
||||
|
||||
return {
|
||||
"compressed_text": compressed,
|
||||
"original_tokens": original_tokens,
|
||||
"compressed_tokens": compressed_tokens,
|
||||
"saved_tokens": saved_tokens,
|
||||
"pct_saved": pct_saved,
|
||||
"original_cost": original_cost,
|
||||
"compressed_cost": compressed_cost,
|
||||
"saved_cost": saved_cost,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("text", nargs="?", help="Text to analyze")
|
||||
parser.add_argument("--file", help="Read text to analyze from a file")
|
||||
parser.add_argument(
|
||||
"--price-per-mtok",
|
||||
type=float,
|
||||
default=3.00,
|
||||
help="Price in USD per 1,000,000 tokens (default: 3.00)",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.file:
|
||||
with open(args.file, "r", encoding="utf-8") as fh:
|
||||
text = fh.read()
|
||||
elif args.text is not None:
|
||||
text = args.text
|
||||
else:
|
||||
text = sys.stdin.read()
|
||||
|
||||
result = estimate_savings(text, args.price_per_mtok)
|
||||
|
||||
print("--- Caveman compression ---")
|
||||
print(result["compressed_text"])
|
||||
print()
|
||||
print("--- Token savings ---")
|
||||
print(f"Original tokens: {result['original_tokens']}")
|
||||
print(f"Compressed tokens: {result['compressed_tokens']}")
|
||||
print(f"Saved tokens: {result['saved_tokens']} ({result['pct_saved']:.1f}%)")
|
||||
print()
|
||||
print(f"--- Cost @ ${args.price_per_mtok:.2f} / MTok ---")
|
||||
print(f"Original cost: ${result['original_cost']:.6f}")
|
||||
print(f"Compressed cost: ${result['compressed_cost']:.6f}")
|
||||
print(f"Saved cost: ${result['saved_cost']:.6f}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user