Files
Clayton Kim 91a99c4c9d Make the advertised Python 3.8+ floor true, and test it in CI
Insert `from __future__ import annotations` in the 17 files that use
PEP 604/585 annotations in module-level positions evaluated at import
time, so scripts/banned_phrase_scan.py and friends no longer raise
TypeError on Python 3.8/3.9. Add a 3.8 leg to the CI matrix so the
floor claim in README.md is actually gated, and correct the two
imprecise "439 deterministic cases" references to "440 deterministic
script cases (439 pass, 1 documented xfail)".

New scripts must carry the future-import until the floor is raised;
if the maintainer later chooses 3.10+, delete the CI 3.8 leg and
README claim together.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01K6CYksdLbXbTAxcAQjvHz5
2026-07-07 06:13:32 -07:00

140 lines
4.3 KiB
Python
Executable File

#!/usr/bin/env python3
"""
Check change percentage between original and transformed text.
Flags if >40% of words changed (may indicate over-editing).
Pure reordering is exempt: change_percentage still reflects the raw
delete+insert diff, but excessive_change stays false when the transformed
text is a word-level permutation of the original (moved sentences are not
over-editing). SKILL.md's 40% guidance refers to the excessive_change flag.
Usage:
python diff_check.py original.txt transformed.txt
"""
from __future__ import annotations
import sys
import json
import re
from collections import Counter
from difflib import SequenceMatcher
from typing import TypedDict
class DiffResult(TypedDict):
original_word_count: int
transformed_word_count: int
similarity_ratio: float
change_percentage: float
words_added: int
words_removed: int
words_changed: int
excessive_change: bool
flags: list[str]
def split_words(text: str) -> list[str]:
"""Split text into comparison tokens.
Punctuation is tokenized separately (not stripped) so that structural edits
like repunctuation register as change instead of reading as identical.
"""
return re.findall(r'\w+|[^\w\s]', text.lower())
def calculate_diff(original: str, transformed: str) -> DiffResult:
"""Calculate difference metrics between two texts."""
original_words = split_words(original)
transformed_words = split_words(transformed)
# Use SequenceMatcher for word-level diff
matcher = SequenceMatcher(None, original_words, transformed_words)
similarity = matcher.ratio()
# Count operations
words_added = 0
words_removed = 0
words_changed = 0
for tag, i1, i2, j1, j2 in matcher.get_opcodes():
if tag == 'replace':
words_changed += max(i2 - i1, j2 - j1)
elif tag == 'delete':
words_removed += i2 - i1
elif tag == 'insert':
words_added += j2 - j1
# Change percentage (relative to original)
total_changes = words_added + words_removed + words_changed
change_percentage = (
(total_changes / len(original_words) * 100)
if original_words else 0
)
shared_tokens = sum((Counter(original_words) & Counter(transformed_words)).values())
token_overlap = shared_tokens / len(original_words) if original_words else 0
# Flags
flags: list[str] = []
excessive = False
if change_percentage > 40:
if token_overlap >= 0.9:
flags.append(f"Mostly reordered ({token_overlap:.0%} tokens shared)")
else:
flags.append(f"Excessive change ({change_percentage:.1f}% > 40% threshold)")
excessive = True
if len(transformed_words) < len(original_words) * 0.3:
flags.append("Transformed text is less than 30% of original length")
excessive = True
if len(transformed_words) > len(original_words) * 1.5:
flags.append("Transformed text is 50%+ longer than original")
# Length change ratio
length_ratio = len(transformed_words) / len(original_words) if original_words else 0
if length_ratio < 0.5:
flags.append(f"Significant condensation ({length_ratio:.0%} of original)")
elif length_ratio > 1.2:
flags.append(f"Text expanded ({length_ratio:.0%} of original)")
return {
"original_word_count": len(original_words),
"transformed_word_count": len(transformed_words),
"similarity_ratio": round(similarity, 3),
"change_percentage": round(change_percentage, 1),
"words_added": words_added,
"words_removed": words_removed,
"words_changed": words_changed,
"excessive_change": excessive,
"flags": flags
}
def main() -> None:
if len(sys.argv) < 3:
print("Usage: diff_check.py <original.txt> <transformed.txt>")
sys.exit(1)
# Read inputs
try:
with open(sys.argv[1], 'r') as f:
original = f.read()
with open(sys.argv[2], 'r') as f:
transformed = f.read()
except OSError as e:
print(json.dumps({"error": f"Could not read input: {e}"}))
sys.exit(2)
result = calculate_diff(original, transformed)
print(json.dumps(result, indent=2))
# Exit with 1 if excessive change
sys.exit(1 if result["excessive_change"] else 0)
if __name__ == "__main__":
main()