fix: report unmeasurable burstiness instead of scoring it as max LLM-like (#141)

* fix: report unmeasurable burstiness instead of scoring it as max LLM-like

compute_burstiness returned cv=0.0 for an empty (or single-sentence)
list, which the tiering read as perfectly uniform prose — the strongest
LLM-likeness signal — so any file whose body yielded no parseable
sentences (e.g. entirely wrapped in a code fence) scored falsely high
while word_count stayed high enough to defeat the small-sample dampener
(#132). On a real export this hit 159 of 162 files as pure artifact.

compute_burstiness now returns None for the CV when it cannot be
measured; the scoring site drops the burstiness component and
renormalizes the composite over AI-phrase density and lexical
diversity, with an explicit note. burstiness_cv is null in JSON for
unmeasurable cases, the CLI prints n/a instead of crashing, and the
uniform-cadence finding guards on cv not being None.

* style: satisfy ruff import sorting and formatting

---------

Co-authored-by: yzxcj797 <yzxcj797@users.noreply.github.com>
Co-authored-by: Guillaume Meyer (The Opinionated Man) <1385518+guillaumemeyer@users.noreply.github.com>
Co-authored-by: Guillaume Meyer <guillaumemeyer@users.noreply.github.com>
This commit is contained in:
yzxcj797
2026-08-18 12:14:27 -07:00
committed by GitHub
co-authored by yzxcj797 Guillaume Meyer Guillaume Meyer
parent dc74db15a2
commit d5fdaae8a1
3 changed files with 95 additions and 12 deletions
+36 -11
View File
@@ -122,7 +122,7 @@ class StylometryReport:
path: str
word_count: int
sentence_count: int
burstiness_cv: float
burstiness_cv: float | None
lexical_diversity: float
ai_ngram_density: float
matched_markers: list[dict[str, Any]]
@@ -137,7 +137,9 @@ class StylometryReport:
"path": self.path,
"word_count": self.word_count,
"sentence_count": self.sentence_count,
"burstiness_cv": round(self.burstiness_cv, 4),
"burstiness_cv": round(self.burstiness_cv, 4)
if self.burstiness_cv is not None
else None,
"lexical_diversity": round(self.lexical_diversity, 4),
"ai_ngram_density": round(self.ai_ngram_density, 4),
"matched_markers": self.matched_markers,
@@ -177,16 +179,22 @@ def extract_words(text: str) -> list[str]:
return [w.lower() for w in RE_WORDS.findall(text)]
def compute_burstiness(sentences: list[str]) -> tuple[float, float, float]:
"""Compute mean sentence word length, standard deviation, and coefficient of variation (CV)."""
def compute_burstiness(sentences: list[str]) -> tuple[float, float, float | None]:
"""Compute mean sentence word length, standard deviation, and coefficient of variation (CV).
CV is ``None`` when it cannot be measured (no sentences, or fewer than two
with words): a one-or-zero-sample CV is undefined, and reporting it as 0.0
made the caller's tiering read "perfectly uniform" — the strongest
LLM-likeness signal — for text whose body yielded no sentences at all (#132).
"""
if not sentences:
return 0.0, 0.0, 0.0
return 0.0, 0.0, None
lengths = [len(extract_words(s)) for s in sentences]
lengths = [L for L in lengths if L > 0]
if len(lengths) < 2:
mean_len = float(lengths[0]) if lengths else 0.0
return mean_len, 0.0, 0.0
return mean_len, 0.0, None
mean_len = sum(lengths) / len(lengths)
variance = sum((x - mean_len) ** 2 for x in lengths) / (len(lengths) - 1)
@@ -264,7 +272,7 @@ def score_text_stylometry(text: str, path: str = "<text>") -> StylometryReport:
path=path,
word_count=word_count,
sentence_count=sentence_count,
burstiness_cv=0.0,
burstiness_cv=None,
lexical_diversity=compute_mattr(words),
ai_ngram_density=0.0,
matched_markers=[m.to_dict() for m in marker_matches],
@@ -286,7 +294,13 @@ def score_text_stylometry(text: str, path: str = "<text>") -> StylometryReport:
# 3. Component Sub-scores (0.0 to 1.0)
# Burstiness subscore: low CV (<0.35) is strongly characteristic of LLMs; high CV (>0.60) is human
if cv < 0.25:
burstiness_score: float | None
if cv is None:
# Fewer than two parseable sentences: burstiness is unmeasurable, not
# maximally LLM-like. The composite renormalizes over the remaining
# components instead of crediting the strongest signal (#132).
burstiness_score = None
elif cv < 0.25:
burstiness_score = 0.95
elif cv < 0.35:
burstiness_score = 0.80
@@ -313,7 +327,13 @@ def score_text_stylometry(text: str, path: str = "<text>") -> StylometryReport:
diversity_score = 0.4 if 0.68 <= mattr <= 0.76 else 0.1
# 4. Composite Scoring & Small-Sample Dampening
raw_composite = (burstiness_score * 0.45) + (ngram_score * 0.45) + (diversity_score * 0.10)
if burstiness_score is None:
notes.append(
"Sentence burstiness unavailable (fewer than 2 parsed sentences — e.g. body wrapped in a code fence); composite renormalized over AI-phrase density and lexical diversity"
)
raw_composite = ((ngram_score * 0.45) + (diversity_score * 0.10)) / 0.55
else:
raw_composite = (burstiness_score * 0.45) + (ngram_score * 0.45) + (diversity_score * 0.10)
# Dampening factor: scales smoothly from 0.4 at MIN_SAMPLE_WORDS up to 1.0 at FULL_WEIGHT_WORDS
if word_count < FULL_WEIGHT_WORDS:
@@ -333,7 +353,7 @@ def score_text_stylometry(text: str, path: str = "<text>") -> StylometryReport:
for m in marker_matches:
findings.append(f"AI cadence phrase '{m.phrase}' ({m.count}x)")
if cv < 0.35 and sentence_count >= 3:
if cv is not None and cv < 0.35 and sentence_count >= 3:
findings.append(f"Unnaturally uniform sentence cadence (CV={cv:.2f} < 0.35)")
if ngram_density >= 1.0:
@@ -372,7 +392,12 @@ def print_human_stylometry_report(report: StylometryReport, explain: bool = Fals
print(f"AI Probability: {report.score * 100:.1f}% (score: {report.score:.3f})")
print(f"Word Count: {report.word_count}")
print(f"Sentence Count: {report.sentence_count}")
print(f"Sentence CV: {report.burstiness_cv:.3f}")
cv_display = (
f"{report.burstiness_cv:.3f}"
if report.burstiness_cv is not None
else "n/a (fewer than 2 sentences)"
)
print(f"Sentence CV: {cv_display}")
print(f"Lexical Diversity: {report.lexical_diversity:.3f} (MATTR)")
print(f"AI Marker Density: {report.ai_ngram_density:.3f} / 100 words")
+2 -1
View File
@@ -42,8 +42,9 @@ def test_sentence_and_word_extraction():
def test_burstiness_variance():
# Empty / single sentence edge cases
assert compute_burstiness([]) == (0.0, 0.0, 0.0)
assert compute_burstiness([]) == (0.0, 0.0, None)
assert compute_burstiness(["Only one sentence here."])[1] == 0.0
assert compute_burstiness(["Only one sentence here."])[2] is None
# Uniform sentences (every sentence is 5 words) -> CV should be 0.0
uniform = [
+57
View File
@@ -0,0 +1,57 @@
"""Regression tests for empty-sentence burstiness handling (#132).
A body whose sentences yield nothing parseable (e.g. entirely wrapped in a
code fence) must be reported as unmeasurable, not as maximally LLM-like.
"""
from __future__ import annotations
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
SCRIPTS = ROOT / "service" / "scripts"
sys.path.insert(0, str(SCRIPTS))
from score_stylometry import compute_burstiness
from score_stylometry import score_text_stylometry as score_text
PLAIN = (
"We ran the probe for three weeks. Traffic was uneven. "
"Some days nothing arrived at all, and then a single autonomous system would saturate the link for "
"forty hours before disappearing without any obvious trigger, which made the early averages useless. "
"I assumed misconfiguration at first. It wasn't. "
"The second dataset disagreed with the first, which was inconvenient, because the whole analysis plan "
"had quietly assumed the two would move together. They do not move together. "
"One caveat remains. Coverage is thin."
)
FENCED = "```tex\n" + PLAIN + "\n```\n"
def test_burstiness_reports_unmeasurable_cv():
assert compute_burstiness([])[2] is None
assert compute_burstiness(["One sentence only."])[2] is None
def test_fenced_body_is_not_scored_as_maximally_llm_like():
plain = score_text(PLAIN, path="plain.md")
fenced = score_text(FENCED, path="fenced.md")
assert plain.sentence_count >= 5
assert fenced.sentence_count == 0
# The unmeasurable case surfaces as null CV plus an explicit note…
assert fenced.burstiness_cv is None
assert any("burstiness unavailable" in n for n in fenced.notes)
# …and the composite no longer credits the strongest LLM signal: the
# 13.6x inflation from the issue collapses to the same neighborhood as
# the plain file (both are ordinary prose without AI markers).
assert fenced.score <= plain.score + 0.05
def test_plain_body_still_scores_with_burstiness():
plain = score_text(PLAIN, path="plain.md")
assert plain.burstiness_cv is not None and plain.burstiness_cv > 0.25
assert not any("burstiness unavailable" in n for n in plain.notes)