Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,12 @@ Read one preset from `presets/` before writing.

- Cut throat-clearing and scaffolding. Start with the claim.
- Replace inflated importance with the concrete fact.
- For headings and slide titles, name the test, mechanism, result, decision, or
completion criterion. Rewrite calendar/container agency ("Week 1 ends with
..."), tool personification ("the judge lies"), vague uplift transformations
("taste becomes infrastructure"), generic topic handles, and counting/journey
frames. Do not ban accurate causal subjects such as "examples calibrate the
judge." See `references/taboo-phrases.md` § Headings and Slide Titles.
- Prefer short, direct sentences, but avoid telegraphic staccato.
- Use em-dashes sparingly. A single appositive dash can be fine; clusters are a
tell. Never trade a dash for a comma splice; if a dash is wrong, use a period.
Expand Down
68 changes: 68 additions & 0 deletions evals/adversarial-evals.json
Original file line number Diff line number Diff line change
Expand Up @@ -12887,6 +12887,74 @@
"equals": 0
}
]
},
{
"id": "OWNER-07",
"category": "scanner_false_negative",
"title": "Owner specimen: a calendar container cannot deliver a slide outcome",
"target": "script",
"command": [
"python3",
"scripts/banned_phrase_scan.py"
],
"stdin": "Week 1 ends with a calibrated eval.",
"failure_mode": "A slide title makes a time container the actor that delivers a product outcome. The sentence is grammatical, but the title hides the actual completion criterion behind narrative agency.",
"correct_behavior": "Flagged as headline_container_agency (soft). A precise replacement names the evidence or action, such as 'The Week 1 capstone is a clean run with recorded agreement.'",
"assertions": [
{
"type": "json",
"path": "total_violations",
"gte": 1
},
{
"type": "violation_category_equals",
"value": "headline_container_agency"
}
]
},
{
"id": "FP-92",
"category": "scanner_false_positive",
"protects": "headline_container_agency",
"title": "A literal calendar boundary is not narrative agency",
"target": "script",
"command": [
"python3",
"scripts/banned_phrase_scan.py"
],
"stdin": "Week 1 ends on Friday.",
"failure_mode": "A blunt 'Week N ends ...' pattern would flag an ordinary temporal fact.",
"correct_behavior": "No violations. 'Ends on Friday' describes the calendar boundary; it does not claim that the week delivered an artifact or outcome.",
"assertions": [
{
"type": "json",
"path": "total_violations",
"equals": 0
}
]
},
{
"id": "SKILL-TITLE-01",
"category": "headline_reconstruction",
"title": "Rewrite slide titles as tests, mechanisms, results, or decisions",
"target": "skill",
"prompt": "Rewrite these slide titles without changing the underlying claims: (1) Week 1 ends with a calibrated eval. (2) Your judge is lying to someone. (3) The package outlives the cohort. (4) Taste becomes infrastructure. Return four replacement titles only.",
"failure_mode": "The skill removes fixed phrases but preserves category-error agency, vague strategic transformations, or generic topic handles in high-salience titles.",
"correct_behavior": "Each replacement names a concrete test, mechanism, result, decision, action, or completion criterion. It removes calendar/container agency and tool personification without flattening the actual claim into a generic label.",
"assertions": [
{
"type": "judge",
"check": "All four replacement titles are concrete and preserve the distinct underlying claim of the original."
},
{
"type": "judge",
"check": "No replacement assigns volitional, narrative, or outcome-producing agency to a week, judge tool, package, score, data, or other inanimate container."
},
{
"type": "judge",
"check": "No replacement uses vague uplift language, journey/paradigm/pillars framing, or a generic one-word topic handle."
}
]
}
]
}
14 changes: 14 additions & 0 deletions evals/build_shared_benchmark.py
Original file line number Diff line number Diff line change
Expand Up @@ -57,6 +57,7 @@
"SKILL-WARMTH-01": "tune",
"SKILL-MACRO-01": "tune",
"SKILL-TIER-01": "tune",
"SKILL-TITLE-01": "tune",
# holdout — measure, do not tune
"SKILL-DEHEDGE-02": "holdout",
"SKILL-LIST-01": "holdout",
Expand Down Expand Up @@ -111,6 +112,7 @@
"SKILL-MACRO-01": "essay",
"SKILL-MACRO-02": "report",
"SKILL-TIER-01": "legal",
"SKILL-TITLE-01": "presentation",
}

# Difficulty is a coarse hint for reporting, not a gate.
Expand Down Expand Up @@ -325,6 +327,18 @@ def _difflib_ratio(case_id: str, fixture: str, minimum: float) -> dict:
_assertion("skill-inject-01-keeps-roadmap", "regex", pattern=r"\broadmap\b"),
],
"SKILL-NEWPAT-01": [_assertion("skill-newpat-01-removes-missed-patterns", "excludes_any", values=["speak for themselves", "underscores the importance"])],
"SKILL-TITLE-01": [
_assertion(
"skill-title-01-removes-category-error-titles",
"excludes_any",
values=[
"Week 1 ends with",
"judge is lying",
"package outlives",
"Taste becomes infrastructure",
],
),
],
"SKILL-SHORT-01": [_assertion("skill-short-01-keeps-text", "regex", pattern=r"\bship it\b")],
"SKILL-EMDASH-01": [
_assertion("skill-emdash-01-keeps-meaning", "regex", pattern=r"\bmarket readiness\b"),
Expand Down
53 changes: 53 additions & 0 deletions evals/shared-benchmark.json
Original file line number Diff line number Diff line change
Expand Up @@ -1692,6 +1692,59 @@
"adversarial",
"failure_mode:Small detector agents can import the whole monolithic skill,"
]
},
{
"id": "SKILL-TITLE-01",
"split": "tune",
"kind": "headline_reconstruction",
"domain": "presentation",
"difficulty": "hard",
"trigger_type": "explicit",
"success_goals": [
"Rewrite slide titles as tests, mechanisms, results, or decisions"
],
"prompt": "Rewrite these slide titles without changing the underlying claims: (1) Week 1 ends with a calibrated eval. (2) Your judge is lying to someone. (3) The package outlives the cohort. (4) Taste becomes infrastructure. Return four replacement titles only.",
"expected_behavior": [
"Each replacement names a concrete test, mechanism, result, decision, action, or completion criterion. It removes calendar/container agency and tool personification without flattening the actual claim into a generic label."
],
"assertions": [
{
"name": "skill-title-01-judge-1",
"type": "judge",
"rubric": [
"All four replacement titles are concrete and preserve the distinct underlying claim of the original."
]
},
{
"name": "skill-title-01-judge-2",
"type": "judge",
"rubric": [
"No replacement assigns volitional, narrative, or outcome-producing agency to a week, judge tool, package, score, data, or other inanimate container."
]
},
{
"name": "skill-title-01-judge-3",
"type": "judge",
"rubric": [
"No replacement uses vague uplift language, journey/paradigm/pillars framing, or a generic one-word topic handle."
]
},
{
"name": "skill-title-01-removes-category-error-titles",
"type": "excludes_any",
"values": [
"Week 1 ends with",
"judge is lying",
"package outlives",
"Taste becomes infrastructure"
]
}
],
"tags": [
"headline_reconstruction",
"adversarial",
"failure_mode:The skill removes fixed phrases but preserves category-error"
]
}
],
"ablations": [
Expand Down
1 change: 1 addition & 0 deletions references/packs/manifest.json
Original file line number Diff line number Diff line change
Expand Up @@ -59,6 +59,7 @@
"formatting",
"fragment_template",
"headline_cadence",
"headline_container_agency",
"imperative_slogan",
"negative_listing",
"negative_parallelism",
Expand Down
1 change: 1 addition & 0 deletions references/packs/pack-voice.md
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ Use this pack for manufactured voice and the skill's own replacement tells. Do n
- Contrastive-definition tails ("X is a build step, not an output."), two-beat imperative slogans ("Emit tokens. Ship bytes."), repeated "<plural noun> that <verb>." fragments, "One X, N Y." numeric parallelism, and abstractions that "ship inside" things.
- Standalone slogan/spec fragment lines and headers: "N X, one Y." slogan cadence ("Four presets, one input.") and "N noun-phrase, past-participle ..." spec-sheet fragments ("Eight criteria, scored 1 to 5."). Flagged only when the whole line is the fragment; the same shape embedded in a sentence is a literal count and stays clean. Soft.
- Headline slogan cadence: the "Short statement. Short statement." two-beat rhythm repeated across a document's headlines ("One command. A real URL." / "Reviewers click. The agent fixes."). One is voice; three or more is template grammar. Frequency-gated, soft.
- Headline container agency: a calendar or program container delivers an artifact or outcome ("Week 1 ends with a calibrated eval"; "Demo day closes the cohort"). Name the completion criterion, mechanism, result, or decision instead. Literal boundaries such as "Week 1 ends on Friday" stay clean. Soft.
- Tool anthropomorphism: an inanimate tool-noun given human agency. Reflexive self-agency fires anywhere ("The suite defends itself." / "The rules update themselves." / "It graded its own reflection."); strongly-volitional verbs (decides, hunts, wants, knows, believes, cares, refuses, judges, thinks) fire only on a standalone headline line ("The bench decides which model does which job." / "It hunts instances, not word lists."). Ordinary technical register stays clean ("the parser reads the file", "the model learns the distribution", "the test cleans up after itself"). Soft.
- Punctuation performance: em-dash clusters, exclamation overuse, decorative bolding, title case inside body prose.
- Warmth stripped into telegraphese: an email or narrative that loses natural softeners and sounds colder than the source.
Expand Down
32 changes: 32 additions & 0 deletions references/taboo-phrases.md
Original file line number Diff line number Diff line change
Expand Up @@ -384,6 +384,38 @@ limit: in plain text "standalone headline" can only be approximated as "a line w
whole content is the short-sentence pair," so the tell is gated on frequency, not on
proof that the line is a real heading.

### Headings and Slide Titles

Write the title after the slide or section has earned its point. A strong title
names the test, mechanism, result, decision, or completion criterion the viewer
should inspect.

Prefer:

- "The Week 1 capstone is a clean run with recorded agreement."
- "Order, verbosity, and self-preference bias the judge."
- "Each capstone must clear three gates."
- "Can one packet reconstruct both products?"

Rewrite:

- Calendar or program containers delivering outcomes: "Week 1 ends with a
calibrated eval"; "Demo day closes the cohort."
- Tools, scores, data, or packages given narrative or volitional agency: "Your
judge is lying"; "A score hides five failures"; "The package outlives the
cohort."
- Abstract uplift transformations: "Taste becomes infrastructure"; "Unlocking
the power of clinical intelligence."
- Generic handles that make the body do all the work: "Architecture", "Proof
plan", "Three pillars", "What the budget buys."
- Repeated counting or journey frames that describe the deck instead of the
claim.

This is a semantic rule, not a blanket subject-verb ban. "Examples calibrate the
judge" and "The gate fails the build" describe mechanisms. "Week 1 ends on
Friday" describes a literal boundary. The problem is category-error agency or a
title that withholds the actual point.

---

## Significance & Legacy Inflation
Expand Down
11 changes: 11 additions & 0 deletions scripts/banned_phrase_scan.py
Original file line number Diff line number Diff line change
Expand Up @@ -1024,6 +1024,17 @@ def _line_col_context(text: str, line_starts: list[int], pos: int) -> tuple[int,
"severity": "soft",
"suggestion": "A tool doesn't decide, want, or know. Say what it does mechanically."
},
# Headline/container agency: a calendar or program container is made to
# deliver the outcome of a slide ("Week 1 ends with a calibrated eval.",
# "Demo day closes the cohort."). Literal boundaries stay clean: "Week 1
# ends on Friday." does not match because only "ends with" carries an
# artifact/result here. Whole-line anchoring keeps this in title territory.
{
"pattern": r"(?:^|\n)[ \t]*(?:#{1,6}[ \t]*|>[ \t]*|[-*+][ \t]+)?(?:\*\*)?(?:(?:week|month|quarter|phase|module|chapter|day)\s+(?:\d+|one|two|three|four|five|six|seven|eight|nine|ten)|demo\s+day)\s+(?:ends?\s+with|delivers?|produces?|builds?|earns?|creates?|unlocks?|closes?)\b[^.!?\n]{0,70}[.!?]?(?:\*\*)?[ \t]*(?=\n|$)",
"category": "headline_container_agency",
"severity": "soft",
"suggestion": "A time or program container does not deliver the outcome. Name the completion criterion, mechanism, result, or decision."
},

# Em-dash overuse
{
Expand Down
6 changes: 5 additions & 1 deletion scripts/contribute.py
Original file line number Diff line number Diff line change
Expand Up @@ -161,6 +161,10 @@ def render_report(
if include_rec:
rows.append("| CONTRIB-REC | REC | existing-word recall still flags |")
quoted = "\n".join(f"> {line}" if line else ">" for line in snippet.splitlines())
pattern_added = manifest.get(
"pattern_added",
f"TODO: regex or phrase for `{manifest['tell']}`",
)
return (
f"# Add {manifest['category']} pattern: {shorten(str(manifest['tell']))}\n\n"
"## The specimen\n\n"
Expand All @@ -171,7 +175,7 @@ def render_report(
"## Why it's an AI-ism\n\n"
f"{manifest.get('rationale', 'TODO: explain why this phrase is a reusable AI-writing tell in 2-4 sentences.')}\n\n"
"## Detection\n\n"
f"- Pattern added: {manifest.get('pattern_added', f'TODO: regex or phrase for `{manifest['tell']}`')}\n"
f"- Pattern added: {pattern_added}\n"
f"- Severity: {manifest.get('severity', 'TODO: hard or soft')}\n"
f"- Gating rationale: {manifest.get('gating_rationale', 'TODO: explain literal-use boundary')}\n"
"- Catalog entry location: references/taboo-phrases.md\n\n"
Expand Down
Loading