Files
nucleic-purpose-classifier/prose_extract.py
T

179 lines
6.4 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Runtime-identical assistant reply-prose extraction.
The Context Switch turn-end watcher classifies agent replies after stripping fenced code
and tail-biasing the remainder (`docs/CONTEXT_SWITCH.md` §3.2). Training the classifier on
raw replies would therefore train it on text the runtime never sees, so the assistant-prose
slice runs its student text through this module, which is a deliberate line-by-line port of
``HeuristicSummary.contextSwitchReplyProse`` in ``Sources/NucleicCore/Intelligence.swift``.
The two implementations are pinned together by
``Tests/NucleicCoreTests/Fixtures/context-switch-prose.json``, which
``tests/test_prose_extract.py`` and ``ContextSwitchTests`` both assert against. Change one
side and the other side's test fails.
Two Swift behaviours need explicit modelling in Python:
* ``String.count``/``suffix`` count **grapheme clusters**, not code points, so the
character limit is applied over :func:`graphemes` rather than ``len``. That segmentation
is a pragmatic subset of UAX #29 (CRLF, combining marks, variation selectors, emoji
ZWJ sequences, skin-tone modifiers, regional-indicator pairs) — the cases that actually
occur in agent prose. Anything it does not model degrades to one cluster per scalar,
which is what Python would have done anyway.
* ``CharacterSet.whitespacesAndNewlines`` is the Unicode ``White_Space`` property, whereas
``str.strip()`` also strips U+001C–U+001F. :data:`WHITESPACE` spells the Swift set out so
a stray information separator in a reply cannot make the two extractions disagree.
"""
from __future__ import annotations
import unicodedata
DEFAULT_CHARACTER_LIMIT = 1_200
#: Unicode ``White_Space``, matching Swift's ``CharacterSet.whitespacesAndNewlines`` and
#: ``Character.isWhitespace``. Deliberately excludes U+001C–U+001F, which ``str.strip()``
#: would otherwise remove.
WHITESPACE = frozenset(
[chr(code) for code in range(0x0009, 0x000E)] # tab, LF, VT, FF, CR
+ [chr(code) for code in range(0x2000, 0x200B)] # en quad through hair space
+ [
" ", # space
"…", # next line
" ", # no-break space
" ", # ogham space mark
"
", # line separator
"
", # paragraph separator
" ", # narrow no-break space
" ", # medium mathematical space
" ", # ideographic space
]
)
_ZERO_WIDTH_JOINER = "‍"
_EXTEND_CATEGORIES = frozenset({"Mn", "Me", "Mc"})
def _is_extend(character: str) -> bool:
"""Scalars that attach to the preceding grapheme cluster."""
if character in ("︎", "️"): # variation selectors 15/16
return True
if "\U000e0100" <= character <= "\U000e01ef": # variation selectors supplement
return True
if "\U0001f3fb" <= character <= "\U0001f3ff": # emoji skin-tone modifiers
return True
return unicodedata.category(character) in _EXTEND_CATEGORIES
def _is_regional_indicator(character: str) -> bool:
return "\U0001f1e6" <= character <= "\U0001f1ff"
def graphemes(text: str) -> list[str]:
"""Segment ``text`` the way Swift's ``Character`` view does, for the cases we see."""
clusters: list[str] = []
index, length = 0, len(text)
while index < length:
base = text[index]
index += 1
if base == "\r" and index < length and text[index] == "\n":
clusters.append("\r\n")
index += 1
continue
if unicodedata.category(base) == "Cc":
# Controls never combine with what follows; CR LF above is the one exception.
clusters.append(base)
continue
cluster = base
if (
_is_regional_indicator(base)
and index < length
and _is_regional_indicator(text[index])
):
cluster += text[index]
index += 1
while index < length:
following = text[index]
if _is_extend(following):
cluster += following
index += 1
elif following == _ZERO_WIDTH_JOINER and index + 1 < length:
cluster += following + text[index + 1]
index += 2
else:
break
clusters.append(cluster)
return clusters
def trim(text: str) -> str:
"""``trimmingCharacters(in: .whitespacesAndNewlines)``."""
start, end = 0, len(text)
while start < end and text[start] in WHITESPACE:
start += 1
while end > start and text[end - 1] in WHITESPACE:
end -= 1
return text[start:end]
def strip_fenced_code(reply: str) -> str:
"""Drop Markdown fenced blocks, including an unclosed trailing fence.
Half-streamed code is code, not evidence that the chat changed purpose. Prose outside
the fences keeps its line structure so sentence boundaries survive.
"""
fence: str | None = None
prose_lines: list[str] = []
for line in reply.split("\n"):
trimmed = line.lstrip(" \t")
if trimmed.startswith("```"):
marker: str | None = "`"
elif trimmed.startswith("~~~"):
marker = "~"
else:
marker = None
if fence is not None:
if marker == fence:
fence = None
continue
if marker is not None:
fence = marker
continue
prose_lines.append(line)
return "\n".join(prose_lines)
def reply_prose(reply: str, character_limit: int = DEFAULT_CHARACTER_LIMIT) -> str | None:
"""The exact text the runtime hands the purpose classifier, or ``None`` for no signal."""
if character_limit <= 0:
return None
prose = trim(strip_fenced_code(reply))
if not prose:
return None
clusters = graphemes(prose)
if len(clusters) <= character_limit:
return prose
tail = clusters[-character_limit:]
# Prefer a whole-word start when the bounded suffix cut through one. If there is no
# whitespace at all, retain the hard suffix rather than returning an empty signal.
boundary = next(
(offset for offset, cluster in enumerate(tail) if cluster[0] in WHITESPACE), None
)
if boundary is not None and boundary + 1 < len(tail):
tail = tail[boundary + 1 :]
bounded = trim("".join(tail))
return bounded or None
def ends_in_question(prose: str | None) -> bool:
"""The runtime's ``replyEndsInQuestion`` signal — a question is not committed drift."""
return prose is not None and prose.endswith("?")