179 lines
6.4 KiB
Python
179 lines
6.4 KiB
Python
"""Runtime-identical assistant reply-prose extraction.
|
||
|
||
The Context Switch turn-end watcher classifies agent replies after stripping fenced code
|
||
and tail-biasing the remainder (`docs/CONTEXT_SWITCH.md` §3.2). Training the classifier on
|
||
raw replies would therefore train it on text the runtime never sees, so the assistant-prose
|
||
slice runs its student text through this module, which is a deliberate line-by-line port of
|
||
``HeuristicSummary.contextSwitchReplyProse`` in ``Sources/NucleicCore/Intelligence.swift``.
|
||
|
||
The two implementations are pinned together by
|
||
``Tests/NucleicCoreTests/Fixtures/context-switch-prose.json``, which
|
||
``tests/test_prose_extract.py`` and ``ContextSwitchTests`` both assert against. Change one
|
||
side and the other side's test fails.
|
||
|
||
Two Swift behaviours need explicit modelling in Python:
|
||
|
||
* ``String.count``/``suffix`` count **grapheme clusters**, not code points, so the
|
||
character limit is applied over :func:`graphemes` rather than ``len``. That segmentation
|
||
is a pragmatic subset of UAX #29 (CRLF, combining marks, variation selectors, emoji
|
||
ZWJ sequences, skin-tone modifiers, regional-indicator pairs) — the cases that actually
|
||
occur in agent prose. Anything it does not model degrades to one cluster per scalar,
|
||
which is what Python would have done anyway.
|
||
* ``CharacterSet.whitespacesAndNewlines`` is the Unicode ``White_Space`` property, whereas
|
||
``str.strip()`` also strips U+001C–U+001F. :data:`WHITESPACE` spells the Swift set out so
|
||
a stray information separator in a reply cannot make the two extractions disagree.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import unicodedata
|
||
|
||
|
||
DEFAULT_CHARACTER_LIMIT = 1_200
|
||
|
||
#: Unicode ``White_Space``, matching Swift's ``CharacterSet.whitespacesAndNewlines`` and
|
||
#: ``Character.isWhitespace``. Deliberately excludes U+001C–U+001F, which ``str.strip()``
|
||
#: would otherwise remove.
|
||
WHITESPACE = frozenset(
|
||
[chr(code) for code in range(0x0009, 0x000E)] # tab, LF, VT, FF, CR
|
||
+ [chr(code) for code in range(0x2000, 0x200B)] # en quad through hair space
|
||
+ [
|
||
" ", # space
|
||
"
", # next line
|
||
" ", # no-break space
|
||
" ", # ogham space mark
|
||
"
", # line separator
|
||
"
", # paragraph separator
|
||
" ", # narrow no-break space
|
||
" ", # medium mathematical space
|
||
" ", # ideographic space
|
||
]
|
||
)
|
||
|
||
_ZERO_WIDTH_JOINER = ""
|
||
_EXTEND_CATEGORIES = frozenset({"Mn", "Me", "Mc"})
|
||
|
||
|
||
def _is_extend(character: str) -> bool:
|
||
"""Scalars that attach to the preceding grapheme cluster."""
|
||
|
||
if character in ("︎", "️"): # variation selectors 15/16
|
||
return True
|
||
if "\U000e0100" <= character <= "\U000e01ef": # variation selectors supplement
|
||
return True
|
||
if "\U0001f3fb" <= character <= "\U0001f3ff": # emoji skin-tone modifiers
|
||
return True
|
||
return unicodedata.category(character) in _EXTEND_CATEGORIES
|
||
|
||
|
||
def _is_regional_indicator(character: str) -> bool:
|
||
return "\U0001f1e6" <= character <= "\U0001f1ff"
|
||
|
||
|
||
def graphemes(text: str) -> list[str]:
|
||
"""Segment ``text`` the way Swift's ``Character`` view does, for the cases we see."""
|
||
|
||
clusters: list[str] = []
|
||
index, length = 0, len(text)
|
||
while index < length:
|
||
base = text[index]
|
||
index += 1
|
||
if base == "\r" and index < length and text[index] == "\n":
|
||
clusters.append("\r\n")
|
||
index += 1
|
||
continue
|
||
if unicodedata.category(base) == "Cc":
|
||
# Controls never combine with what follows; CR LF above is the one exception.
|
||
clusters.append(base)
|
||
continue
|
||
cluster = base
|
||
if (
|
||
_is_regional_indicator(base)
|
||
and index < length
|
||
and _is_regional_indicator(text[index])
|
||
):
|
||
cluster += text[index]
|
||
index += 1
|
||
while index < length:
|
||
following = text[index]
|
||
if _is_extend(following):
|
||
cluster += following
|
||
index += 1
|
||
elif following == _ZERO_WIDTH_JOINER and index + 1 < length:
|
||
cluster += following + text[index + 1]
|
||
index += 2
|
||
else:
|
||
break
|
||
clusters.append(cluster)
|
||
return clusters
|
||
|
||
|
||
def trim(text: str) -> str:
|
||
"""``trimmingCharacters(in: .whitespacesAndNewlines)``."""
|
||
|
||
start, end = 0, len(text)
|
||
while start < end and text[start] in WHITESPACE:
|
||
start += 1
|
||
while end > start and text[end - 1] in WHITESPACE:
|
||
end -= 1
|
||
return text[start:end]
|
||
|
||
|
||
def strip_fenced_code(reply: str) -> str:
|
||
"""Drop Markdown fenced blocks, including an unclosed trailing fence.
|
||
|
||
Half-streamed code is code, not evidence that the chat changed purpose. Prose outside
|
||
the fences keeps its line structure so sentence boundaries survive.
|
||
"""
|
||
|
||
fence: str | None = None
|
||
prose_lines: list[str] = []
|
||
for line in reply.split("\n"):
|
||
trimmed = line.lstrip(" \t")
|
||
if trimmed.startswith("```"):
|
||
marker: str | None = "`"
|
||
elif trimmed.startswith("~~~"):
|
||
marker = "~"
|
||
else:
|
||
marker = None
|
||
if fence is not None:
|
||
if marker == fence:
|
||
fence = None
|
||
continue
|
||
if marker is not None:
|
||
fence = marker
|
||
continue
|
||
prose_lines.append(line)
|
||
return "\n".join(prose_lines)
|
||
|
||
|
||
def reply_prose(reply: str, character_limit: int = DEFAULT_CHARACTER_LIMIT) -> str | None:
|
||
"""The exact text the runtime hands the purpose classifier, or ``None`` for no signal."""
|
||
|
||
if character_limit <= 0:
|
||
return None
|
||
prose = trim(strip_fenced_code(reply))
|
||
if not prose:
|
||
return None
|
||
|
||
clusters = graphemes(prose)
|
||
if len(clusters) <= character_limit:
|
||
return prose
|
||
|
||
tail = clusters[-character_limit:]
|
||
# Prefer a whole-word start when the bounded suffix cut through one. If there is no
|
||
# whitespace at all, retain the hard suffix rather than returning an empty signal.
|
||
boundary = next(
|
||
(offset for offset, cluster in enumerate(tail) if cluster[0] in WHITESPACE), None
|
||
)
|
||
if boundary is not None and boundary + 1 < len(tail):
|
||
tail = tail[boundary + 1 :]
|
||
bounded = trim("".join(tail))
|
||
return bounded or None
|
||
|
||
|
||
def ends_in_question(prose: str | None) -> bool:
|
||
"""The runtime's ``replyEndsInQuestion`` signal — a question is not committed drift."""
|
||
|
||
return prose is not None and prose.endswith("?")
|