179 lines
6.4 KiB
Python
179 lines
6.4 KiB
Python
"""Runtime-identical assistant reply-prose extraction.
|
||||
|
|
|
|||
|
|
The Context Switch turn-end watcher classifies agent replies after stripping fenced code
|
|||
|
|
and tail-biasing the remainder (`docs/CONTEXT_SWITCH.md` §3.2). Training the classifier on
|
|||
|
|
raw replies would therefore train it on text the runtime never sees, so the assistant-prose
|
|||
|
|
slice runs its student text through this module, which is a deliberate line-by-line port of
|
|||
|
|
``HeuristicSummary.contextSwitchReplyProse`` in ``Sources/NucleicCore/Intelligence.swift``.
|
|||
|
|
|
|||
|
|
The two implementations are pinned together by
|
|||
|
|
``Tests/NucleicCoreTests/Fixtures/context-switch-prose.json``, which
|
|||
|
|
``tests/test_prose_extract.py`` and ``ContextSwitchTests`` both assert against. Change one
|
|||
|
|
side and the other side's test fails.
|
|||
|
|
|
|||
|
|
Two Swift behaviours need explicit modelling in Python:
|
|||
|
|
|
|||
|
|
* ``String.count``/``suffix`` count **grapheme clusters**, not code points, so the
|
|||
|
|
character limit is applied over :func:`graphemes` rather than ``len``. That segmentation
|
|||
|
|
is a pragmatic subset of UAX #29 (CRLF, combining marks, variation selectors, emoji
|
|||
|
|
ZWJ sequences, skin-tone modifiers, regional-indicator pairs) — the cases that actually
|
|||
|
|
occur in agent prose. Anything it does not model degrades to one cluster per scalar,
|
|||
|
|
which is what Python would have done anyway.
|
|||
|
|
* ``CharacterSet.whitespacesAndNewlines`` is the Unicode ``White_Space`` property, whereas
|
|||
|
|
``str.strip()`` also strips U+001C–U+001F. :data:`WHITESPACE` spells the Swift set out so
|
|||
|
|
a stray information separator in a reply cannot make the two extractions disagree.
|
|||
|
|
"""
|
|||
|
|
|
|||
|
|
from __future__ import annotations
|
|||
|
|
|
|||
|
|
import unicodedata
|
|||
|
|
|
|||
|
|
|
|||
|
|
DEFAULT_CHARACTER_LIMIT = 1_200
|
|||
|
|
|
|||
|
|
#: Unicode ``White_Space``, matching Swift's ``CharacterSet.whitespacesAndNewlines`` and
|
|||
|
|
#: ``Character.isWhitespace``. Deliberately excludes U+001C–U+001F, which ``str.strip()``
|
|||
|
|
#: would otherwise remove.
|
|||
|
|
WHITESPACE = frozenset(
|
|||
|
|
[chr(code) for code in range(0x0009, 0x000E)] # tab, LF, VT, FF, CR
|
|||
|
|
+ [chr(code) for code in range(0x2000, 0x200B)] # en quad through hair space
|
|||
|
|
+ [
|
|||
|
|
" ", # space
|
|||
|
|
"
", # next line
|
|||
|
|
" ", # no-break space
|
|||
|
|
" ", # ogham space mark
|
|||
|
|
"
", # line separator
|
|||
|
|
"
", # paragraph separator
|
|||
|
|
" ", # narrow no-break space
|
|||
|
|
" ", # medium mathematical space
|
|||
|
|
" ", # ideographic space
|
|||
|
|
]
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
_ZERO_WIDTH_JOINER = ""
|
|||
|
|
_EXTEND_CATEGORIES = frozenset({"Mn", "Me", "Mc"})
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _is_extend(character: str) -> bool:
|
|||
|
|
"""Scalars that attach to the preceding grapheme cluster."""
|
|||
|
|
|
|||
|
|
if character in ("︎", "️"): # variation selectors 15/16
|
|||
|
|
return True
|
|||
|
|
if "\U000e0100" <= character <= "\U000e01ef": # variation selectors supplement
|
|||
|
|
return True
|
|||
|
|
if "\U0001f3fb" <= character <= "\U0001f3ff": # emoji skin-tone modifiers
|
|||
|
|
return True
|
|||
|
|
return unicodedata.category(character) in _EXTEND_CATEGORIES
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _is_regional_indicator(character: str) -> bool:
|
|||
|
|
return "\U0001f1e6" <= character <= "\U0001f1ff"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def graphemes(text: str) -> list[str]:
|
|||
|
|
"""Segment ``text`` the way Swift's ``Character`` view does, for the cases we see."""
|
|||
|
|
|
|||
|
|
clusters: list[str] = []
|
|||
|
|
index, length = 0, len(text)
|
|||
|
|
while index < length:
|
|||
|
|
base = text[index]
|
|||
|
|
index += 1
|
|||
|
|
if base == "\r" and index < length and text[index] == "\n":
|
|||
|
|
clusters.append("\r\n")
|
|||
|
|
index += 1
|
|||
|
|
continue
|
|||
|
|
if unicodedata.category(base) == "Cc":
|
|||
|
|
# Controls never combine with what follows; CR LF above is the one exception.
|
|||
|
|
clusters.append(base)
|
|||
|
|
continue
|
|||
|
|
cluster = base
|
|||
|
|
if (
|
|||
|
|
_is_regional_indicator(base)
|
|||
|
|
and index < length
|
|||
|
|
and _is_regional_indicator(text[index])
|
|||
|
|
):
|
|||
|
|
cluster += text[index]
|
|||
|
|
index += 1
|
|||
|
|
while index < length:
|
|||
|
|
following = text[index]
|
|||
|
|
if _is_extend(following):
|
|||
|
|
cluster += following
|
|||
|
|
index += 1
|
|||
|
|
elif following == _ZERO_WIDTH_JOINER and index + 1 < length:
|
|||
|
|
cluster += following + text[index + 1]
|
|||
|
|
index += 2
|
|||
|
|
else:
|
|||
|
|
break
|
|||
|
|
clusters.append(cluster)
|
|||
|
|
return clusters
|
|||
|
|
|
|||
|
|
|
|||
|
|
def trim(text: str) -> str:
|
|||
|
|
"""``trimmingCharacters(in: .whitespacesAndNewlines)``."""
|
|||
|
|
|
|||
|
|
start, end = 0, len(text)
|
|||
|
|
while start < end and text[start] in WHITESPACE:
|
|||
|
|
start += 1
|
|||
|
|
while end > start and text[end - 1] in WHITESPACE:
|
|||
|
|
end -= 1
|
|||
|
|
return text[start:end]
|
|||
|
|
|
|||
|
|
|
|||
|
|
def strip_fenced_code(reply: str) -> str:
|
|||
|
|
"""Drop Markdown fenced blocks, including an unclosed trailing fence.
|
|||
|
|
|
|||
|
|
Half-streamed code is code, not evidence that the chat changed purpose. Prose outside
|
|||
|
|
the fences keeps its line structure so sentence boundaries survive.
|
|||
|
|
"""
|
|||
|
|
|
|||
|
|
fence: str | None = None
|
|||
|
|
prose_lines: list[str] = []
|
|||
|
|
for line in reply.split("\n"):
|
|||
|
|
trimmed = line.lstrip(" \t")
|
|||
|
|
if trimmed.startswith("```"):
|
|||
|
|
marker: str | None = "`"
|
|||
|
|
elif trimmed.startswith("~~~"):
|
|||
|
|
marker = "~"
|
|||
|
|
else:
|
|||
|
|
marker = None
|
|||
|
|
if fence is not None:
|
|||
|
|
if marker == fence:
|
|||
|
|
fence = None
|
|||
|
|
continue
|
|||
|
|
if marker is not None:
|
|||
|
|
fence = marker
|
|||
|
|
continue
|
|||
|
|
prose_lines.append(line)
|
|||
|
|
return "\n".join(prose_lines)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def reply_prose(reply: str, character_limit: int = DEFAULT_CHARACTER_LIMIT) -> str | None:
|
|||
|
|
"""The exact text the runtime hands the purpose classifier, or ``None`` for no signal."""
|
|||
|
|
|
|||
|
|
if character_limit <= 0:
|
|||
|
|
return None
|
|||
|
|
prose = trim(strip_fenced_code(reply))
|
|||
|
|
if not prose:
|
|||
|
|
return None
|
|||
|
|
|
|||
|
|
clusters = graphemes(prose)
|
|||
|
|
if len(clusters) <= character_limit:
|
|||
|
|
return prose
|
|||
|
|
|
|||
|
|
tail = clusters[-character_limit:]
|
|||
|
|
# Prefer a whole-word start when the bounded suffix cut through one. If there is no
|
|||
|
|
# whitespace at all, retain the hard suffix rather than returning an empty signal.
|
|||
|
|
boundary = next(
|
|||
|
|
(offset for offset, cluster in enumerate(tail) if cluster[0] in WHITESPACE), None
|
|||
|
|
)
|
|||
|
|
if boundary is not None and boundary + 1 < len(tail):
|
|||
|
|
tail = tail[boundary + 1 :]
|
|||
|
|
bounded = trim("".join(tail))
|
|||
|
|
return bounded or None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def ends_in_question(prose: str | None) -> bool:
|
|||
|
|
"""The runtime's ``replyEndsInQuestion`` signal — a question is not committed drift."""
|
|||
|
|
|
|||
|
|
return prose is not None and prose.endswith("?")
|