"""Runtime-identical assistant reply-prose extraction. The Context Switch turn-end watcher classifies agent replies after stripping fenced code and tail-biasing the remainder (`docs/CONTEXT_SWITCH.md` §3.2). Training the classifier on raw replies would therefore train it on text the runtime never sees, so the assistant-prose slice runs its student text through this module, which is a deliberate line-by-line port of ``HeuristicSummary.contextSwitchReplyProse`` in ``Sources/NucleicCore/Intelligence.swift``. The two implementations are pinned together by ``Tests/NucleicCoreTests/Fixtures/context-switch-prose.json``, which ``tests/test_prose_extract.py`` and ``ContextSwitchTests`` both assert against. Change one side and the other side's test fails. Two Swift behaviours need explicit modelling in Python: * ``String.count``/``suffix`` count **grapheme clusters**, not code points, so the character limit is applied over :func:`graphemes` rather than ``len``. That segmentation is a pragmatic subset of UAX #29 (CRLF, combining marks, variation selectors, emoji ZWJ sequences, skin-tone modifiers, regional-indicator pairs) — the cases that actually occur in agent prose. Anything it does not model degrades to one cluster per scalar, which is what Python would have done anyway. * ``CharacterSet.whitespacesAndNewlines`` is the Unicode ``White_Space`` property, whereas ``str.strip()`` also strips U+001C–U+001F. :data:`WHITESPACE` spells the Swift set out so a stray information separator in a reply cannot make the two extractions disagree. """ from __future__ import annotations import unicodedata DEFAULT_CHARACTER_LIMIT = 1_200 #: Unicode ``White_Space``, matching Swift's ``CharacterSet.whitespacesAndNewlines`` and #: ``Character.isWhitespace``. Deliberately excludes U+001C–U+001F, which ``str.strip()`` #: would otherwise remove. WHITESPACE = frozenset( [chr(code) for code in range(0x0009, 0x000E)] # tab, LF, VT, FF, CR + [chr(code) for code in range(0x2000, 0x200B)] # en quad through hair space + [ " ", # space "…", # next line " ", # no-break space " ", # ogham space mark "
", # line separator "
", # paragraph separator " ", # narrow no-break space " ", # medium mathematical space " ", # ideographic space ] ) _ZERO_WIDTH_JOINER = "‍" _EXTEND_CATEGORIES = frozenset({"Mn", "Me", "Mc"}) def _is_extend(character: str) -> bool: """Scalars that attach to the preceding grapheme cluster.""" if character in ("︎", "️"): # variation selectors 15/16 return True if "\U000e0100" <= character <= "\U000e01ef": # variation selectors supplement return True if "\U0001f3fb" <= character <= "\U0001f3ff": # emoji skin-tone modifiers return True return unicodedata.category(character) in _EXTEND_CATEGORIES def _is_regional_indicator(character: str) -> bool: return "\U0001f1e6" <= character <= "\U0001f1ff" def graphemes(text: str) -> list[str]: """Segment ``text`` the way Swift's ``Character`` view does, for the cases we see.""" clusters: list[str] = [] index, length = 0, len(text) while index < length: base = text[index] index += 1 if base == "\r" and index < length and text[index] == "\n": clusters.append("\r\n") index += 1 continue if unicodedata.category(base) == "Cc": # Controls never combine with what follows; CR LF above is the one exception. clusters.append(base) continue cluster = base if ( _is_regional_indicator(base) and index < length and _is_regional_indicator(text[index]) ): cluster += text[index] index += 1 while index < length: following = text[index] if _is_extend(following): cluster += following index += 1 elif following == _ZERO_WIDTH_JOINER and index + 1 < length: cluster += following + text[index + 1] index += 2 else: break clusters.append(cluster) return clusters def trim(text: str) -> str: """``trimmingCharacters(in: .whitespacesAndNewlines)``.""" start, end = 0, len(text) while start < end and text[start] in WHITESPACE: start += 1 while end > start and text[end - 1] in WHITESPACE: end -= 1 return text[start:end] def strip_fenced_code(reply: str) -> str: """Drop Markdown fenced blocks, including an unclosed trailing fence. Half-streamed code is code, not evidence that the chat changed purpose. Prose outside the fences keeps its line structure so sentence boundaries survive. """ fence: str | None = None prose_lines: list[str] = [] for line in reply.split("\n"): trimmed = line.lstrip(" \t") if trimmed.startswith("```"): marker: str | None = "`" elif trimmed.startswith("~~~"): marker = "~" else: marker = None if fence is not None: if marker == fence: fence = None continue if marker is not None: fence = marker continue prose_lines.append(line) return "\n".join(prose_lines) def reply_prose(reply: str, character_limit: int = DEFAULT_CHARACTER_LIMIT) -> str | None: """The exact text the runtime hands the purpose classifier, or ``None`` for no signal.""" if character_limit <= 0: return None prose = trim(strip_fenced_code(reply)) if not prose: return None clusters = graphemes(prose) if len(clusters) <= character_limit: return prose tail = clusters[-character_limit:] # Prefer a whole-word start when the bounded suffix cut through one. If there is no # whitespace at all, retain the hard suffix rather than returning an empty signal. boundary = next( (offset for offset, cluster in enumerate(tail) if cluster[0] in WHITESPACE), None ) if boundary is not None and boundary + 1 < len(tail): tail = tail[boundary + 1 :] bounded = trim("".join(tail)) return bounded or None def ends_in_question(prose: str | None) -> bool: """The runtime's ``replyEndsInQuestion`` signal — a question is not committed drift.""" return prose is not None and prose.endswith("?")