Merge nucleic/jolly-coral-egret-smoz into dev

This commit is contained in:
2026-08-04 16:15:55 -07:00
parent 1a41febf73
commit 931e180f1c
8 changed files with 1653 additions and 12 deletions
+178
View File
@@ -0,0 +1,178 @@
"""Runtime-identical assistant reply-prose extraction.
The Context Switch turn-end watcher classifies agent replies after stripping fenced code
and tail-biasing the remainder (`docs/CONTEXT_SWITCH.md` §3.2). Training the classifier on
raw replies would therefore train it on text the runtime never sees, so the assistant-prose
slice runs its student text through this module, which is a deliberate line-by-line port of
``HeuristicSummary.contextSwitchReplyProse`` in ``Sources/NucleicCore/Intelligence.swift``.
The two implementations are pinned together by
``Tests/NucleicCoreTests/Fixtures/context-switch-prose.json``, which
``tests/test_prose_extract.py`` and ``ContextSwitchTests`` both assert against. Change one
side and the other side's test fails.
Two Swift behaviours need explicit modelling in Python:
* ``String.count``/``suffix`` count **grapheme clusters**, not code points, so the
character limit is applied over :func:`graphemes` rather than ``len``. That segmentation
is a pragmatic subset of UAX #29 (CRLF, combining marks, variation selectors, emoji
ZWJ sequences, skin-tone modifiers, regional-indicator pairs) — the cases that actually
occur in agent prose. Anything it does not model degrades to one cluster per scalar,
which is what Python would have done anyway.
* ``CharacterSet.whitespacesAndNewlines`` is the Unicode ``White_Space`` property, whereas
``str.strip()`` also strips U+001C–U+001F. :data:`WHITESPACE` spells the Swift set out so
a stray information separator in a reply cannot make the two extractions disagree.
"""
from __future__ import annotations
import unicodedata
DEFAULT_CHARACTER_LIMIT = 1_200
#: Unicode ``White_Space``, matching Swift's ``CharacterSet.whitespacesAndNewlines`` and
#: ``Character.isWhitespace``. Deliberately excludes U+001C–U+001F, which ``str.strip()``
#: would otherwise remove.
WHITESPACE = frozenset(
[chr(code) for code in range(0x0009, 0x000E)] # tab, LF, VT, FF, CR
+ [chr(code) for code in range(0x2000, 0x200B)] # en quad through hair space
+ [
" ", # space
"…", # next line
" ", # no-break space
" ", # ogham space mark
"
", # line separator
"
", # paragraph separator
" ", # narrow no-break space
" ", # medium mathematical space
" ", # ideographic space
]
)
_ZERO_WIDTH_JOINER = "‍"
_EXTEND_CATEGORIES = frozenset({"Mn", "Me", "Mc"})
def _is_extend(character: str) -> bool:
"""Scalars that attach to the preceding grapheme cluster."""
if character in ("︎", "️"): # variation selectors 15/16
return True
if "\U000e0100" <= character <= "\U000e01ef": # variation selectors supplement
return True
if "\U0001f3fb" <= character <= "\U0001f3ff": # emoji skin-tone modifiers
return True
return unicodedata.category(character) in _EXTEND_CATEGORIES
def _is_regional_indicator(character: str) -> bool:
return "\U0001f1e6" <= character <= "\U0001f1ff"
def graphemes(text: str) -> list[str]:
"""Segment ``text`` the way Swift's ``Character`` view does, for the cases we see."""
clusters: list[str] = []
index, length = 0, len(text)
while index < length:
base = text[index]
index += 1
if base == "\r" and index < length and text[index] == "\n":
clusters.append("\r\n")
index += 1
continue
if unicodedata.category(base) == "Cc":
# Controls never combine with what follows; CR LF above is the one exception.
clusters.append(base)
continue
cluster = base
if (
_is_regional_indicator(base)
and index < length
and _is_regional_indicator(text[index])
):
cluster += text[index]
index += 1
while index < length:
following = text[index]
if _is_extend(following):
cluster += following
index += 1
elif following == _ZERO_WIDTH_JOINER and index + 1 < length:
cluster += following + text[index + 1]
index += 2
else:
break
clusters.append(cluster)
return clusters
def trim(text: str) -> str:
"""``trimmingCharacters(in: .whitespacesAndNewlines)``."""
start, end = 0, len(text)
while start < end and text[start] in WHITESPACE:
start += 1
while end > start and text[end - 1] in WHITESPACE:
end -= 1
return text[start:end]
def strip_fenced_code(reply: str) -> str:
"""Drop Markdown fenced blocks, including an unclosed trailing fence.
Half-streamed code is code, not evidence that the chat changed purpose. Prose outside
the fences keeps its line structure so sentence boundaries survive.
"""
fence: str | None = None
prose_lines: list[str] = []
for line in reply.split("\n"):
trimmed = line.lstrip(" \t")
if trimmed.startswith("```"):
marker: str | None = "`"
elif trimmed.startswith("~~~"):
marker = "~"
else:
marker = None
if fence is not None:
if marker == fence:
fence = None
continue
if marker is not None:
fence = marker
continue
prose_lines.append(line)
return "\n".join(prose_lines)
def reply_prose(reply: str, character_limit: int = DEFAULT_CHARACTER_LIMIT) -> str | None:
"""The exact text the runtime hands the purpose classifier, or ``None`` for no signal."""
if character_limit <= 0:
return None
prose = trim(strip_fenced_code(reply))
if not prose:
return None
clusters = graphemes(prose)
if len(clusters) <= character_limit:
return prose
tail = clusters[-character_limit:]
# Prefer a whole-word start when the bounded suffix cut through one. If there is no
# whitespace at all, retain the hard suffix rather than returning an empty signal.
boundary = next(
(offset for offset, cluster in enumerate(tail) if cluster[0] in WHITESPACE), None
)
if boundary is not None and boundary + 1 < len(tail):
tail = tail[boundary + 1 :]
bounded = trim("".join(tail))
return bounded or None
def ends_in_question(prose: str | None) -> bool:
"""The runtime's ``replyEndsInQuestion`` signal — a question is not committed drift."""
return prose is not None and prose.endswith("?")