Merge nucleic/jolly-coral-egret-smoz into dev
This commit is contained in:
@@ -0,0 +1,178 @@
|
||||
"""Runtime-identical assistant reply-prose extraction.
|
||||
|
||||
The Context Switch turn-end watcher classifies agent replies after stripping fenced code
|
||||
and tail-biasing the remainder (`docs/CONTEXT_SWITCH.md` §3.2). Training the classifier on
|
||||
raw replies would therefore train it on text the runtime never sees, so the assistant-prose
|
||||
slice runs its student text through this module, which is a deliberate line-by-line port of
|
||||
``HeuristicSummary.contextSwitchReplyProse`` in ``Sources/NucleicCore/Intelligence.swift``.
|
||||
|
||||
The two implementations are pinned together by
|
||||
``Tests/NucleicCoreTests/Fixtures/context-switch-prose.json``, which
|
||||
``tests/test_prose_extract.py`` and ``ContextSwitchTests`` both assert against. Change one
|
||||
side and the other side's test fails.
|
||||
|
||||
Two Swift behaviours need explicit modelling in Python:
|
||||
|
||||
* ``String.count``/``suffix`` count **grapheme clusters**, not code points, so the
|
||||
character limit is applied over :func:`graphemes` rather than ``len``. That segmentation
|
||||
is a pragmatic subset of UAX #29 (CRLF, combining marks, variation selectors, emoji
|
||||
ZWJ sequences, skin-tone modifiers, regional-indicator pairs) — the cases that actually
|
||||
occur in agent prose. Anything it does not model degrades to one cluster per scalar,
|
||||
which is what Python would have done anyway.
|
||||
* ``CharacterSet.whitespacesAndNewlines`` is the Unicode ``White_Space`` property, whereas
|
||||
``str.strip()`` also strips U+001C–U+001F. :data:`WHITESPACE` spells the Swift set out so
|
||||
a stray information separator in a reply cannot make the two extractions disagree.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import unicodedata
|
||||
|
||||
|
||||
DEFAULT_CHARACTER_LIMIT = 1_200
|
||||
|
||||
#: Unicode ``White_Space``, matching Swift's ``CharacterSet.whitespacesAndNewlines`` and
|
||||
#: ``Character.isWhitespace``. Deliberately excludes U+001C–U+001F, which ``str.strip()``
|
||||
#: would otherwise remove.
|
||||
WHITESPACE = frozenset(
|
||||
[chr(code) for code in range(0x0009, 0x000E)] # tab, LF, VT, FF, CR
|
||||
+ [chr(code) for code in range(0x2000, 0x200B)] # en quad through hair space
|
||||
+ [
|
||||
" ", # space
|
||||
"
", # next line
|
||||
" ", # no-break space
|
||||
" ", # ogham space mark
|
||||
"
", # line separator
|
||||
"
", # paragraph separator
|
||||
" ", # narrow no-break space
|
||||
" ", # medium mathematical space
|
||||
" ", # ideographic space
|
||||
]
|
||||
)
|
||||
|
||||
_ZERO_WIDTH_JOINER = ""
|
||||
_EXTEND_CATEGORIES = frozenset({"Mn", "Me", "Mc"})
|
||||
|
||||
|
||||
def _is_extend(character: str) -> bool:
|
||||
"""Scalars that attach to the preceding grapheme cluster."""
|
||||
|
||||
if character in ("︎", "️"): # variation selectors 15/16
|
||||
return True
|
||||
if "\U000e0100" <= character <= "\U000e01ef": # variation selectors supplement
|
||||
return True
|
||||
if "\U0001f3fb" <= character <= "\U0001f3ff": # emoji skin-tone modifiers
|
||||
return True
|
||||
return unicodedata.category(character) in _EXTEND_CATEGORIES
|
||||
|
||||
|
||||
def _is_regional_indicator(character: str) -> bool:
|
||||
return "\U0001f1e6" <= character <= "\U0001f1ff"
|
||||
|
||||
|
||||
def graphemes(text: str) -> list[str]:
|
||||
"""Segment ``text`` the way Swift's ``Character`` view does, for the cases we see."""
|
||||
|
||||
clusters: list[str] = []
|
||||
index, length = 0, len(text)
|
||||
while index < length:
|
||||
base = text[index]
|
||||
index += 1
|
||||
if base == "\r" and index < length and text[index] == "\n":
|
||||
clusters.append("\r\n")
|
||||
index += 1
|
||||
continue
|
||||
if unicodedata.category(base) == "Cc":
|
||||
# Controls never combine with what follows; CR LF above is the one exception.
|
||||
clusters.append(base)
|
||||
continue
|
||||
cluster = base
|
||||
if (
|
||||
_is_regional_indicator(base)
|
||||
and index < length
|
||||
and _is_regional_indicator(text[index])
|
||||
):
|
||||
cluster += text[index]
|
||||
index += 1
|
||||
while index < length:
|
||||
following = text[index]
|
||||
if _is_extend(following):
|
||||
cluster += following
|
||||
index += 1
|
||||
elif following == _ZERO_WIDTH_JOINER and index + 1 < length:
|
||||
cluster += following + text[index + 1]
|
||||
index += 2
|
||||
else:
|
||||
break
|
||||
clusters.append(cluster)
|
||||
return clusters
|
||||
|
||||
|
||||
def trim(text: str) -> str:
|
||||
"""``trimmingCharacters(in: .whitespacesAndNewlines)``."""
|
||||
|
||||
start, end = 0, len(text)
|
||||
while start < end and text[start] in WHITESPACE:
|
||||
start += 1
|
||||
while end > start and text[end - 1] in WHITESPACE:
|
||||
end -= 1
|
||||
return text[start:end]
|
||||
|
||||
|
||||
def strip_fenced_code(reply: str) -> str:
|
||||
"""Drop Markdown fenced blocks, including an unclosed trailing fence.
|
||||
|
||||
Half-streamed code is code, not evidence that the chat changed purpose. Prose outside
|
||||
the fences keeps its line structure so sentence boundaries survive.
|
||||
"""
|
||||
|
||||
fence: str | None = None
|
||||
prose_lines: list[str] = []
|
||||
for line in reply.split("\n"):
|
||||
trimmed = line.lstrip(" \t")
|
||||
if trimmed.startswith("```"):
|
||||
marker: str | None = "`"
|
||||
elif trimmed.startswith("~~~"):
|
||||
marker = "~"
|
||||
else:
|
||||
marker = None
|
||||
if fence is not None:
|
||||
if marker == fence:
|
||||
fence = None
|
||||
continue
|
||||
if marker is not None:
|
||||
fence = marker
|
||||
continue
|
||||
prose_lines.append(line)
|
||||
return "\n".join(prose_lines)
|
||||
|
||||
|
||||
def reply_prose(reply: str, character_limit: int = DEFAULT_CHARACTER_LIMIT) -> str | None:
|
||||
"""The exact text the runtime hands the purpose classifier, or ``None`` for no signal."""
|
||||
|
||||
if character_limit <= 0:
|
||||
return None
|
||||
prose = trim(strip_fenced_code(reply))
|
||||
if not prose:
|
||||
return None
|
||||
|
||||
clusters = graphemes(prose)
|
||||
if len(clusters) <= character_limit:
|
||||
return prose
|
||||
|
||||
tail = clusters[-character_limit:]
|
||||
# Prefer a whole-word start when the bounded suffix cut through one. If there is no
|
||||
# whitespace at all, retain the hard suffix rather than returning an empty signal.
|
||||
boundary = next(
|
||||
(offset for offset, cluster in enumerate(tail) if cluster[0] in WHITESPACE), None
|
||||
)
|
||||
if boundary is not None and boundary + 1 < len(tail):
|
||||
tail = tail[boundary + 1 :]
|
||||
bounded = trim("".join(tail))
|
||||
return bounded or None
|
||||
|
||||
|
||||
def ends_in_question(prose: str | None) -> bool:
|
||||
"""The runtime's ``replyEndsInQuestion`` signal — a question is not committed drift."""
|
||||
|
||||
return prose is not None and prose.endswith("?")
|
||||
Reference in New Issue
Block a user