Files
nucleic-purpose-classifier/tests/test_prose_extract.py
T

111 lines
4.1 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import json
import sys
import unittest
from pathlib import Path
MODULE_DIR = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(MODULE_DIR))
import prose_extract
FIXTURE = (
MODULE_DIR.parents[1] / "Tests/NucleicCoreTests/Fixtures/context-switch-prose.json"
)
class ProseParityTests(unittest.TestCase):
"""The shared fixture is the contract between this port and the Swift runtime.
``ContextSwitchTests.replyProseMatchesTheSharedParityFixture`` asserts the same file
against ``HeuristicSummary.contextSwitchReplyProse``. A change to either extraction
that is not mirrored in the other fails one of the two suites.
"""
def setUp(self):
self.cases = json.loads(FIXTURE.read_text(encoding="utf-8"))
self.assertTrue(self.cases, "shared prose parity fixture is empty")
def test_matches_the_shared_runtime_fixture(self):
for case in self.cases:
with self.subTest(case["name"]):
prose = prose_extract.reply_prose(case["reply"], case["characterLimit"])
self.assertEqual(prose, case["expectedProse"])
self.assertEqual(
prose_extract.ends_in_question(prose), case["expectedEndsInQuestion"]
)
def test_extracted_prose_never_exceeds_the_limit_in_graphemes(self):
for case in self.cases:
prose = prose_extract.reply_prose(case["reply"], case["characterLimit"])
if prose is None:
continue
with self.subTest(case["name"]):
self.assertLessEqual(
len(prose_extract.graphemes(prose)), case["characterLimit"]
)
class GraphemeSegmentationTests(unittest.TestCase):
def test_models_the_clusters_swift_treats_as_one_character(self):
for text, expected in [
("á", 1), # combining acute
("\U0001f468‍\U0001f469‍\U0001f467", 1), # family ZWJ sequence
("\U0001f44d\U0001f3fd", 1), # thumbs up + skin tone
("\U0001f1fa\U0001f1f8", 1), # regional indicator pair
("\U0001f1fa\U0001f1f8\U0001f1e9\U0001f1ea", 2), # two flags, not one run
("\r\n", 1),
("\n\n", 2),
("2️⃣", 1), # keycap
("plain", 5),
]:
with self.subTest(repr(text)):
self.assertEqual(len(prose_extract.graphemes(text)), expected)
def test_reassembling_clusters_is_lossless(self):
for text in ["", "ábc", "\U0001f468‍\U0001f469 x", "a\r\nb"]:
with self.subTest(repr(text)):
self.assertEqual("".join(prose_extract.graphemes(text)), text)
class TrimTests(unittest.TestCase):
def test_uses_the_swift_whitespace_set_rather_than_str_strip(self):
# str.strip() would remove the information separators; Swift's
# .whitespacesAndNewlines does not, so neither does the port.
self.assertEqual(prose_extract.trim("Prose."), "Prose.")
self.assertEqual(prose_extract.trim("  Prose. "), "Prose.")
self.assertEqual(prose_extract.trim(" \t\r\n "), "")
def test_whitespace_set_is_exactly_unicode_white_space(self):
expected = {
*range(0x0009, 0x000E),
0x0020,
0x0085,
0x00A0,
0x1680,
*range(0x2000, 0x200B),
0x2028,
0x2029,
0x202F,
0x205F,
0x3000,
}
self.assertEqual({ord(c) for c in prose_extract.WHITESPACE}, expected)
class FenceStrippingTests(unittest.TestCase):
def test_drops_fenced_blocks_and_keeps_outside_line_structure(self):
self.assertEqual(
prose_extract.strip_fenced_code("a\n```\ncode\n```\nb"), "a\nb"
)
# An unclosed fence takes the remainder: half-streamed code is still code.
self.assertEqual(prose_extract.strip_fenced_code("a\n```\ncode"), "a")
# A backtick fence does not close a tilde fence.
self.assertEqual(
prose_extract.strip_fenced_code("a\n~~~\n```\nx\n```\n~~~\nb"), "a\nb"
)
if __name__ == "__main__":
unittest.main()