111 lines
4.1 KiB
Python
111 lines
4.1 KiB
Python
import json
|
||
import sys
|
||
import unittest
|
||
from pathlib import Path
|
||
|
||
|
||
MODULE_DIR = Path(__file__).resolve().parents[1]
|
||
sys.path.insert(0, str(MODULE_DIR))
|
||
|
||
import prose_extract
|
||
|
||
FIXTURE = (
|
||
MODULE_DIR.parents[1] / "Tests/NucleicCoreTests/Fixtures/context-switch-prose.json"
|
||
)
|
||
|
||
|
||
class ProseParityTests(unittest.TestCase):
|
||
"""The shared fixture is the contract between this port and the Swift runtime.
|
||
|
||
``ContextSwitchTests.replyProseMatchesTheSharedParityFixture`` asserts the same file
|
||
against ``HeuristicSummary.contextSwitchReplyProse``. A change to either extraction
|
||
that is not mirrored in the other fails one of the two suites.
|
||
"""
|
||
|
||
def setUp(self):
|
||
self.cases = json.loads(FIXTURE.read_text(encoding="utf-8"))
|
||
self.assertTrue(self.cases, "shared prose parity fixture is empty")
|
||
|
||
def test_matches_the_shared_runtime_fixture(self):
|
||
for case in self.cases:
|
||
with self.subTest(case["name"]):
|
||
prose = prose_extract.reply_prose(case["reply"], case["characterLimit"])
|
||
self.assertEqual(prose, case["expectedProse"])
|
||
self.assertEqual(
|
||
prose_extract.ends_in_question(prose), case["expectedEndsInQuestion"]
|
||
)
|
||
|
||
def test_extracted_prose_never_exceeds_the_limit_in_graphemes(self):
|
||
for case in self.cases:
|
||
prose = prose_extract.reply_prose(case["reply"], case["characterLimit"])
|
||
if prose is None:
|
||
continue
|
||
with self.subTest(case["name"]):
|
||
self.assertLessEqual(
|
||
len(prose_extract.graphemes(prose)), case["characterLimit"]
|
||
)
|
||
|
||
|
||
class GraphemeSegmentationTests(unittest.TestCase):
|
||
def test_models_the_clusters_swift_treats_as_one_character(self):
|
||
for text, expected in [
|
||
("á", 1), # combining acute
|
||
("\U0001f468\U0001f469\U0001f467", 1), # family ZWJ sequence
|
||
("\U0001f44d\U0001f3fd", 1), # thumbs up + skin tone
|
||
("\U0001f1fa\U0001f1f8", 1), # regional indicator pair
|
||
("\U0001f1fa\U0001f1f8\U0001f1e9\U0001f1ea", 2), # two flags, not one run
|
||
("\r\n", 1),
|
||
("\n\n", 2),
|
||
("2️⃣", 1), # keycap
|
||
("plain", 5),
|
||
]:
|
||
with self.subTest(repr(text)):
|
||
self.assertEqual(len(prose_extract.graphemes(text)), expected)
|
||
|
||
def test_reassembling_clusters_is_lossless(self):
|
||
for text in ["", "ábc", "\U0001f468\U0001f469 x", "a\r\nb"]:
|
||
with self.subTest(repr(text)):
|
||
self.assertEqual("".join(prose_extract.graphemes(text)), text)
|
||
|
||
|
||
class TrimTests(unittest.TestCase):
|
||
def test_uses_the_swift_whitespace_set_rather_than_str_strip(self):
|
||
# str.strip() would remove the information separators; Swift's
|
||
# .whitespacesAndNewlines does not, so neither does the port.
|
||
self.assertEqual(prose_extract.trim("Prose."), "Prose.")
|
||
self.assertEqual(prose_extract.trim(" Prose. "), "Prose.")
|
||
self.assertEqual(prose_extract.trim(" \t\r\n "), "")
|
||
|
||
def test_whitespace_set_is_exactly_unicode_white_space(self):
|
||
expected = {
|
||
*range(0x0009, 0x000E),
|
||
0x0020,
|
||
0x0085,
|
||
0x00A0,
|
||
0x1680,
|
||
*range(0x2000, 0x200B),
|
||
0x2028,
|
||
0x2029,
|
||
0x202F,
|
||
0x205F,
|
||
0x3000,
|
||
}
|
||
self.assertEqual({ord(c) for c in prose_extract.WHITESPACE}, expected)
|
||
|
||
|
||
class FenceStrippingTests(unittest.TestCase):
|
||
def test_drops_fenced_blocks_and_keeps_outside_line_structure(self):
|
||
self.assertEqual(
|
||
prose_extract.strip_fenced_code("a\n```\ncode\n```\nb"), "a\nb"
|
||
)
|
||
# An unclosed fence takes the remainder: half-streamed code is still code.
|
||
self.assertEqual(prose_extract.strip_fenced_code("a\n```\ncode"), "a")
|
||
# A backtick fence does not close a tilde fence.
|
||
self.assertEqual(
|
||
prose_extract.strip_fenced_code("a\n~~~\n```\nx\n```\n~~~\nb"), "a\nb"
|
||
)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|