"""Unit tests for kb_mail.text.sanitize_surrogates / strip_nul.""" from __future__ import annotations import json from kb_mail.text import sanitize_surrogates, strip_nul def test_none_stays_none(): assert sanitize_surrogates(None) is None def test_plain_ascii_unchanged(): assert sanitize_surrogates("hello") == "hello" def test_real_unicode_passes_through(): assert sanitize_surrogates("Kąpała") == "Kąpała" def test_lone_surrogate_degraded_and_json_safe(): # A compat32 bytes-parse can leave a lone surrogate (undecodable byte); # postgres jsonb and json.dumps().encode('utf-8') both reject it. dirty = "Pr\udce9sent" # 0xe9 smuggled in as a surrogate escape cleaned = sanitize_surrogates(dirty) assert cleaned == "Pr?sent" json.dumps(cleaned, ensure_ascii=False).encode("utf-8") # must not raise def test_replacement_char_preserved(): # U+FFFD is already valid UTF-8 and must survive untouched. assert sanitize_surrogates("bad�id") == "bad�id" def test_sanitize_surrogates_does_not_remove_nul(): # Why strip_nul has to exist: NUL is a valid code point, so it survives the # encode/decode round-trip untouched -- and then postgres rejects the insert. assert sanitize_surrogates("a\x00b") == "a\x00b" def test_strip_nul_none_stays_none(): assert strip_nul(None) is None def test_strip_nul_removes_every_occurrence(): assert strip_nul("a\x00b\x00\x00c") == "abc" def test_strip_nul_leaves_ordinary_text_untouched(): text = "Zwykły tekst\nz nową linią\ti tabulatorem\r\n" assert strip_nul(text) is text # same object: no NUL, no rewrite def test_strip_nul_keeps_other_c0_controls(): # Postgres accepts these in `text`; silently rewriting mail bodies is not the job here. assert strip_nul("a\x01b\x0cc\x1fd") == "a\x01b\x0cc\x1fd"