Chunki z NUL wywalaly insert do postgresa (asyncpg CharacterNotInRepertoireError: invalid byte sequence for encoding "UTF8": 0x00) - 3 przypadki na mailach z 2007 w plastrze offset 50000 Etapu B. sanitize_surrogates tego nie lapie, bo NUL to poprawny code point, nie osierocony surogat. Strip dzieje sie zaraz po strip_quotes, czyli PRZED chunk_text i przed wywolaniem Ollamy - dzieki temu embedding liczy sie z dokladnie tego samego stringa, ktory trafia do document_chunk.text. Sanityzacja dopiero przy insercie zostawialaby wektor opisujacy tekst, ktorego DB nigdy nie zobaczyla. Skala widoczna w progress/summary: nul_bytes_stripped (ile znakow) oraz mails_nul_sanitized (ilu maili dotyczylo). Zadne z nich nie wchodzi do rownan balansu i nie wplywa na exit code - to normalizacja, nie blad. Ten sam strip w extract_threading (In-Reply-To / References): json.dumps zamienia NUL na escape u0000, ktory jsonb odrzuca tym samym bledem. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
57 lines
1.8 KiB
Python
57 lines
1.8 KiB
Python
"""Unit tests for kb_mail.text.sanitize_surrogates / strip_nul."""
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
|
||
from kb_mail.text import sanitize_surrogates, strip_nul
|
||
|
||
|
||
def test_none_stays_none():
|
||
assert sanitize_surrogates(None) is None
|
||
|
||
|
||
def test_plain_ascii_unchanged():
|
||
assert sanitize_surrogates("hello") == "hello"
|
||
|
||
|
||
def test_real_unicode_passes_through():
|
||
assert sanitize_surrogates("Kąpała") == "Kąpała"
|
||
|
||
|
||
def test_lone_surrogate_degraded_and_json_safe():
|
||
# A compat32 bytes-parse can leave a lone surrogate (undecodable byte);
|
||
# postgres jsonb and json.dumps().encode('utf-8') both reject it.
|
||
dirty = "Pr\udce9sent" # 0xe9 smuggled in as a surrogate escape
|
||
cleaned = sanitize_surrogates(dirty)
|
||
assert cleaned == "Pr?sent"
|
||
json.dumps(cleaned, ensure_ascii=False).encode("utf-8") # must not raise
|
||
|
||
|
||
def test_replacement_char_preserved():
|
||
# U+FFFD is already valid UTF-8 and must survive untouched.
|
||
assert sanitize_surrogates("bad<EFBFBD>id") == "bad<EFBFBD>id"
|
||
|
||
|
||
def test_sanitize_surrogates_does_not_remove_nul():
|
||
# Why strip_nul has to exist: NUL is a valid code point, so it survives the
|
||
# encode/decode round-trip untouched -- and then postgres rejects the insert.
|
||
assert sanitize_surrogates("a\x00b") == "a\x00b"
|
||
|
||
|
||
def test_strip_nul_none_stays_none():
|
||
assert strip_nul(None) is None
|
||
|
||
|
||
def test_strip_nul_removes_every_occurrence():
|
||
assert strip_nul("a\x00b\x00\x00c") == "abc"
|
||
|
||
|
||
def test_strip_nul_leaves_ordinary_text_untouched():
|
||
text = "Zwykły tekst\nz nową linią\ti tabulatorem\r\n"
|
||
assert strip_nul(text) is text # same object: no NUL, no rewrite
|
||
|
||
|
||
def test_strip_nul_keeps_other_c0_controls():
|
||
# Postgres accepts these in `text`; silently rewriting mail bodies is not the job here.
|
||
assert strip_nul("a\x01b\x0cc\x1fd") == "a\x01b\x0cc\x1fd"
|