NERGAL / test_nergal.py
ppuzio's picture
Claude Opus 5.5
Prepare 1.1.0: batch API (predict_many, scrub_many), opt-in float16
fdc79fc
Raw
History Blame
3.99 kB
"""Synthetic NERGAL tests. Invented strings only; no corpus text or real identifiers."""
import hashlib
import json
import unittest
from pathlib import Path
HERE = Path(__file__).resolve().parent
RULES_SHA = 'f32d5c5452fc47178e109d4bc248a0d8234ea6e59e8cf79407f4eb8451581d67'
class NergalTests(unittest.TestCase):
def test_card_and_rules_hash(self):
from nergal import GAP_IDS, GAPS, HUB_ID, RULES_SHA as PINNED, THRESHOLD, VERSION
card = json.loads((HERE / 'hybrid.json').read_text())
self.assertEqual(HUB_ID, 'SlayerLab/NERGAL')
self.assertEqual(VERSION, '1.1.0')
self.assertEqual(card['version'], VERSION)
self.assertEqual(card['eval']['union_fp'], 123)
self.assertEqual(card['eval']['rules_fp'], 98)
self.assertEqual(GAPS, card['gaps'])
self.assertEqual(GAP_IDS, card['gap_ids'])
self.assertEqual(THRESHOLD, card['threshold'])
self.assertEqual(PINNED, RULES_SHA)
digest = hashlib.sha256((HERE / 'scrub_pii.py').read_bytes()).hexdigest()
self.assertEqual(digest, RULES_SHA)
def test_real_tokenizer_preserves_batch_and_unit_alignment(self):
from transformers import AutoTokenizer
from nergal import Encoding
tokenizer = AutoTokenizer.from_pretrained(str(HERE), local_files_only=True, fix_mistral_regex=False)
encoding = Encoding(tokenizer)
words = ['A', '[PII_SPACE]', '1']
encoded, first = encoding.encode(words)
self.assertIsInstance(encoded['input_ids'][0], list)
self.assertEqual(len(first), len(words))
self.assertEqual([encoded.word_ids(0)[i] for i in first], [0, 1, 2])
def test_window_token_count_matches_the_encoded_window(self):
from transformers import AutoTokenizer
from nergal import Encoding
tokenizer = AutoTokenizer.from_pretrained(str(HERE), local_files_only=True, fix_mistral_regex=False)
encoding = Encoding(tokenizer)
text = ' '.join(f'Zdanie {i}: tel. 22 123 45 67,\nNIP 1234567802.' for i in range(120))
units, chunks = encoding.prepare(text)
self.assertGreater(len(chunks), 1)
for w in chunks:
encoded, _ = encoding.encode([u.model for u in units[w['start']:w['end']]])
self.assertEqual(w['tokens'], len(encoded['input_ids'][0]))
self.assertLessEqual(w['tokens'], 512)
def test_float16_is_opt_in_and_needs_an_accelerator(self):
from nergal import Nergal
with self.assertRaises(ValueError):
Nergal(HERE, device='cpu', dtype='float16')
with self.assertRaises(ValueError):
Nergal(HERE, dtype='bfloat16')
def test_existing_placeholders_do_not_switch_the_rules_off(self):
from nergal import rules
text = 'Kontakt [Telefon], NIP 1234567802.' # invented, checksum-valid
[span] = rules(text)
self.assertEqual(text[span['start']:span['end']], '1234567802')
self.assertEqual(rules('a [PII] b [Telefon] c'), [])
def test_union_keeps_regex_and_adds_model_spans(self):
from nergal import apply_union, scrub_spans
text = 'Ring 000000000 then extra.'
rules = [{'start': 5, 'end': 14, 'label': 'phone', 'score': 1.0}]
model = [
{'start': 5, 'end': 14, 'label': 'phone', 'score': 0.99},
{'start': 20, 'end': 25, 'label': 'pii', 'score': 0.97},
]
masked, counts = scrub_spans(text, rules, model, threshold=0.95)
self.assertIn('[Telefon]', masked)
self.assertIn('[PII]', masked)
self.assertGreater(counts['union_placeholder_chars'], counts['rules_placeholder_chars'])
self.assertEqual(counts['model_extra_spans'], 1)
_, rules_chars, _, _ = apply_union(text, rules)
self.assertEqual(counts['rules_placeholder_chars'], rules_chars)
self.assertNotIn('000000000', masked)
self.assertNotIn('extra', masked)
if __name__ == '__main__':
unittest.main()