Instructions to use SlayerLab/NERGAL with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use SlayerLab/NERGAL with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("token-classification", model="SlayerLab/NERGAL")# Load model directly from transformers import AutoTokenizer, AutoModelForTokenClassification tokenizer = AutoTokenizer.from_pretrained("SlayerLab/NERGAL") model = AutoModelForTokenClassification.from_pretrained("SlayerLab/NERGAL", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Add synthetic NERGAL unit tests
Browse files- test_nergal.py +43 -0
test_nergal.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Synthetic NERGAL tests. Invented strings only; no corpus text or real identifiers."""
|
| 2 |
+
import hashlib
|
| 3 |
+
import json
|
| 4 |
+
import unittest
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
|
| 7 |
+
HERE = Path(__file__).resolve().parent
|
| 8 |
+
RULES_SHA = '547c0428b0799bf051566d6ac489987eff27f09e1bda36a5665452fe155b3966'
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
class NergalTests(unittest.TestCase):
|
| 12 |
+
def test_card_and_rules_hash(self):
|
| 13 |
+
from nergal import GAP_IDS, GAPS, HUB_ID, RULES_SHA as PINNED, THRESHOLD
|
| 14 |
+
card = json.loads((HERE / 'hybrid.json').read_text())
|
| 15 |
+
self.assertEqual(HUB_ID, 'SlayerLab/NERGAL')
|
| 16 |
+
self.assertEqual(GAPS, card['gaps'])
|
| 17 |
+
self.assertEqual(GAP_IDS, card['gap_ids'])
|
| 18 |
+
self.assertEqual(THRESHOLD, card['threshold'])
|
| 19 |
+
self.assertEqual(PINNED, RULES_SHA)
|
| 20 |
+
digest = hashlib.sha256((HERE / 'scrub_pii.py').read_bytes()).hexdigest()
|
| 21 |
+
self.assertEqual(digest, RULES_SHA)
|
| 22 |
+
|
| 23 |
+
def test_union_keeps_regex_and_adds_model_spans(self):
|
| 24 |
+
from nergal import apply_union, scrub_spans
|
| 25 |
+
text = 'Ring 000000000 then extra.'
|
| 26 |
+
rules = [{'start': 5, 'end': 14, 'label': 'phone', 'score': 1.0}]
|
| 27 |
+
model = [
|
| 28 |
+
{'start': 5, 'end': 14, 'label': 'phone', 'score': 0.99},
|
| 29 |
+
{'start': 20, 'end': 25, 'label': 'pii', 'score': 0.97},
|
| 30 |
+
]
|
| 31 |
+
masked, counts = scrub_spans(text, rules, model, threshold=0.95)
|
| 32 |
+
self.assertIn('[Telefon]', masked)
|
| 33 |
+
self.assertIn('[PII]', masked)
|
| 34 |
+
self.assertGreater(counts['union_placeholder_chars'], counts['rules_placeholder_chars'])
|
| 35 |
+
self.assertEqual(counts['model_extra_spans'], 1)
|
| 36 |
+
_, rules_chars, _, _ = apply_union(text, rules)
|
| 37 |
+
self.assertEqual(counts['rules_placeholder_chars'], rules_chars)
|
| 38 |
+
self.assertNotIn('000000000', masked)
|
| 39 |
+
self.assertNotIn('extra', masked)
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
if __name__ == '__main__':
|
| 43 |
+
unittest.main()
|