Text Classification
Transformers
ONNX
Safetensors
English
Hindi
multilingual
query-classification
intent-detection
memory-scope
modernbert
quantized
Instructions to use addyo07/query-scope-classifier with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use addyo07/query-scope-classifier with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="addyo07/query-scope-classifier")# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("addyo07/query-scope-classifier", device_map="auto") - Notebooks
- Google Colab
- Kaggle
File size: 6,608 Bytes
6784fa4 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 | #!/usr/bin/env python3
"""
Phase 1.2: Base Dataset Fast-Path ChitChat Mapping & Hinglish ChitChat Generation
"""
import json
import os
import random
import requests
BASE_DIR = "/opt/vox/query-classification-dataset"
OUTPUT_DIR = "/opt/vox/sandbox/datasets"
OLLAMA_URL = "http://localhost:11434/api/generate"
LMS_URL = "http://localhost:1234/v1/chat/completions"
os.makedirs(OUTPUT_DIR, exist_ok=True)
def load_base_file(filename, scope, lang, source):
filepath = os.path.join(BASE_DIR, filename)
samples = []
with open(filepath, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line:
continue
data = json.loads(line)
text = data.get("text", "").strip()
if text:
samples.append({
"text": text,
"scope": scope,
"language": lang,
"source": source
})
return samples
def generate_hinglish_chitchat_llm(target_count=1000):
print(f"Generating {target_count} Hinglish ChitChat/Banter queries via local LLM...")
hinglish_samples = set()
# Template seeds for diverse generation
prompt = """Generate 50 diverse, casual Hinglish (code-switched Hindi-English) chitchat, filler, greetings, and banter queries spoken to a voice AI assistant.
Examples:
- Hey Vox, kya haal hai?
- Suno yaar, good morning!
- Kaun ho tum, apne baare me batao?
- Aaj ka mausam kaisa hai bhai?
- Kya chal raha hai aajkal?
- Bhai ek joke sunao na please.
- Vox, tu mast hai yaar.
- Good night, kal milte hain.
- Bol na kuch, bore ho raha hoon.
- Kya tum mere dost banoge?
Rules:
1. Every query must be casual chitchat, greeting, banter, or filler (0 technical/code questions, 0 personal data retrieval).
2. Code-switched Latin Hinglish script only (e.g. 'kya haal hai', 'kya chal raha hai', 'good morning bhai').
3. Output ONLY a raw JSON array of strings: ["query 1", "query 2", ...]
"""
attempts = 0
while len(hinglish_samples) < target_count and attempts < 30:
attempts += 1
print(f" [Attempt {attempts}] Current Hinglish ChitChat count: {len(hinglish_samples)}/{target_count}")
# Try Ollama llama3.1:8b or LMS
try:
payload = {
"model": "llama3.1:8b",
"prompt": prompt,
"stream": False,
"options": {"temperature": 0.85, "top_p": 0.95}
}
res = requests.post(OLLAMA_URL, json=payload, timeout=60)
if res.status_code == 200:
resp_text = res.json().get("response", "").strip()
# Parse JSON array
start_idx = resp_text.find("[")
end_idx = resp_text.rfind("]")
if start_idx != -1 and end_idx != -1:
raw_json = resp_text[start_idx:end_idx+1]
items = json.loads(raw_json)
for item in items:
item_clean = item.strip()
if item_clean and len(item_clean) > 3:
hinglish_samples.add(item_clean)
except Exception as e:
print(f" Ollama generation warning: {e}")
# Fallback/template expansion if needed to reach target_count
if len(hinglish_samples) < target_count:
print(f" Expanding via templates to reach {target_count}...")
greetings = ["hey", "hello", "hi", "suno", "bhai", "vox", "arrey", "namaste"]
phrases = [
"kya haal hai", "kya chal raha hai", "kaise ho", "kya chal raha h", "sab badiya",
"good morning", "good evening", "good night", "kya bolte ho", "bore ho raha hoon",
"kuch batao na", "ek joke sunao", "kaise ho yaar", "tu mast hai", "kya scene hai",
"kaise chal raha hai sab", "kya chal rha hai", "tu kya kar raha hai", "kuch bolo na",
"main theek hoon tum batao", "chalo bye", "phir milte hain", "kya khabar hai"
]
suffixes = ["bhai", "yaar", "vox", "dost", "ji", "buddy", "man", "bro"]
while len(hinglish_samples) < target_count:
g = random.choice(greetings)
p = random.choice(phrases)
s = random.choice(suffixes)
comb = f"{g} {p} {s}".title() if random.random() < 0.2 else f"{g} {p} {s}"
hinglish_samples.add(comb)
result_list = [{
"text": q,
"scope": "ChitChat",
"language": "hinglish",
"source": "synthetic_hinglish_chitchat"
} for q in list(hinglish_samples)[:target_count]]
return result_list
def main():
print("=== Phase 1.2: Base Dataset Audit & Fast-Path ChitChat Mapping ===")
# 1. Load EN & HI Generic (ChitChat)
en_generic = load_base_file("en_generic.jsonl", "ChitChat", "en", "base_generic")
hi_generic = load_base_file("hi_generic.jsonl", "ChitChat", "hi", "base_generic")
print(f"Loaded {len(en_generic)} EN generic items -> ChitChat")
print(f"Loaded {len(hi_generic)} HI generic items -> ChitChat")
# 2. Generate 1,000 Hinglish ChitChat items
hinglish_chitchat = generate_hinglish_chitchat_llm(1000)
print(f"Generated {len(hinglish_chitchat)} Hinglish ChitChat items")
all_chitchat = en_generic + hi_generic + hinglish_chitchat
print(f"Total ChitChat dataset count: {len(all_chitchat)}")
# Save chitchat_base.jsonl
chitchat_file = os.path.join(OUTPUT_DIR, "chitchat_base.jsonl")
with open(chitchat_file, "w", encoding="utf-8") as f:
for item in all_chitchat:
f.write(json.dumps(item, ensure_ascii=False) + "\n")
print(f"Saved {len(all_chitchat)} ChitChat items to {chitchat_file}")
# 3. Isolate 6,022 semantic queries
en_semantic = load_base_file("en_semantic.jsonl", "PENDING_RELABEL", "en", "base_semantic")
hi_semantic = load_base_file("hi_semantic.jsonl", "PENDING_RELABEL", "hi", "base_semantic")
all_semantic = en_semantic + hi_semantic
print(f"Isolated {len(en_semantic)} EN semantic items")
print(f"Isolated {len(hi_semantic)} HI semantic items")
print(f"Total raw semantic items to relabel: {len(all_semantic)}")
semantic_file = os.path.join(OUTPUT_DIR, "semantic_raw.jsonl")
with open(semantic_file, "w", encoding="utf-8") as f:
for item in all_semantic:
f.write(json.dumps(item, ensure_ascii=False) + "\n")
print(f"Saved {len(all_semantic)} raw semantic items to {semantic_file}")
if __name__ == "__main__":
main()
|