luispoveda93's picture
download
raw
2.7 kB
import json, re
KW = {
'sd-wan': r'SD-WAN|OMP|vSmart|cEdge|TLOC|vBond',
'sd-access': r'SD-Access|fabric edge|LISP|VXLAN|SXP|CPNd|ISE',
'routing': r'OSPF|BGP|EIGRP|IS-IS|ISIS|area 0|stub|adjacency|IGP',
'multicast': r'multicast|IGMP|PIM|rendezvous|MSDP|Anycast-RP|mDNS',
'wan-dmvpn': r'DMVPN|IPsec|MPLS|NHRP|FlexVPN|branch|VRF|WAN link|failover|HSRP|VRRP',
'qos': r'QoS|CoS|DSCP|shaping|queuing|LLQ|policing|jitter|video quality',
'mgmt-automation': r'NETCONF|RESTCONF|YANG|SNMP|NetFlow|telemetry|Cisco DNA|automation|Ansible|API',
}
TOPICS = list(KW)
def topic(r):
t = r['stem'] + ' ' + ' '.join(r['options'].values())
for name, pat in KW.items():
if re.search(pat, t, re.I):
return name
return 'other'
recs = [json.loads(l) for l in open('/work/questions_all.jsonl')]
for r in recs:
r['topic'] = topic(r)
exhibit_first = [r for r in recs if r['has_exhibit'] and r['explanation'] and r['options'] and r['correct']]
# sort exhibit candidates to maximize topic spread
picked = []
used_topics = set()
# pass 1: one exhibit per topic
for r in sorted(exhibit_first, key=lambda x: (x['topic'] != 'multicast', x['topic'] != 'mgmt-automation', x['topic'] != 'qos')):
pass
remaining = sorted(exhibit_first, key=lambda x: x['topic'])
for r in remaining:
if r['topic'] not in used_topics:
picked.append(r); used_topics.add(r['topic'])
for r in remaining:
if r not in picked:
picked.append(r)
if len([p for p in picked if p['has_exhibit']]) >= 8:
break
pilot = list(picked)
assert all(p['has_exhibit'] for p in pilot)
# fill up to 20 with topic spread, preferring explanation, skipping DD/empty options
rest = [r for r in recs if r not in pilot and r['options'] and r['correct']]
# order: has explanation first, then by topic rarity of unpicked
counts = {}
for p in pilot:
counts[p['topic']] = counts.get(p['topic'], 0) + 1
rest.sort(key=lambda r: (0 if r['explanation'] else 1, counts.get(r['topic'], 0), TOPICS.index(r['topic']) if r['topic'] in TOPICS else 99))
for r in rest:
if len(pilot) >= 20:
break
pilot.append(r)
counts[r['topic']] = counts.get(r['topic'], 0) + 1
assert len(pilot) == 20
n_ex = sum(p['has_exhibit'] for p in pilot)
assert n_ex >= 6, n_ex
topics_seen = set(p['topic'] for p in pilot)
print('pilot size:', len(pilot), 'exhibits:', n_ex, 'topics:', sorted(topics_seen))
with open('/work/pilot_questions.jsonl', 'w') as f:
for p in pilot:
f.write(json.dumps(p) + '\n')
print('wrote pilot_questions.jsonl')
for p in pilot:
print(p['exam'], p['qnum'], p['page'], p['topic'], p['has_exhibit'], bool(p['explanation']), repr(p['correct']))

Xet Storage Details

Size:
2.7 kB
·
Xet hash:
62f5f3e023e86d48e266bd155a95cc99c08a2537985836dabf4d9374f9196f85

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.