Buckets:
| import json, re | |
| KW = { | |
| 'sd-wan': r'SD-WAN|OMP|vSmart|cEdge|TLOC|vBond', | |
| 'sd-access': r'SD-Access|fabric edge|LISP|VXLAN|SXP|CPNd|ISE', | |
| 'routing': r'OSPF|BGP|EIGRP|IS-IS|ISIS|area 0|stub|adjacency|IGP', | |
| 'multicast': r'multicast|IGMP|PIM|rendezvous|MSDP|Anycast-RP|mDNS', | |
| 'wan-dmvpn': r'DMVPN|IPsec|MPLS|NHRP|FlexVPN|branch|VRF|WAN link|failover|HSRP|VRRP', | |
| 'qos': r'QoS|CoS|DSCP|shaping|queuing|LLQ|policing|jitter|video quality', | |
| 'mgmt-automation': r'NETCONF|RESTCONF|YANG|SNMP|NetFlow|telemetry|Cisco DNA|automation|Ansible|API', | |
| } | |
| TOPICS = list(KW) | |
| def topic(r): | |
| t = r['stem'] + ' ' + ' '.join(r['options'].values()) | |
| for name, pat in KW.items(): | |
| if re.search(pat, t, re.I): | |
| return name | |
| return 'other' | |
| recs = [json.loads(l) for l in open('/work/questions_all.jsonl')] | |
| for r in recs: | |
| r['topic'] = topic(r) | |
| exhibit_first = [r for r in recs if r['has_exhibit'] and r['explanation'] and r['options'] and r['correct']] | |
| # sort exhibit candidates to maximize topic spread | |
| picked = [] | |
| used_topics = set() | |
| # pass 1: one exhibit per topic | |
| for r in sorted(exhibit_first, key=lambda x: (x['topic'] != 'multicast', x['topic'] != 'mgmt-automation', x['topic'] != 'qos')): | |
| pass | |
| remaining = sorted(exhibit_first, key=lambda x: x['topic']) | |
| for r in remaining: | |
| if r['topic'] not in used_topics: | |
| picked.append(r); used_topics.add(r['topic']) | |
| for r in remaining: | |
| if r not in picked: | |
| picked.append(r) | |
| if len([p for p in picked if p['has_exhibit']]) >= 8: | |
| break | |
| pilot = list(picked) | |
| assert all(p['has_exhibit'] for p in pilot) | |
| # fill up to 20 with topic spread, preferring explanation, skipping DD/empty options | |
| rest = [r for r in recs if r not in pilot and r['options'] and r['correct']] | |
| # order: has explanation first, then by topic rarity of unpicked | |
| counts = {} | |
| for p in pilot: | |
| counts[p['topic']] = counts.get(p['topic'], 0) + 1 | |
| rest.sort(key=lambda r: (0 if r['explanation'] else 1, counts.get(r['topic'], 0), TOPICS.index(r['topic']) if r['topic'] in TOPICS else 99)) | |
| for r in rest: | |
| if len(pilot) >= 20: | |
| break | |
| pilot.append(r) | |
| counts[r['topic']] = counts.get(r['topic'], 0) + 1 | |
| assert len(pilot) == 20 | |
| n_ex = sum(p['has_exhibit'] for p in pilot) | |
| assert n_ex >= 6, n_ex | |
| topics_seen = set(p['topic'] for p in pilot) | |
| print('pilot size:', len(pilot), 'exhibits:', n_ex, 'topics:', sorted(topics_seen)) | |
| with open('/work/pilot_questions.jsonl', 'w') as f: | |
| for p in pilot: | |
| f.write(json.dumps(p) + '\n') | |
| print('wrote pilot_questions.jsonl') | |
| for p in pilot: | |
| print(p['exam'], p['qnum'], p['page'], p['topic'], p['has_exhibit'], bool(p['explanation']), repr(p['correct'])) | |
Xet Storage Details
- Size:
- 2.7 kB
- Xet hash:
- 62f5f3e023e86d48e266bd155a95cc99c08a2537985836dabf4d9374f9196f85
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.