| """Scan filtered/filtered_*.jsonl, report: |
| 1. Top template-style basenames (counts + total bytes + duplicate ratio per file). |
| 2. Top high-frequency lines (stripped) across the whole corpus, with sample coverage. |
| 3. Top short-phrase n-grams (3-gram and 5-gram over tokens of stripped lines), sampled. |
| |
| Output: filtered_repetition_report.json + .md |
| Pure-stdlib, single-pass over disk, ~10GB. |
| """ |
| from __future__ import annotations |
|
|
| import json |
| import os |
| import re |
| import sys |
| from collections import Counter, defaultdict |
| from glob import glob |
| from pathlib import Path |
|
|
| SRC_DIR = Path("/raid/data/weifeng/Datasets/filtered") |
| OUT_JSON = Path("/raid/data/weifeng/Datasets/filtered_repetition_report.json") |
| OUT_MD = Path("/raid/data/weifeng/Datasets/filtered_repetition_report.md") |
|
|
| TEMPLATE_BASENAMES = { |
| "__init__.py", "setup.py", "config.py", "conf.py", "settings.py", |
| "constants.py", "version.py", "_version.py", "manage.py", |
| "decorators.py", "exceptions.py", "errors.py", "utils.py", |
| "types.py", "schemas.py", "logging.py", "logger.py", |
| "__main__.py", "wsgi.py", "asgi.py", "urls.py", "apps.py", |
| "models.py", "admin.py", |
| } |
|
|
| |
| TOP_LINES_KEEP = 2000 |
| TOP_BASENAMES_KEEP = 500 |
| NGRAM_TOP_KEEP = 500 |
| NGRAM_SAMPLE_EVERY = 20 |
|
|
| |
| MIN_LINE_LEN = 8 |
| MAX_LINE_LEN = 200 |
|
|
| |
| TOKEN_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*|[^\s\w]") |
|
|
|
|
| def main() -> None: |
| files = sorted(glob(str(SRC_DIR / "filtered_*.jsonl"))) |
| print(f"found {len(files)} jsonl files", flush=True) |
|
|
| basename_counter: Counter[str] = Counter() |
| basename_bytes: Counter[str] = Counter() |
| basename_dup_lines: Counter[str] = Counter() |
| line_counter: Counter[str] = Counter() |
| line_sample_coverage: Counter[str] = Counter() |
| ngram3_counter: Counter[tuple] = Counter() |
| ngram5_counter: Counter[tuple] = Counter() |
|
|
| total_samples = 0 |
| total_bytes = 0 |
| py_samples = 0 |
|
|
| for fi, fp in enumerate(files): |
| do_ngram = (fi % NGRAM_SAMPLE_EVERY == 0) |
| try: |
| with open(fp, encoding="utf-8") as f: |
| for line in f: |
| try: |
| o = json.loads(line) |
| except Exception: |
| continue |
| path = o.get("path", "") or "" |
| text = o.get("text", "") or "" |
| if not text: |
| continue |
| total_samples += 1 |
| total_bytes += len(text) |
| basename = os.path.basename(path) if path else "<unknown>" |
| basename_counter[basename] += 1 |
| basename_bytes[basename] += len(text) |
| if path.endswith(".py"): |
| py_samples += 1 |
|
|
| |
| file_lines = [ln.strip() for ln in text.split("\n")] |
| file_line_cnt = Counter(ln for ln in file_lines if MIN_LINE_LEN <= len(ln) <= MAX_LINE_LEN) |
| dup_here = sum(c - 1 for c in file_line_cnt.values() if c > 1) |
| basename_dup_lines[basename] += dup_here |
|
|
| |
| seen_in_this_sample: set[str] = set() |
| for ln, c in file_line_cnt.items(): |
| line_counter[ln] += c |
| if ln not in seen_in_this_sample: |
| line_sample_coverage[ln] += 1 |
| seen_in_this_sample.add(ln) |
|
|
| if do_ngram and len(file_lines) > 5: |
| |
| for ln in file_lines[:400]: |
| toks = TOKEN_RE.findall(ln) |
| if len(toks) < 3: |
| continue |
| for i in range(len(toks) - 2): |
| ngram3_counter[tuple(toks[i:i + 3])] += 1 |
| if len(toks) >= 5: |
| for i in range(len(toks) - 4): |
| ngram5_counter[tuple(toks[i:i + 5])] += 1 |
|
|
| |
| if total_samples % 100000 == 0: |
| print(f" [{total_samples}] processed; " |
| f"line_counter={len(line_counter):,} ngram3={len(ngram3_counter):,}", |
| flush=True) |
| if len(line_counter) > 5_000_000: |
| |
| cutoff = line_counter.most_common(2_000_000) |
| line_counter = Counter(dict(cutoff)) |
| line_sample_coverage = Counter({k: line_sample_coverage[k] for k, _ in cutoff}) |
| if len(ngram3_counter) > 3_000_000: |
| ngram3_counter = Counter(dict(ngram3_counter.most_common(1_000_000))) |
| if len(ngram5_counter) > 3_000_000: |
| ngram5_counter = Counter(dict(ngram5_counter.most_common(1_000_000))) |
| except Exception as e: |
| print(f" ERR {fp}: {e}", file=sys.stderr, flush=True) |
|
|
| print(f"DONE: {total_samples:,} samples, {total_bytes/1e9:.2f} GB", flush=True) |
|
|
| |
| top_basenames = basename_counter.most_common(TOP_BASENAMES_KEEP) |
| top_lines = line_counter.most_common(TOP_LINES_KEEP) |
| top_ngram3 = ngram3_counter.most_common(NGRAM_TOP_KEEP) |
| top_ngram5 = ngram5_counter.most_common(NGRAM_TOP_KEEP) |
|
|
| report = { |
| "total_samples": total_samples, |
| "total_bytes": total_bytes, |
| "py_samples": py_samples, |
| "n_files_scanned": len(files), |
| "top_basenames": [ |
| { |
| "basename": b, |
| "count": c, |
| "pct_of_corpus": round(c / total_samples * 100, 3), |
| "total_bytes": basename_bytes[b], |
| "dup_lines_within_files": basename_dup_lines[b], |
| } |
| for b, c in top_basenames |
| ], |
| "top_lines_corpus_wide": [ |
| { |
| "line": ln, |
| "occurrences": c, |
| "n_samples_containing": line_sample_coverage[ln], |
| "pct_samples": round(line_sample_coverage[ln] / total_samples * 100, 3), |
| } |
| for ln, c in top_lines[:500] |
| ], |
| "top_3grams": [{"tokens": list(t), "count": c} for t, c in top_ngram3], |
| "top_5grams": [{"tokens": list(t), "count": c} for t, c in top_ngram5], |
| "ngram_sampling_note": f"n-gram counted on every {NGRAM_SAMPLE_EVERY}th file, first 400 lines per sample", |
| } |
| OUT_JSON.write_text(json.dumps(report, ensure_ascii=False, indent=2)) |
| print(f"wrote {OUT_JSON} ({OUT_JSON.stat().st_size/1e6:.1f} MB)", flush=True) |
|
|
| |
| md = [] |
| md.append(f"# Filtered pretrain repetition report\n") |
| md.append(f"- samples: **{total_samples:,}**, bytes: **{total_bytes/1e9:.2f} GB**, .py: {py_samples:,}\n") |
| md.append(f"- files scanned: {len(files)}\n\n") |
|
|
| md.append("## Top template basenames (by sample count)\n\n") |
| md.append("| basename | count | % corpus | total bytes | dup lines (within-file) |\n|---|---:|---:|---:|---:|\n") |
| for b, c in top_basenames[:60]: |
| md.append(f"| `{b}` | {c} | {c/total_samples*100:.2f}% | {basename_bytes[b]/1e6:.1f} MB | {basename_dup_lines[b]} |\n") |
|
|
| md.append("\n## Top corpus-wide lines (frequency)\n\n") |
| md.append("| line | occurrences | samples containing | % samples |\n|---|---:|---:|---:|\n") |
| for ln, c in top_lines[:80]: |
| disp = ln.replace("|", "\\|")[:120] |
| md.append(f"| `{disp}` | {c} | {line_sample_coverage[ln]} | {line_sample_coverage[ln]/total_samples*100:.2f}% |\n") |
|
|
| md.append("\n## Top 5-grams (token-level)\n\n") |
| md.append("| 5-gram | count |\n|---|---:|\n") |
| for t, c in top_ngram5[:60]: |
| disp = " ".join(t).replace("|", "\\|")[:120] |
| md.append(f"| `{disp}` | {c} |\n") |
|
|
| md.append("\n## Top 3-grams (token-level)\n\n") |
| md.append("| 3-gram | count |\n|---|---:|\n") |
| for t, c in top_ngram3[:60]: |
| disp = " ".join(t).replace("|", "\\|")[:120] |
| md.append(f"| `{disp}` | {c} |\n") |
|
|
| OUT_MD.write_text("".join(md)) |
| print(f"wrote {OUT_MD}", flush=True) |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|