""" Quality filtering, contamination control, and token-density validation. Takes the deduplicated documents and applies final quality gates: QUALITY FILTERS: - Real source code (language signatures at line starts) - Balanced delimiters (braces/parens/brackets roughly match) - Reasonable length (200–100k chars) - Low repetition (unique line ratio) - Clean ASCII (low non-ASCII ratio) - High code-to-prose ratio - Has structure (function/class/struct definitions) CONTAMINATION CONTROL: - No test files (already filtered in download, double-check here) - No auto-generated code markers - No license-only files - No binary/garbage content - No files with extremely high repetition (copy-paste blocks) TOKEN DENSITY: - Every kept document must be "dense in tokens" — meaning the content tokenizes to a meaningful number of tokens relative to its character length (no whitespace-padding, no huge comment blocks). - Reports token density stats using the project tokenizer. Input: data/dedup/documents_dedup.jsonl Output: data/quality/documents_quality.jsonl + quality_stats.json """ import json import os import re import sys PROJECT_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) DATA_DIR = os.path.join(PROJECT_DIR, "data") DEDUP_DOCS_PATH = os.path.join(DATA_DIR, "dedup", "documents_dedup.jsonl") QUALITY_DIR = os.path.join(DATA_DIR, "quality") QUALITY_DOCS_PATH = os.path.join(QUALITY_DIR, "documents_quality.jsonl") STATS_PATH = os.path.join(QUALITY_DIR, "quality_stats.json") # ─── Code structure patterns (must be at line start) ───────────────────────── CODE_SIGNATURES = { "python": re.compile(r"^(def |class |import |from \S+ import |if __name__|@|async def )", re.M), "js_ts": re.compile(r"^(function |const |let |var |class |export |import |async function|interface |type \w+ =|enum )", re.M), "rust": re.compile(r"^(fn |pub fn |impl |struct |enum |trait |mod |use |pub struct|pub enum|macro_rules!|pub trait)", re.M), "go": re.compile(r"^(func |package |import |type \w+ struct|var |const )", re.M), "c": re.compile(r"^(#include|#define|#ifndef|#ifdef|#if |#endif|typedef |struct \w+|static |void |int |char )", re.M), "cpp": re.compile(r"^(#include|#define|#ifndef|#ifdef|#if |#endif|template |class \w+|namespace |struct \w+|void |int )", re.M), "java": re.compile(r"^(public |private |protected |class \w+|import |package |interface )", re.M), "csharp": re.compile(r"^(public |private |protected |internal |class \w+|using |namespace |interface )", re.M), "ruby": re.compile(r"^(def |class |module |require |require_relative |attr_|include )", re.M), } # ─── Contamination / exclusion patterns ────────────────────────────────────── AUTO_GEN = re.compile( r"(?:auto[- ]generated|do not edit|generated by|code generated|" r"DO NOT MODIFY|@generated|automatically generated|" r"this file was generated)", re.IGNORECASE, ) LICENSE_ONLY = re.compile(r"^(?:/\*|//|#)\s*(?:copyright|licensed|mit license|apache license|bsd license|gnu|gpl)", re.I) PROSE_LINE = re.compile(r"^[A-Z][a-z]+ .* [a-z]+\.$", re.M) # ─── Quality scoring ───────────────────────────────────────────────────────── def detect_language_signatures(doc: dict) -> int: """Count code-structure keywords at line starts. Returns count.""" content = doc["content"] lang = doc["language"] pat = CODE_SIGNATURES.get(lang) if pat is None: # Try all return sum(len(p.findall(content)) for p in CODE_SIGNATURES.values()) return len(pat.findall(content)) def score_document(doc: dict) -> tuple[float, str | None]: """Score a document 0.0–1.0 on quality. Returns (score, reject_reason).""" content = doc["content"] lines = content.split("\n") n_lines = len(lines) length = len(content) # ─── Hard rejects (contamination) ────────────────────────────────────── if length < 200: return 0.0, "too_short" if length > 200_000: return 0.0, "too_long" if n_lines < 5: return 0.0, "too_few_lines" # Auto-generated code if AUTO_GEN.search(content[:2000]): return 0.0, "auto_generated" # License-only files (first 10 lines are all license comments) first_lines = "\n".join(lines[:10]) if LICENSE_ONLY.search(first_lines) and n_lines < 30: return 0.0, "license_only" # High non-ASCII (garbage/encoding issues) non_ascii = sum(1 for c in content if ord(c) > 127) if length > 0 and non_ascii / length > 0.03: return 0.0, "high_non_ascii" # Extremely high repetition (copy-paste blocks) unique_lines = len(set(lines)) unique_ratio = unique_lines / max(n_lines, 1) if unique_ratio < 0.20: return 0.0, "high_repetition" # ─── Soft scoring ────────────────────────────────────────────────────── score = 0.0 # Base: passes hard filters score += 0.15 # Length quality (sweet spot: 500–30000 chars) if 500 <= length <= 30000: score += 0.15 elif 200 <= length <= 80000: score += 0.08 # Delimiter balance braces = content.count("{") - content.count("}") parens = content.count("(") - content.count(")") brackets = content.count("[") - content.count("]") total_delims = content.count("{") + content.count("(") + content.count("[") if total_delims > 0: imbalance = abs(braces) + abs(parens) + abs(brackets) balance_ratio = 1.0 - (imbalance / max(total_delims, 1)) score += 0.15 * max(balance_ratio, 0.0) # Code structure density struct_count = detect_language_signatures(doc) struct_density = min(struct_count / max(n_lines, 1) * 10, 1.0) score += 0.15 * struct_density # Low repetition (unique line ratio) if unique_ratio > 0.7: score += 0.10 elif unique_ratio > 0.5: score += 0.05 else: score -= 0.05 # Clean ASCII if length > 0 and non_ascii / length < 0.005: score += 0.05 # Indentation quality (indented lines indicate real code structure) indented = sum(1 for l in lines if l.startswith(" ") or l.startswith("\t")) if indented > 0 and indented / max(n_lines, 1) > 0.15: score += 0.05 # Penalize high prose ratio (documentation, not code) prose_lines = len(PROSE_LINE.findall(content)) prose_ratio = prose_lines / max(n_lines, 1) if prose_ratio > 0.20: score -= 0.15 # Comment density (sweet spot: 3–30%) comment_lines = 0 for line in lines: s = line.strip() if s.startswith("#") or s.startswith("//") or s.startswith("/*") \ or s.startswith("*") or s.startswith('"""') or s.startswith("'''") \ or s.startswith("///") or s.startswith("//!"): comment_lines += 1 comment_ratio = comment_lines / max(n_lines, 1) if 0.03 <= comment_ratio <= 0.30: score += 0.10 elif comment_ratio > 0.50: score -= 0.10 # too many comments = doc, not code return min(max(score, 0.0), 1.0), None def check_token_density(doc: dict) -> tuple[bool, float]: """Check that a document is dense in tokens (not whitespace-padded). Returns (passes, chars_per_token_ratio). A good code document should have ~2.5-5 chars per token. If the ratio is very high (>15), it's likely whitespace/garbage. If very low (<1.5), it may be all symbols. """ content = doc["content"] # Rough estimate: count non-whitespace characters as a proxy # Real tokenization happens with the tokenizer, but this is a fast filter non_ws = len(content) - content.count(" ") - content.count("\n") - content.count("\t") - content.count("\r") # Approximate token count: split on whitespace + common code delimiters approx_tokens = len(re.findall(r"\w+|[^\w\s]", content)) if approx_tokens == 0: return False, 0.0 chars_per_token = len(content) / approx_tokens # Good density: 2.0 - 8.0 chars per token passes = 2.0 <= chars_per_token <= 12.0 return passes, chars_per_token def main(): os.makedirs(QUALITY_DIR, exist_ok=True) print(f"Loading deduped documents from {DEDUP_DOCS_PATH}...") docs = [] with open(DEDUP_DOCS_PATH, "r", encoding="utf-8") as f: for line in f: docs.append(json.loads(line)) print(f" Loaded {len(docs):,} documents") kept = [] reject_reasons = {} scores = [] token_densities = [] for i, doc in enumerate(docs): score, reject = score_document(doc) if reject: reject_reasons[reject] = reject_reasons.get(reject, 0) + 1 continue if score < 0.40: reject_reasons["low_score"] = reject_reasons.get("low_score", 0) + 1 continue # Token density check dense, cpt = check_token_density(doc) token_densities.append(cpt) if not dense: reject_reasons["low_token_density"] = reject_reasons.get("low_token_density", 0) + 1 continue doc = dict(doc) doc["quality_score"] = round(score, 4) kept.append(doc) scores.append(score) if (i + 1) % 5000 == 0: print(f" Processed {i+1}/{len(docs)} | kept {len(kept)} | " f"rejected {i+1 - len(kept)}") # Stats final_size = sum(len(d["content"]) for d in kept) avg_score = sum(scores) / len(scores) if scores else 0 avg_cpt = sum(token_densities) / len(token_densities) if token_densities else 0 lang_counts = {} for d in kept: lang_counts[d["language"]] = lang_counts.get(d["language"], 0) + 1 stats = { "input": len(docs), "kept": len(kept), "rejected": len(docs) - len(kept), "reject_reasons": reject_reasons, "final_size_mb": final_size / 1e6, "avg_quality_score": round(avg_score, 4), "avg_chars_per_token": round(avg_cpt, 2), "language_distribution": lang_counts, "min_score_threshold": 0.40, } print("\n" + "=" * 60) print("QUALITY FILTERING COMPLETE") print("=" * 60) print(f" Input: {len(docs):,}") print(f" Kept: {len(kept):,}") print(f" Rejected: {len(docs) - len(kept):,}") print(f" Reject reasons: {reject_reasons}") print(f" Final size: {final_size / 1e6:.1f} MB") print(f" Avg quality score: {avg_score:.3f}") print(f" Avg chars/token: {avg_cpt:.2f}") print(f" Languages: {lang_counts}") # Write quality-filtered documents with open(QUALITY_DOCS_PATH, "w", encoding="utf-8") as f: for doc in kept: f.write(json.dumps(doc, ensure_ascii=False) + "\n") print(f"\nQuality documents written to {QUALITY_DOCS_PATH}") with open(STATS_PATH, "w") as f: json.dump(stats, f, indent=2) print(f"Stats written to {STATS_PATH}") if __name__ == "__main__": main()