-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathpreprocess_data.py
More file actions
98 lines (81 loc) · 3.51 KB
/
Copy pathpreprocess_data.py
File metadata and controls
98 lines (81 loc) · 3.51 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
from datasketch import MinHash, MinHashLSH
from langdetect import detect
from bs4 import BeautifulSoup
import re
from collections import Counter
import pandas as pd
import json
# step 1 : remove duplicates using MinHash and Locality Sensitive Hashing (LSH)
def minhash_deduplication(texts, threshold=0.7):
lsh = MinHashLSH(threshold=threshold, num_perm=128)
unique_texts = []
for i, doc in enumerate(texts):
m = MinHash(num_perm=128)
for word in set(doc.split()):
m.update(word.encode('utf8'))
if not lsh.query(m):
lsh.insert(f"doc{i}", m)
unique_texts.append(doc)
return unique_texts
# step 2: clean HTML tags and filter by language
def clean_html_and_filter_lang(texts, lang='en'):
filtered = []
for txt in texts:
txt = BeautifulSoup(txt, 'html.parser').get_text()
try:
if detect(txt.strip()) == lang:
filtered.append(txt.strip())
except:
continue
return filtered
# step3: remove personally identifiable information (PII)
def strip_pii(text):
text = re.sub(r'[\w\.-]+@[\w\.-]+', '[EMAIL]', text)
text = re.sub(r'\b\d{12,19}\b', '[CREDIT_CARD]', text)
text = re.sub(r'\b(?:\d{3}-){2}\d{4}\b', '[PHONE]', text)
return text
# step4: remove repetitive n-grams
def remove_repetitive_ngrams(text, n=3, threshold=3):
words = text.split()
ngrams = [' '.join(words[i:i+n]) for i in range(len(words)-n+1)]
counts = Counter(ngrams)
repetitive = [ngram for ngram, count in counts.items() if count >= threshold]
for phrase in repetitive:
# regex-safe version of the phrase
escaped_phrase = re.escape(phrase)
# match the phrase repeated 2+ times with optional whitespace
text = re.sub(rf'(?:{escaped_phrase}\s*){{{threshold},}}', phrase + ' ', text)
# Remove extra spaces
text = re.sub(r'\s{2,}', ' ', text).strip()
return text
def load_jsonl(file_path):
records = []
with open(file_path, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if line:
records.append(json.loads(line))
return records
def preprocess_from_jsonl(file_path):
# ── 讀取 .jsonl ───────────────────────────────────
records = load_jsonl(file_path)
print(f"Loaded {len(records)} records from {file_path}")
# 取出 content 欄位作為 raw_dataset
raw_dataset = [r["content"] for r in records if "content" in r]
print(f"Raw dataset size: {len(raw_dataset)}")
# ── 前處理步驟 ────────────────────────────────────
step1 = clean_html_and_filter_lang(raw_dataset)
print(f"Step 1 - After HTML cleaning and language filtering: {len(step1)} docs")
step2 = minhash_deduplication(step1)
print(f"Step 2 - After deduplication: {len(step2)} docs")
step3 = [strip_pii(t) for t in step2]
print(f"Step 3 - After stripping PII: {len(step3)} docs")
cleaned_data = [remove_repetitive_ngrams(t) for t in step3]
print(f"Step 4 - After removing repetitive n-grams: {len(cleaned_data)} docs")
return cleaned_data[:200] # return a sample of cleaned data for inspection
if __name__ == "__main__":
cleaned = preprocess_from_jsonl("./output/web_outputs.jsonl")
print("\n✅ Cleaned dataset sample:")
for idx, text in enumerate(cleaned):
print(f"--- Article {idx + 1} ---")
print(text[:300])