Repository navigation
Expand file tree
/
Copy pathexamples.py
More file actions
203 lines (155 loc) · 5.86 KB
/
Copy pathexamples.py
File metadata and controls
203 lines (155 loc) · 5.86 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
"""
Simple usage examples for the clean Arabic normalizer.
Enhanced with massive dataset processing examples.
"""
from pathlib import Path
from config import Config
from arabic_normalizer import ArabicNormalizer
def example_basic_usage():
"""Basic usage example."""
print("=== Basic Usage Example ===")
# Initialize with default configuration
normalizer = ArabicNormalizer()
# Setup LLM (requires API key)
try:
normalizer.setup_llm()
print("✓ LLM setup successful")
# Process a single file
success = normalizer.process_file("input.txt", "output.txt")
if success:
print("✓ File processed successfully")
else:
print("✗ File processing failed")
except Exception as e:
print(f"✗ LLM setup failed: {e}")
def example_custom_config():
"""Example with custom configuration."""
print("\n=== Custom Configuration Example ===")
# Create custom configuration
config = Config(
llm_model="gemini-1.5-flash",
max_tokens=1500,
temperature=0.1,
chunk_size=600,
preserve_sudanese_expressions=True,
min_paragraph_length=25,
max_paragraph_length=1500,
api_delay=2.0
)
# Initialize normalizer with custom config
normalizer = ArabicNormalizer(config)
try:
normalizer.setup_llm()
# Process batch of files
count = normalizer.process_batch("input_dir", "output_dir")
print(f"✓ Processed {count} files")
except Exception as e:
print(f"✗ Error: {e}")
def example_config_file():
"""Example using configuration file."""
print("\n=== Configuration File Example ===")
# Load configuration from file
config = Config.load_from_file("config.json")
normalizer = ArabicNormalizer(config)
try:
normalizer.setup_llm()
# Process single file
normalizer.process_file("large_document.txt", "normalized_document.txt")
except Exception as e:
print(f"✗ Error: {e}")
def example_massive_processing():
"""Example for processing massive datasets (3+ billion words)."""
print("\n=== Massive Dataset Processing Example ===")
# Ultra-conservative config for massive processing
config = Config(
llm_model="gemini-1.5-flash",
max_tokens=800,
temperature=0.05,
chunk_size=300,
requests_per_minute=3, # Very conservative
api_delay=15.0, # Long delays
max_retries=8, # More retries
checkpoint_every=50, # Frequent checkpoints
enable_resume=True
)
normalizer = ArabicNormalizer(config)
try:
normalizer.setup_llm()
# Estimate processing time first
huge_file = "3_billion_words.txt"
if Path(huge_file).exists():
estimate = normalizer.estimate_processing_time(huge_file)
print(f"File size: {estimate['file_size_gb']:.2f} GB")
print(f"Estimated processing time: {estimate['estimated_days']:.1f} days")
print(f"Rate: {estimate['requests_per_hour']:.1f} requests/hour")
# Process with massive mode and resume capability
success = normalizer.process_file_massive(
huge_file,
"normalized_3_billion_words.txt",
resume=True
)
if success:
print("✓ Massive processing completed!")
else:
print("✗ Massive processing failed")
else:
print(f"Demo file {huge_file} not found")
except Exception as e:
print(f"✗ Error: {e}")
def example_batch_massive():
"""Example for batch processing multiple huge files."""
print("\n=== Massive Batch Processing Example ===")
config = Config(
requests_per_minute=2, # Ultra-conservative
api_delay=20.0,
checkpoint_every=25,
enable_resume=True
)
normalizer = ArabicNormalizer(config)
try:
normalizer.setup_llm()
# Process entire directory of massive files
count = normalizer.process_batch_massive(
"huge_files_dir",
"normalized_huge_files"
)
print(f"✓ Processed {count} massive files")
except Exception as e:
print(f"✗ Error: {e}")
def example_text_normalization():
"""Example of normalizing text directly."""
print("\n=== Direct Text Normalization Example ===")
# Sample Arabic text with issues
sample_text = """
هذا نص عربي بمشاكل في المسافات
ويحتوي على أخطاء إملائية وتطويل زاااائد
"""
config = Config()
normalizer = ArabicNormalizer(config)
try:
normalizer.setup_llm()
# Normalize the text
normalized = normalizer.normalize_text(sample_text)
print("Original text:")
print(repr(sample_text))
print("\nNormalized text:")
print(repr(normalized))
except Exception as e:
print(f"✗ Error: {e}")
if __name__ == "__main__":
print("Arabic Text Normalizer - Usage Examples")
print("=" * 60)
example_basic_usage()
example_custom_config()
example_config_file()
example_text_normalization()
example_massive_processing()
example_batch_massive()
print("\n" + "=" * 60)
print("MASSIVE PROCESSING TIPS:")
print("• Use --estimate-only first to plan your processing")
print("• Always use --massive --resume for huge files")
print("• Set GEMINI_API_KEY environment variable")
print("• Monitor checkpoints/ directory for progress")
print("• Expect 3+ days for 3 billion words with conservative settings")
print("=" * 60)