Repository navigation
Expand file tree
/
Copy pathsimulate.py
More file actions
137 lines (113 loc) · 4.18 KB
/
Copy pathsimulate.py
File metadata and controls
137 lines (113 loc) · 4.18 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
"""
simulate.py — Gerçekçi test verisi üretici.
100 simüle pipeline çalışması yazar:
- 2 farklı prompt versiyonu (v1 ve v2)
- v1: retrieval %35 başarısız, generation %20 başarısız
- v2: retrieval %18 başarısız, generation %10 başarısız
- Farklı günlere dağıtılmış timestamp'ler
- Bilerek anomali günü: 3 gün öncesine yüksek failure rate
"""
import os
import sys
import random
from datetime import datetime, timezone, timedelta
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
from failure_forensics.logger import log_step, new_run_id, clear_logs
random.seed(42)
STEPS = ["retrieval", "reranking", "generation", "citation"]
QUERIES = [
"Transformer mimarisinin geleneksel RNN modellerine göre avantajı nedir?",
"A/B testi nasıl yapılır?",
"Cross-encoder ile bi-encoder farkı nedir?",
"LLM halüsinasyonları nasıl önlenir?",
"RAG mimarisinde BM25 ne işe yarar?"
]
# Versiyon bazında adım failure olasılıkları
VERSION_FAIL_PROBS = {
"v1": {
"retrieval": 0.35,
"reranking": 0.15,
"generation": 0.20,
"citation": 0.12,
},
"v2": {
"retrieval": 0.18,
"reranking": 0.08,
"generation": 0.10,
"citation": 0.06,
},
"v3": { # Kasıtlı regresyon: retrieval çok bozulmuş
"retrieval": 0.40,
"reranking": 0.08,
"generation": 0.10,
"citation": 0.06,
},
}
ERROR_TYPES = {
"retrieval": ["no_results", "low_score", "timeout"],
"reranking": ["json_parse", "rerank_failed", "timeout"],
"generation": ["empty_response", "hallucination", "timeout"],
"citation": ["no_citation", "citation_missing"],
}
MODELS = ["gemini-flash-latest"]
def _random_ts(days_ago: float) -> str:
"""Belirli gün öncesine rastgele bir timestamp üretir."""
base = datetime.now(timezone.utc) - timedelta(days=days_ago)
jitter = timedelta(hours=random.uniform(0, 20), minutes=random.randint(0, 59))
return (base + jitter).isoformat()
def simulate(n_runs: int = 150, clear: bool = True):
"""
n_runs kadar simüle pipeline çalışması üretir.
"""
if clear:
clear_logs()
print(f"\n[SIMULATE] {n_runs} run simüle ediliyor...")
run_count = 0
for i in range(n_runs):
# Versiyon dağılımı: ilk 50 → v1, sonraki 50 → v2, son 50 → v3
if i < 50:
version = "v1"
elif i < 100:
version = "v2"
else:
version = "v3"
fail_probs = VERSION_FAIL_PROBS[version]
model = random.choice(MODELS)
run_id = new_run_id()
query = random.choice(QUERIES)
# Timestamp dağılımı: son 8 güne yay
# 3 gün öncesi → anomali günü (2x failure prob)
days_ago = random.uniform(0, 7)
is_anomaly_day = 2.5 <= days_ago <= 3.5
anomaly_multiplier = 2.5 if is_anomaly_day else 1.0
run_success = True
for step in STEPS:
base_prob = fail_probs[step]
effective_prob = min(base_prob * anomaly_multiplier, 0.95)
failed = random.random() < effective_prob
if failed:
error_type = random.choice(ERROR_TYPES[step])
run_success = False
else:
error_type = None
ts = _random_ts(days_ago)
log_step(
run_id=run_id,
step_name=step,
success=not failed,
latency_ms=random.uniform(50, 3000),
token_count=random.randint(10, 500),
input_summary=query,
output_summary=f"Simulated output for {step} #{i}" if not failed else "",
error_type=error_type,
prompt_version=version,
model_name=model,
timestamp=ts,
)
run_count += 1
print(f"[SIMULATE] ✅ {run_count} run tamamlandı.")
print(f"[SIMULATE] v1: 50 run | v2: 50 run | v3: 50 run (v3 = Kasıtlı regresyon)")
print(f"[SIMULATE] Anomali günü: ~3 gün önce (2.5x failure rate)")
print(f"[SIMULATE] Loglar: data/logs/requests.jsonl\n")
if __name__ == "__main__":
simulate()