Skip to main content

stygian_charon/
types.rs

1use std::collections::BTreeMap;
2
3use serde::{Deserialize, Serialize};
4
5/// Target website classification for SLO thresholds.
6///
7/// Used to determine acceptable blocked ratios and risk assessments based on expected
8/// anti-bot posture. Different sites have different security requirements:
9///
10/// - **API**: Machine-to-machine communication; expects very low block ratio.
11/// - **`ContentSite`**: Public web content; moderate block tolerance.
12/// - **`HighSecurity`**: Banking, auth, sensitive data; higher block ratio acceptable.
13/// - **Unknown**: Default classification when unable to determine target type.
14#[derive(
15    Debug, Default, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize,
16)]
17#[serde(rename_all = "snake_case")]
18pub enum TargetClass {
19    /// REST API or GraphQL endpoint; expect clean machine-to-machine paths.
20    Api,
21    /// General content site or e-commerce; browser-like requests expected.
22    ContentSite,
23    /// High-security property (banking, auth, sensitive data); strict anti-bot expected.
24    HighSecurity,
25    /// Unknown or unclassified target.
26    #[default]
27    Unknown,
28}
29
30/// Blocked ratio service-level objectives (SLOs) by target class.
31///
32/// Defines acceptable and concerning block ratios for different target types.
33/// These thresholds guide requirement inference and risk scoring.
34#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
35pub struct BlockedRatioSlo {
36    /// Target class for these SLOs.
37    pub target_class: TargetClass,
38    /// Acceptable block ratio (green threshold); below this is normal.
39    pub acceptable: f64,
40    /// Warning threshold; above this triggers adaptive rate requirement.
41    pub warning: f64,
42    /// Critical threshold; above this indicates severe anti-bot posture.
43    pub critical: f64,
44}
45
46impl BlockedRatioSlo {
47    /// Default SLOs for API targets (0-5% blocks, 10% warning, 15% critical).
48    #[must_use]
49    pub const fn api() -> Self {
50        Self {
51            target_class: TargetClass::Api,
52            acceptable: 0.05,
53            warning: 0.10,
54            critical: 0.15,
55        }
56    }
57
58    /// Default SLOs for content sites (0-15% blocks, 25% warning, 40% critical).
59    #[must_use]
60    pub const fn content_site() -> Self {
61        Self {
62            target_class: TargetClass::ContentSite,
63            acceptable: 0.15,
64            warning: 0.25,
65            critical: 0.40,
66        }
67    }
68
69    /// Default SLOs for high-security sites (0-30% blocks, 50% warning, 70% critical).
70    #[must_use]
71    pub const fn high_security() -> Self {
72        Self {
73            target_class: TargetClass::HighSecurity,
74            acceptable: 0.30,
75            warning: 0.50,
76            critical: 0.70,
77        }
78    }
79
80    /// Default SLOs for unknown targets (conservative: API thresholds).
81    #[must_use]
82    pub const fn unknown() -> Self {
83        Self {
84            target_class: TargetClass::Unknown,
85            acceptable: 0.05, // Same as API
86            warning: 0.10,
87            critical: 0.15,
88        }
89    }
90
91    /// Get SLO for a target class.
92    #[must_use]
93    pub const fn for_class(class: TargetClass) -> Self {
94        match class {
95            TargetClass::Api => Self::api(),
96            TargetClass::ContentSite => Self::content_site(),
97            TargetClass::HighSecurity => Self::high_security(),
98            TargetClass::Unknown => Self::unknown(),
99        }
100    }
101
102    /// Assess blocked ratio against SLO thresholds.
103    ///
104    /// Returns `(is_acceptable, is_warning, is_critical)`.
105    #[must_use]
106    pub fn assess(&self, blocked_ratio: f64) -> (bool, bool, bool) {
107        (
108            blocked_ratio <= self.acceptable,
109            blocked_ratio > self.acceptable && blocked_ratio <= self.warning,
110            blocked_ratio > self.critical,
111        )
112    }
113}
114
115/// A simplified view of one HTTP transaction used for provider classification.
116#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
117pub struct TransactionView {
118    /// Request URL.
119    pub url: String,
120    /// HTTP status code.
121    pub status: u16,
122    /// Response headers (lower/upper case are normalized by the classifier).
123    pub response_headers: BTreeMap<String, String>,
124    /// Optional response body snippet.
125    pub response_body_snippet: Option<String>,
126}
127
128/// Known anti-bot providers recognized by the classifier.
129#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize)]
130pub enum AntiBotProvider {
131    /// `DataDome`.
132    DataDome,
133    /// Cloudflare bot/challenge stack.
134    Cloudflare,
135    /// Akamai bot manager indicators.
136    Akamai,
137    /// Human Security / `PerimeterX` indicators.
138    PerimeterX,
139    /// Kasada indicators.
140    Kasada,
141    /// Fingerprint.com markers.
142    FingerprintCom,
143    /// Catch-all when no provider-specific signatures were found.
144    Unknown,
145}
146
147/// Classification result with evidence markers.
148#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
149pub struct Detection {
150    /// Most likely provider.
151    pub provider: AntiBotProvider,
152    /// Simple confidence score in [0.0, 1.0].
153    pub confidence: f64,
154    /// Marker strings that matched.
155    pub markers: Vec<String>,
156}
157
158/// Scorecard for one provider.
159#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
160pub struct ProviderScore {
161    /// Provider represented by this score.
162    pub provider: AntiBotProvider,
163    /// Weighted score from marker matches.
164    pub score: u32,
165    /// Evidence used to produce the score.
166    pub markers: Vec<String>,
167}
168
169/// Minimal per-request summary extracted from a HAR file.
170#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
171pub struct HarRequestSummary {
172    /// URL requested.
173    pub url: String,
174    /// HTTP status code.
175    pub status: u16,
176    /// Best-effort resource type from HAR metadata.
177    pub resource_type: Option<String>,
178    /// Detection result for this request.
179    pub detection: Detection,
180}
181
182/// Full HAR classification report.
183#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
184pub struct HarClassificationReport {
185    /// URL/title from HAR page metadata when available.
186    pub page_title: Option<String>,
187    /// Summary classification for all entries.
188    pub aggregate: Detection,
189    /// Request-level classification outputs.
190    pub requests: Vec<HarRequestSummary>,
191}
192
193/// Frequency count for a normalized marker string.
194#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
195pub struct MarkerCount {
196    /// Marker text.
197    pub marker: String,
198    /// Number of requests where the marker appears.
199    pub count: u64,
200}
201
202/// Aggregated request metrics per host.
203#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
204pub struct HostSummary {
205    /// Hostname extracted from request URL.
206    pub host: String,
207    /// Total requests observed for this host.
208    pub total_requests: u64,
209    /// Requests that returned HTTP 403 or 429.
210    pub blocked_requests: u64,
211}
212
213/// Full-featured HAR investigation output suitable for diffs and alerting.
214#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
215pub struct InvestigationReport {
216    /// URL/title from HAR page metadata when available.
217    pub page_title: Option<String>,
218    /// Total requests in the capture.
219    pub total_requests: u64,
220    /// Count of blocked/challenged requests (403/429).
221    pub blocked_requests: u64,
222    /// Status-code histogram.
223    pub status_histogram: BTreeMap<u16, u64>,
224    /// Resource-type histogram from HAR metadata.
225    pub resource_type_histogram: BTreeMap<String, u64>,
226    /// Provider histogram inferred from signatures.
227    pub provider_histogram: BTreeMap<AntiBotProvider, u64>,
228    /// Full marker histogram inferred from signatures.
229    pub marker_histogram: BTreeMap<String, u64>,
230    /// Most frequent signature markers.
231    pub top_markers: Vec<MarkerCount>,
232    /// Top hosts by request volume.
233    pub hosts: Vec<HostSummary>,
234    /// Suspicious requests (blocked/challenged or with known provider markers).
235    pub suspicious_requests: Vec<HarRequestSummary>,
236    /// Aggregate provider classification.
237    pub aggregate: Detection,
238    /// Target website class for SLO assessment (optional; defaults to Unknown).
239    #[serde(default)]
240    pub target_class: Option<TargetClass>,
241}
242
243/// Delta between a baseline report and a candidate report.
244#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
245pub struct InvestigationDiff {
246    /// Baseline request count.
247    pub baseline_total_requests: u64,
248    /// Candidate request count.
249    pub candidate_total_requests: u64,
250    /// Baseline blocked requests.
251    pub baseline_blocked_requests: u64,
252    /// Candidate blocked requests.
253    pub candidate_blocked_requests: u64,
254    /// Candidate blocked ratio minus baseline blocked ratio.
255    pub blocked_ratio_delta: f64,
256    /// Whether blocked ratio increased by at least 2 percentage points.
257    pub likely_regression: bool,
258    /// Provider count delta: candidate minus baseline.
259    pub provider_delta: BTreeMap<AntiBotProvider, i64>,
260    /// New markers observed in candidate but not baseline.
261    pub new_markers: Vec<String>,
262}
263
264/// Severity/importance level for an inferred operational requirement.
265#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize)]
266pub enum RequirementLevel {
267    /// Helpful, but usually not mandatory.
268    Low,
269    /// Strongly recommended for reliable automation.
270    Medium,
271    /// Typically required to avoid frequent blocks/challenges.
272    High,
273}
274
275/// One inferred operational requirement derived from telemetry.
276#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
277pub struct AntiBotRequirement {
278    /// Stable identifier for the requirement.
279    pub id: String,
280    /// Human-friendly requirement title.
281    pub title: String,
282    /// Why this requirement appears to matter.
283    pub why: String,
284    /// Marker evidence or metrics supporting the inference.
285    pub evidence: Vec<String>,
286    /// Estimated requirement importance.
287    pub level: RequirementLevel,
288}
289
290/// High-level integration strategy for Stygian execution.
291#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
292pub enum AdapterStrategy {
293    /// Standard HTTP adapter path appears sufficient.
294    DirectHttp,
295    /// Browser-backed execution is recommended.
296    BrowserStealth,
297    /// Sticky session + proxy continuity should be applied.
298    StickyProxy,
299    /// Warm-up/session priming before data collection is advised.
300    SessionWarmup,
301    /// Unknown/ambiguous conditions: keep in investigation mode.
302    InvestigateOnly,
303}
304
305/// Suggested Stygian integration plan derived from investigation signals.
306#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
307pub struct IntegrationRecommendation {
308    /// Selected strategy.
309    pub strategy: AdapterStrategy,
310    /// Why this strategy was selected.
311    pub rationale: String,
312    /// Suggested feature flags/components for Stygian wiring.
313    pub required_stygian_features: Vec<String>,
314    /// Suggested runtime configuration hints.
315    pub config_hints: BTreeMap<String, String>,
316}
317
318/// Provider-aware operational profile and integration guidance.
319#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
320pub struct RequirementsProfile {
321    /// Aggregate inferred provider.
322    pub provider: AntiBotProvider,
323    /// Confidence for the provider assignment.
324    pub confidence: f64,
325    /// Inferred operational requirements.
326    pub requirements: Vec<AntiBotRequirement>,
327    /// Suggested Stygian integration strategy.
328    pub recommendation: IntegrationRecommendation,
329}
330
331/// High-level execution mode for a target.
332#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
333#[serde(rename_all = "snake_case")]
334pub enum ExecutionMode {
335    /// Standard HTTP adapters.
336    Http,
337    /// Browser-backed execution.
338    Browser,
339}
340
341/// Session persistence mode.
342#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
343#[serde(rename_all = "snake_case")]
344pub enum SessionMode {
345    /// No explicit session persistence.
346    Stateless,
347    /// Reuse a sticky proxy/session identity.
348    Sticky,
349}
350
351/// Recommended anti-bot telemetry level.
352#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
353#[serde(rename_all = "snake_case")]
354pub enum TelemetryLevel {
355    /// Minimal telemetry.
356    Basic,
357    /// Normal diagnostics.
358    Standard,
359    /// Deep diagnostics and marker tracking.
360    Deep,
361}
362
363/// Concrete runtime policy that can be mapped to Stygian config.
364#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
365pub struct RuntimePolicy {
366    /// Recommended execution mode.
367    pub execution_mode: ExecutionMode,
368    /// Recommended session mode.
369    pub session_mode: SessionMode,
370    /// Recommended telemetry level.
371    pub telemetry_level: TelemetryLevel,
372    /// Requests per second budget.
373    pub rate_limit_rps: f64,
374    /// Max retries per request.
375    pub max_retries: u32,
376    /// Baseline backoff in milliseconds.
377    pub backoff_base_ms: u64,
378    /// Whether warm-up navigation/requests are recommended.
379    pub enable_warmup: bool,
380    /// Whether browser context should block WebRTC non-proxied paths.
381    pub enforce_webrtc_proxy_only: bool,
382    /// Suggested sticky-session TTL in seconds (if relevant).
383    pub sticky_session_ttl_secs: Option<u64>,
384    /// Required Stygian features/components.
385    pub required_stygian_features: Vec<String>,
386    /// Additional hints mapped by key.
387    pub config_hints: BTreeMap<String, String>,
388    /// Composite risk score in [0.0, 1.0].
389    pub risk_score: f64,
390}
391
392/// End-to-end result from HAR analysis, requirements inference, and policy planning.
393#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
394pub struct InvestigationBundle {
395    /// Parsed/aggregated investigation report.
396    pub report: InvestigationReport,
397    /// Inferred requirements profile.
398    pub requirements: RequirementsProfile,
399    /// Planned runtime policy.
400    pub policy: RuntimePolicy,
401}
402
403#[cfg(test)]
404#[allow(
405    clippy::unwrap_used,
406    clippy::expect_used,
407    clippy::panic,
408    clippy::indexing_slicing
409)]
410mod tests {
411    use super::*;
412
413    #[test]
414    fn test_blocked_ratio_slo_api_thresholds() {
415        let slo = BlockedRatioSlo::api();
416        assert_eq!(slo.target_class, TargetClass::Api);
417        assert!((slo.acceptable - 0.05).abs() < f64::EPSILON);
418        assert!((slo.warning - 0.10).abs() < f64::EPSILON);
419        assert!((slo.critical - 0.15).abs() < f64::EPSILON);
420    }
421
422    #[test]
423    fn test_blocked_ratio_slo_content_site_thresholds() {
424        let slo = BlockedRatioSlo::content_site();
425        assert_eq!(slo.target_class, TargetClass::ContentSite);
426        assert!((slo.acceptable - 0.15).abs() < f64::EPSILON);
427        assert!((slo.warning - 0.25).abs() < f64::EPSILON);
428        assert!((slo.critical - 0.40).abs() < f64::EPSILON);
429    }
430
431    #[test]
432    fn test_blocked_ratio_slo_high_security_thresholds() {
433        let slo = BlockedRatioSlo::high_security();
434        assert_eq!(slo.target_class, TargetClass::HighSecurity);
435        assert!((slo.acceptable - 0.30).abs() < f64::EPSILON);
436        assert!((slo.warning - 0.50).abs() < f64::EPSILON);
437        assert!((slo.critical - 0.70).abs() < f64::EPSILON);
438    }
439
440    #[test]
441    fn test_blocked_ratio_slo_unknown_defaults_to_api() {
442        let slo = BlockedRatioSlo::unknown();
443        assert_eq!(slo.target_class, TargetClass::Unknown);
444        assert!((slo.acceptable - 0.05).abs() < f64::EPSILON); // Same thresholds as API
445        assert!((slo.warning - 0.10).abs() < f64::EPSILON);
446        assert!((slo.critical - 0.15).abs() < f64::EPSILON);
447    }
448
449    #[test]
450    fn test_blocked_ratio_slo_for_class_api() {
451        let slo = BlockedRatioSlo::for_class(TargetClass::Api);
452        assert_eq!(slo.target_class, TargetClass::Api);
453        assert!((slo.acceptable - 0.05).abs() < f64::EPSILON);
454    }
455
456    #[test]
457    fn test_blocked_ratio_slo_for_class_content_site() {
458        let slo = BlockedRatioSlo::for_class(TargetClass::ContentSite);
459        assert_eq!(slo.target_class, TargetClass::ContentSite);
460        assert!((slo.acceptable - 0.15).abs() < f64::EPSILON);
461    }
462
463    #[test]
464    fn test_blocked_ratio_slo_assess_below_acceptable() {
465        let slo = BlockedRatioSlo::api();
466        let (acceptable, warning, critical) = slo.assess(0.02);
467        assert!(acceptable);
468        assert!(!warning);
469        assert!(!critical);
470    }
471
472    #[test]
473    fn test_blocked_ratio_slo_assess_at_acceptable() {
474        let slo = BlockedRatioSlo::api();
475        let (acceptable, warning, critical) = slo.assess(0.05);
476        assert!(acceptable);
477        assert!(!warning);
478        assert!(!critical);
479    }
480
481    #[test]
482    fn test_blocked_ratio_slo_assess_in_warning_zone() {
483        let slo = BlockedRatioSlo::api();
484        let (acceptable, warning, critical) = slo.assess(0.075);
485        assert!(!acceptable);
486        assert!(warning);
487        assert!(!critical);
488    }
489
490    #[test]
491    fn test_blocked_ratio_slo_assess_at_warning() {
492        let slo = BlockedRatioSlo::api();
493        let (acceptable, warning, critical) = slo.assess(0.10);
494        // At exactly 0.10 (warning threshold), warning should be true
495        // because warning is true when > acceptable && <= warning
496        assert!(!acceptable);
497        assert!(warning); // 0.10 is in the warning zone (0.05 < 0.10 <= 0.10)
498        assert!(!critical);
499    }
500
501    #[test]
502    fn test_blocked_ratio_slo_assess_between_warning_and_critical() {
503        let slo = BlockedRatioSlo::api();
504        let (acceptable, warning, critical) = slo.assess(0.125);
505        assert!(!acceptable);
506        assert!(!warning);
507        assert!(!critical);
508    }
509
510    #[test]
511    fn test_blocked_ratio_slo_assess_above_critical() {
512        let slo = BlockedRatioSlo::api();
513        let (acceptable, warning, critical) = slo.assess(0.20);
514        assert!(!acceptable);
515        assert!(!warning);
516        assert!(critical);
517    }
518
519    #[test]
520    fn test_blocked_ratio_slo_content_site_assessment() {
521        let slo = BlockedRatioSlo::content_site();
522
523        // Below acceptable (green)
524        let (acc, warn, crit) = slo.assess(0.10);
525        assert!(acc && !warn && !crit);
526
527        // In warning zone (yellow): 0.15 < 0.20 <= 0.25
528        let (acc, warn, crit) = slo.assess(0.20);
529        assert!(!acc && warn && !crit);
530
531        // In critical zone: 0.45 > 0.40
532        let (acc, warn, crit) = slo.assess(0.45);
533        assert!(!acc && !warn && crit);
534
535        // Exactly at critical threshold
536        let (acc, warn, crit) = slo.assess(0.40);
537        assert!(!acc && !warn && !crit); // Exactly at threshold is not > critical
538    }
539
540    #[test]
541    fn test_target_class_derives() {
542        // Verify that TargetClass can be compared and hashed
543        let api1 = TargetClass::Api;
544        let api2 = TargetClass::Api;
545        let content = TargetClass::ContentSite;
546
547        assert_eq!(api1, api2);
548        assert_ne!(api1, content);
549    }
550
551    #[test]
552    fn test_blocked_ratio_slo_serialization() {
553        let slo = BlockedRatioSlo::content_site();
554        let json = serde_json::to_string(&slo).unwrap_or_default();
555        if let Ok(deserialized) = serde_json::from_str::<BlockedRatioSlo>(&json) {
556            assert_eq!(slo, deserialized);
557        }
558    }
559
560    #[test]
561    fn test_target_class_serialization() {
562        let target = TargetClass::HighSecurity;
563        let json = serde_json::to_string(&target).unwrap_or_default();
564        if let Ok(deserialized) = serde_json::from_str::<TargetClass>(&json) {
565            assert_eq!(target, deserialized);
566        }
567    }
568}