Skip to main content

hpr_validate/
report.rs

1//! Reports: one line per metric, in Markdown for people and JSON for machines.
2
3use serde::{Deserialize, Serialize};
4
5use crate::case::Tolerance;
6
7/// How one metric came out.
8#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
9#[serde(rename_all = "snake_case")]
10#[non_exhaustive]
11pub enum Verdict {
12    /// Inside its tolerance.
13    Pass,
14    /// Outside it.
15    Fail,
16    /// Measured and printed, but held to no tolerance, for the written reason beside it. It is
17    /// not a pass: the summary counts it apart, and it never makes a run green on its own.
18    NotScored,
19    /// Predicted mode ([`crate::DragMode::Predicted`]): inside the target its case sets. A target
20    /// is reported, not gated, so this counts neither as a pass nor towards the verdict.
21    WithinTarget,
22    /// Predicted mode: outside the target. Reported, with the reason in the case file, and it
23    /// never fails the run: neither code's drag is the truth there.
24    OutsideTarget,
25}
26
27/// One metric of one case: what the oracle said, what hpr said, and whether that is close enough.
28#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
29pub struct Comparison {
30    /// The case's id.
31    pub case: String,
32    /// The metric's name.
33    pub metric: String,
34    /// What hpr measured.
35    pub measured: f64,
36    /// What the reference says.
37    pub reference: f64,
38    /// Where the reference value came from.
39    pub source: String,
40    /// The tolerance it was held to, if it was held to one.
41    pub tolerance: Tolerance,
42    /// `measured − reference`.
43    pub difference: f64,
44    /// That difference as a fraction of the reference, where the reference is not zero.
45    pub relative: Option<f64>,
46    /// Pass, fail, not scored, or within or outside a predicted-mode target.
47    pub verdict: Verdict,
48    /// Why it is not scored, for a [`Verdict::NotScored`] row.
49    #[serde(default, skip_serializing_if = "Option::is_none")]
50    pub note: Option<String>,
51}
52
53impl Comparison {
54    /// Compares `measured` against `reference` under `tolerance`.
55    #[must_use]
56    pub fn new(
57        case: &str,
58        metric: &str,
59        measured: f64,
60        reference: f64,
61        source: &str,
62        tolerance: Tolerance,
63    ) -> Self {
64        let verdict = if tolerance.accepts(measured, reference) {
65            Verdict::Pass
66        } else {
67            Verdict::Fail
68        };
69        Self::build(
70            case, metric, measured, reference, source, tolerance, verdict, None,
71        )
72    }
73
74    /// Compares `measured` against `reference` under `tolerance` as a target, not a gate: predicted
75    /// mode's rows, which are reported and never fail a run for being outside it.
76    ///
77    /// A row with no bound, or a number that is not finite, is not a miss to explain but a broken
78    /// comparison, so it fails the run, as a blank reason for not scoring does.
79    #[must_use]
80    pub fn targeted(
81        case: &str,
82        metric: &str,
83        measured: f64,
84        reference: f64,
85        source: &str,
86        tolerance: Tolerance,
87    ) -> Self {
88        let verdict = if !tolerance.is_set() || !measured.is_finite() || !reference.is_finite() {
89            Verdict::Fail
90        } else if tolerance.accepts(measured, reference) {
91            Verdict::WithinTarget
92        } else {
93            Verdict::OutsideTarget
94        };
95        Self::build(
96            case, metric, measured, reference, source, tolerance, verdict, None,
97        )
98    }
99
100    /// Records `measured` against `reference` without scoring it, for the written `reason`.
101    #[must_use]
102    pub fn not_scored(
103        case: &str,
104        metric: &str,
105        measured: f64,
106        reference: f64,
107        source: &str,
108        reason: &str,
109    ) -> Self {
110        let reason = reason.trim();
111        Self::build(
112            case,
113            metric,
114            measured,
115            reference,
116            source,
117            Tolerance::default(),
118            if reason.is_empty() {
119                // An excuse nobody wrote down is the thing this mechanism exists to prevent, so a
120                // blank one fails rather than quietly not counting.
121                Verdict::Fail
122            } else {
123                Verdict::NotScored
124            },
125            (!reason.is_empty()).then(|| reason.to_owned()),
126        )
127    }
128
129    /// The common part of both.
130    #[expect(
131        clippy::too_many_arguments,
132        reason = "a report row has this many columns"
133    )]
134    fn build(
135        case: &str,
136        metric: &str,
137        measured: f64,
138        reference: f64,
139        source: &str,
140        tolerance: Tolerance,
141        verdict: Verdict,
142        note: Option<String>,
143    ) -> Self {
144        let difference = measured - reference;
145        Self {
146            case: case.to_owned(),
147            metric: metric.to_owned(),
148            measured,
149            reference,
150            source: source.to_owned(),
151            tolerance,
152            difference,
153            relative: (reference != 0.0).then(|| difference / reference),
154            verdict,
155            note,
156        }
157    }
158
159    /// Whether this row counts towards the suite's verdict.
160    #[must_use]
161    pub fn scored(&self) -> bool {
162        matches!(self.verdict, Verdict::Pass | Verdict::Fail)
163    }
164
165    /// Whether this is a predicted-mode row, held to a target rather than a gate.
166    #[must_use]
167    pub fn targeted_row(&self) -> bool {
168        matches!(self.verdict, Verdict::WithinTarget | Verdict::OutsideTarget)
169    }
170}
171
172/// A case that ran into a limit of hpr's it declares in writing ([`crate::Case::known_gap`]): it
173/// was flown, hpr refused it as the case said it would, and none of its metrics was scored.
174///
175/// It is not a pass and not a quiet omission. The report prints it in a section of its own, with
176/// the case's reason and hpr's refusal, and the summary counts it apart.
177#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
178#[non_exhaustive]
179pub struct Gap {
180    /// The case's id.
181    pub case: String,
182    /// The case's written reason.
183    pub reason: String,
184    /// What hpr said when it refused the flight.
185    pub refusal: String,
186    /// The Mach number hpr refused.
187    pub mach: f64,
188    /// How many metrics the case would have scored, with the tolerances they will be held to.
189    pub metric_count: usize,
190}
191
192/// Where a case's reference came from.
193#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
194pub struct Source {
195    /// The case's id.
196    pub case: String,
197    /// The oracle, as its generator records it.
198    pub oracle: String,
199    /// The generator script.
200    pub generator: String,
201    /// The command that regenerates the reference.
202    pub command: String,
203    /// The reference file, relative to the repository root.
204    pub file: String,
205    /// Its SHA-256, so an edited reference shows up in the report and not only in git history.
206    pub sha256: String,
207    /// What the oracle modeled, in its generator's words.
208    pub model: String,
209    /// What the generator had to override to make the comparison like-for-like.
210    pub overrides: String,
211}
212
213/// A whole run: every comparison, and what it adds up to.
214#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
215pub struct Report {
216    /// The harness's version.
217    pub harness_version: String,
218    /// Whether the run was `--fast`.
219    pub fast: bool,
220    /// The cases it covered, in report order.
221    pub cases: Vec<String>,
222    /// The locked cases it left out, which only a fast run has.
223    #[serde(default)]
224    pub skipped: Vec<String>,
225    /// Every metric compared.
226    pub comparisons: Vec<Comparison>,
227    /// The cases that ran into a limit of hpr's they declare, in case order.
228    #[serde(default, skip_serializing_if = "Vec::is_empty")]
229    pub gaps: Vec<Gap>,
230    /// Where each case's reference came from, in case order.
231    pub sources: Vec<Source>,
232}
233
234impl Report {
235    /// Whether every scored comparison passed.
236    #[must_use]
237    pub fn passed(&self) -> bool {
238        self.failures().is_empty()
239    }
240
241    /// The comparisons that failed.
242    #[must_use]
243    pub fn failures(&self) -> Vec<&Comparison> {
244        self.comparisons
245            .iter()
246            .filter(|comparison| comparison.verdict == Verdict::Fail)
247            .collect()
248    }
249
250    /// The comparisons that were measured but held to no tolerance.
251    #[must_use]
252    pub fn not_scored(&self) -> Vec<&Comparison> {
253        self.comparisons
254            .iter()
255            .filter(|comparison| comparison.verdict == Verdict::NotScored)
256            .collect()
257    }
258
259    /// The largest relative difference among the scored comparisons, as a fraction.
260    #[must_use]
261    pub fn worst_scored(&self) -> Option<(&Comparison, f64)> {
262        self.comparisons
263            .iter()
264            .filter(|comparison| comparison.scored())
265            .filter_map(|comparison| comparison.relative.map(|value| (comparison, value.abs())))
266            .max_by(|(_, a), (_, b)| a.total_cmp(b))
267    }
268
269    /// Whether this run reproduces a committed report, given the text of its `latest.md` and
270    /// `latest.json`.
271    ///
272    /// hpr is bit-identical on one platform, not across three ([ADR-015][adr-015], the validation
273    /// harness's decisions), so the numbers are compared at full precision from the JSON, where
274    /// hpr's value and the reference's may each differ from the committed one by 2e-6 or by 1e-7
275    /// of itself, whichever is larger. Everything else is compared exactly: the harness version,
276    /// the cases, what was left out, the sources, each comparison's case, metric, source,
277    /// tolerance, verdict and note, and each gap but for the Mach number the integrator narrowed
278    /// onto 1. The two committed files come from one run on one platform, so `latest.md` must be
279    /// `latest.json`'s rendering letter for letter, and each committed difference must be its
280    /// own two values' difference. A tighter check would assert a cross-platform bit-identity hpr
281    /// does not claim; comparing the rendered Markdown instead would fail on a platform whose last
282    /// digit rounds a printed percentage the other way.
283    ///
284    /// # Errors
285    ///
286    /// [`NotReproduced`], naming the first difference found, or why the committed JSON does not
287    /// parse.
288    ///
289    /// [adr-015]: https://github.com/nrdptel/hpr-sim/blob/main/docs/DECISIONS.md#adr-015-the-validation-harness-cases-references-tolerances-and-reports-2026-09-17
290    pub fn reproduces(
291        &self,
292        committed_markdown: &str,
293        committed_json: &str,
294    ) -> Result<(), NotReproduced> {
295        let differs = |what: String| Err(NotReproduced(format!("latest.json: {what}")));
296        let committed: Self = serde_json::from_str(committed_json)
297            .map_err(|error| NotReproduced(format!("latest.json does not parse: {error}")))?;
298        if let Some((line, (file, rendered))) = committed_markdown
299            .lines()
300            .map(Some)
301            .chain(std::iter::repeat(None))
302            .zip(
303                committed
304                    .to_markdown()
305                    .lines()
306                    .map(Some)
307                    .chain(std::iter::repeat(None)),
308            )
309            .take_while(|pair| *pair != (None, None))
310            .enumerate()
311            .find(|(_, (file, rendered))| file != rendered)
312        {
313            return Err(NotReproduced(format!(
314                "latest.md is not latest.json's rendering at line {}:\n  latest.md:   {}\n  \
315                 latest.json: {}",
316                line + 1,
317                file.unwrap_or("(the end)"),
318                rendered.unwrap_or("(the end)")
319            )));
320        }
321        if committed.harness_version != self.harness_version {
322            return differs("the harness versions differ".to_owned());
323        }
324        if committed.fast != self.fast || committed.skipped != self.skipped {
325            return differs("the cases left out differ".to_owned());
326        }
327        if committed.cases != self.cases {
328            return differs("the cases differ".to_owned());
329        }
330        // A gap's Mach number is where the integrator narrowed onto 1, to the last bits of which
331        // the platforms need not agree; everything else about it must.
332        let gap_shape = |report: &Self| {
333            report
334                .gaps
335                .iter()
336                .map(|gap| {
337                    (
338                        gap.case.clone(),
339                        gap.reason.clone(),
340                        gap.refusal.clone(),
341                        gap.metric_count,
342                        (gap.mach - 1.0).abs() < 1e-9,
343                    )
344                })
345                .collect::<Vec<_>>()
346        };
347        if gap_shape(&committed) != gap_shape(self) {
348            return differs("the known gaps differ".to_owned());
349        }
350        if committed.sources != self.sources {
351            return differs("the sources differ".to_owned());
352        }
353        let shape = |report: &Self| {
354            report
355                .comparisons
356                .iter()
357                .map(|comparison| {
358                    (
359                        comparison.case.clone(),
360                        comparison.metric.clone(),
361                        comparison.source.clone(),
362                        comparison.tolerance,
363                        comparison.verdict,
364                        comparison.note.clone(),
365                    )
366                })
367                .collect::<Vec<_>>()
368        };
369        if shape(&committed) != shape(self) {
370            return differs(
371                "the comparisons' metrics, sources, tolerances, verdicts or notes differ"
372                    .to_owned(),
373            );
374        }
375        for (old, new) in committed.comparisons.iter().zip(&self.comparisons) {
376            let row = format!("{}'s {}", old.case, old.metric);
377            // Derived on the committing platform by the formula `build` uses, so exactly.
378            if old.difference.to_bits() != (old.measured - old.reference).to_bits()
379                || old.relative.map(f64::to_bits)
380                    != (old.reference != 0.0).then(|| (old.difference / old.reference).to_bits())
381            {
382                return differs(format!(
383                    "{row}'s difference is not its own values' difference"
384                ));
385            }
386            for (what, was, is) in [
387                ("hpr value", old.measured, new.measured),
388                ("reference value", old.reference, new.reference),
389            ] {
390                if !same_but_for_platform_rounding(was, is) {
391                    return differs(format!(
392                        "{row}'s {what} is {was} there and {is} in this run"
393                    ));
394                }
395            }
396        }
397        Ok(())
398    }
399
400    /// The report as Markdown: a summary, one table row per gated or not-scored metric, then
401    /// predicted mode's rows in their own table.
402    ///
403    /// It carries no date, so a run that changes nothing changes no bytes and the committed report
404    /// only moves when a number does.
405    #[must_use]
406    pub fn to_markdown(&self) -> String {
407        let scored = self.comparisons.iter().filter(|c| c.scored()).count();
408        let mut out = String::new();
409        out.push_str("# Validation report\n\n");
410        out.push_str(&format!(
411            "hpr-validate {}, {} case(s), {} metric(s): {}.\n\n",
412            self.harness_version,
413            self.cases.len(),
414            self.comparisons.len(),
415            if self.passed() {
416                format!("{scored} scored, all within tolerance")
417            } else {
418                format!(
419                    "{scored} scored, {} OUT OF TOLERANCE",
420                    self.failures().len()
421                )
422            }
423        ));
424        if let Some((worst, relative)) = self.worst_scored() {
425            out.push_str(&format!(
426                "The largest scored difference is {}'s {} at {:+.3}%, against a {} gate.\n\n",
427                worst.case,
428                worst.metric,
429                100.0 * worst.relative.unwrap_or(relative),
430                worst.tolerance.describe()
431            ));
432        }
433        if !self.not_scored().is_empty() {
434            out.push_str(&format!(
435                "{} metric(s) are measured and printed but held to no tolerance; each says why in \
436                 its row, and none of them counts towards the verdict.\n\n",
437                self.not_scored().len()
438            ));
439        }
440        let targeted: Vec<&Comparison> = self
441            .comparisons
442            .iter()
443            .filter(|comparison| comparison.targeted_row())
444            .collect();
445        if !targeted.is_empty() {
446            out.push_str(&format!(
447                "{} predicted-mode metric(s) are reported against a target, not a gate, {} of them \
448                 within it; they are under Predicted mode, below the table, and none counts \
449                 towards the verdict.\n\n",
450                targeted.len(),
451                targeted
452                    .iter()
453                    .filter(|comparison| comparison.verdict == Verdict::WithinTarget)
454                    .count()
455            ));
456        }
457        if !self.gaps.is_empty() {
458            out.push_str(&format!(
459                "{} case(s) are known gaps: flown, refused by hpr for the reason the case gives, \
460                 and not scored. They are listed under Known gaps, below the table.\n\n",
461                self.gaps.len()
462            ));
463        }
464        if self.fast {
465            out.push_str(&if self.skipped.is_empty() {
466                "Run with `--fast`. The lock marks no case slow, so nothing was left out.\n\n"
467                    .to_owned()
468            } else {
469                format!(
470                    "Run with `--fast`, which left out {} of the locked cases: {}.\n\n",
471                    self.skipped.len(),
472                    self.skipped.join(", ")
473                )
474            });
475        }
476        out.push_str(
477            "Generated by `cargo xtask validate`. Every reference value names its own \
478                      source; the generators are under `validation/oracles/`.\n\n",
479        );
480        table(
481            &mut out,
482            "tolerance",
483            self.comparisons.iter().filter(|c| !c.targeted_row()),
484        );
485        if !targeted.is_empty() {
486            out.push_str(
487                "\n## Predicted mode\n\nhpr flies its own aerodynamics; the reference is RocketPy \
488                 flying each example's own drag. So a difference here is mostly the two drags, and \
489                 neither is the truth. Each metric is held to a target, reported and never gated, \
490                 and each case file explains its misses.\n\n",
491            );
492            table(&mut out, "target", targeted.into_iter());
493        }
494        if !self.gaps.is_empty() {
495            out.push_str("\n## Known gaps\n\n");
496            for gap in &self.gaps {
497                out.push_str(&format!(
498                    "- **{}**: {} metric(s), none scored. {} hpr: {}\n",
499                    gap.case, gap.metric_count, gap.reason, gap.refusal
500                ));
501            }
502        }
503        out.push_str("\n## References\n\n");
504        for source in &self.sources {
505            out.push_str(&format!(
506                "- **{}**: {}, `{}` (sha256 `{}`).\n  - Oracle: {}\n  - Overrides: {}\n",
507                source.case,
508                source.oracle,
509                source.file,
510                source.sha256.get(..16).unwrap_or(&source.sha256),
511                source.model,
512                source.overrides
513            ));
514        }
515        // Each reference's own command, once, in the order the cases first name it.
516        let mut commands: Vec<&str> = Vec::new();
517        for source in &self.sources {
518            if !commands.contains(&source.command.as_str()) {
519                commands.push(&source.command);
520            }
521        }
522        if !commands.is_empty() {
523            out.push_str(&format!(
524                "\nRegenerate with {}, after `cargo xtask refs fetch` has put the oracle in the \
525                 gitignored `refs/`. A reference moves only when its generator runs, which is a \
526                 deliberate step: it is never regenerated to make a comparison pass (Loft lesson \
527                 L76).\n",
528                commands
529                    .iter()
530                    .map(|command| format!("`{command}`"))
531                    .collect::<Vec<_>>()
532                    .join(" and ")
533            ));
534        }
535        out
536    }
537}
538
539/// Writes one table of comparisons, the column after the difference headed `bound`.
540fn table<'a>(out: &mut String, bound: &str, comparisons: impl Iterator<Item = &'a Comparison>) {
541    out.push_str(&format!(
542        "| case | metric | hpr | reference | difference | {bound} | verdict | note |\n"
543    ));
544    out.push_str("|---|---|---|---|---|---|---|---|\n");
545    for comparison in comparisons {
546        let relative = comparison.relative.map_or_else(
547            || format!("{:+.6}", comparison.difference),
548            |relative| format!("{:+.3}%", 100.0 * relative),
549        );
550        out.push_str(&format!(
551            "| {} | {} | {:.6} | {:.6} | {} | {} | {} | {} |\n",
552            comparison.case,
553            comparison.metric,
554            comparison.measured,
555            comparison.reference,
556            relative,
557            comparison.tolerance.describe(),
558            match comparison.verdict {
559                Verdict::Pass => "pass",
560                Verdict::Fail => "**fail**",
561                Verdict::NotScored => "not scored",
562                Verdict::WithinTarget => "within target",
563                Verdict::OutsideTarget => "outside target",
564            },
565            comparison.note.as_deref().unwrap_or("")
566        ));
567    }
568}
569
570/// Whether a committed number and this run's are the same but for the last digits the platforms
571/// round differently: within 2e-6, or 1e-7 of this run's value, whichever is larger.
572///
573/// hpr is bit-identical on one platform, not across three (ADR-015). The descents reproduce to
574/// about 1e-12; a whole flight does to about 1e-8 of each value: NDRT 2020's landing drift is
575/// 354.240893 m on macOS and 354.240895 m on Linux, after 84 s of six-degree-of-freedom flight in
576/// a sheared wind.
577pub(crate) fn same_but_for_platform_rounding(committed: f64, computed: f64) -> bool {
578    (committed - computed).abs() <= (2e-6_f64).max(1e-7 * computed.abs())
579}
580
581/// Why a run does not reproduce a committed report ([`Report::reproduces`]): the first difference
582/// found, in words.
583#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
584#[error("{0}")]
585pub struct NotReproduced(pub String);