Skip to main content

hpr_io/netcdf/
mod.rs

1//! netCDF classic files: the classic (`CDF\x01`) and 64-bit offset (`CDF\x02`) formats, read
2//! from bytes.
3//!
4//! **Source.** Unidata, *NetCDF File Format Specifications*, "The Classic Format" and "The
5//! 64-bit Offset Format" (netCDF-C documentation, captured 2026-09-26), whose grammar this reader
6//! follows: a header of dimensions, global attributes and variables, then each non-record
7//! variable's values contiguously at its `begin` offset, then the records, each holding one slab of
8//! every record variable. Every number is big-endian; names and short values pad to four bytes.
9//! The two formats differ only in the width of `begin`, 32 or 64 bits.
10//!
11//! **Packed data.** [`Variable::packing`] reads the attribute conventions of the netCDF Users
12//! Guide and CF Conventions 1.11 §8.1 ("Packed Data") and §2.5.1 ("Missing data, valid and actual
13//! range of data"): a stored value `p` means `p · scale_factor + add_offset`, and is missing if it
14//! equals `_FillValue` or `missing_value` or lies outside `valid_min`, `valid_max` or
15//! `valid_range`. With no valid bound given, the fill bounds the valid range on its own side, as
16//! the Users Guide says: the valid range ends one step (integers) or two units in the last place
17//! (floats) short of the fill, toward zero, so with a fill of −32767 a stored −32768 is missing
18//! too. A byte variable with no `_FillValue` has no fill. netCDF4-python 1.7.4 masks only values equal to the
19//! fill, and masks the byte default; the tests pin each value where the two differ.
20//!
21//! **Not read.** netCDF-4 files are HDF5 underneath (they begin `\x89HDF`) and the CDF-5 format
22//! (`CDF\x05`) is not in the cited specification; both are refused with the conversion that makes
23//! them readable ([`CONVERSION`]). A file whose record count was never written (`numrecs` "streaming") is
24//! refused, as the specification leaves it unimplemented.
25
26use std::collections::BTreeSet;
27use std::sync::Arc;
28
29use serde::{Deserialize, Serialize};
30use thiserror::Error;
31
32/// Header tag of a dimension list.
33const NC_DIMENSION: u32 = 0x0A;
34/// Header tag of a variable list.
35const NC_VARIABLE: u32 = 0x0B;
36/// Header tag of an attribute list.
37const NC_ATTRIBUTE: u32 = 0x0C;
38/// `numrecs` of a file whose record count was never written.
39const STREAMING: u32 = 0xFFFF_FFFF;
40
41/// How to rewrite a netCDF-4 or CDF-5 file as a 64-bit offset file, which this reader reads: with
42/// Python's xarray, dropping any variable the classic formats cannot hold (64-bit integers other
43/// than times, strings). ERA5 files from the Climate Data Store carry two, `number` and `expver`.
44pub const CONVERSION: &str = "xarray.open_dataset(\"in.nc\").drop_vars([\"number\", \"expver\"], errors=\"ignore\").to_netcdf(\"out.nc\", format=\"NETCDF3_64BIT\")";
45
46/// Why a netCDF file could not be read.
47#[derive(Debug, Clone, PartialEq, Eq, Error)]
48#[non_exhaustive]
49pub enum NetCdfError {
50    /// The bytes are an HDF5 file, as netCDF-4 files are.
51    #[error(
52        "this is a netCDF-4 (HDF5) file, which is not read; rewrite it as netCDF classic, for example with `{CONVERSION}`"
53    )]
54    NetCdf4,
55    /// The bytes are the CDF-5 (64-bit data) format.
56    #[error(
57        "this is a CDF-5 (64-bit data) netCDF file, which is not read; rewrite it as netCDF classic, for example with `{CONVERSION}`"
58    )]
59    Cdf5,
60    /// The bytes begin with neither a netCDF nor an HDF5 signature.
61    #[error("not a netCDF file: it begins with {head}")]
62    NotNetCdf {
63        /// The first bytes, printed as hex.
64        head: String,
65    },
66    /// The header or a variable's values run past the end of the bytes.
67    #[error("the file ends early: {what} needs bytes {start}..{end} of {len}")]
68    Truncated {
69        /// What was being read.
70        what: String,
71        /// First byte needed.
72        start: u64,
73        /// One past the last byte needed.
74        end: u64,
75        /// The number of bytes there are.
76        len: u64,
77    },
78    /// The header breaks the specification's grammar.
79    #[error("malformed header: {reason}")]
80    Malformed {
81        /// What is wrong.
82        reason: String,
83    },
84    /// The record count was never written.
85    #[error(
86        "the record count is \"streaming\" (never written), which the specification leaves unimplemented"
87    )]
88    Streaming,
89    /// A variable's attributes use a convention this reader does not apply.
90    #[error("variable `{variable}`: {reason}")]
91    Unsupported {
92        /// The variable.
93        variable: String,
94        /// What is not supported.
95        reason: String,
96    },
97}
98
99fn malformed(reason: impl Into<String>) -> NetCdfError {
100    NetCdfError::Malformed {
101        reason: reason.into(),
102    }
103}
104
105/// Which of the two classic formats a file is written in.
106#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
107#[non_exhaustive]
108pub enum Format {
109    /// `CDF\x01`: offsets are 32 bits.
110    Classic,
111    /// `CDF\x02`: offsets are 64 bits.
112    Offset64,
113}
114
115/// A netCDF external type.
116#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
117#[non_exhaustive]
118pub enum Type {
119    /// `NC_BYTE`: 8-bit signed integers.
120    Byte,
121    /// `NC_CHAR`: 8-bit characters.
122    Char,
123    /// `NC_SHORT`: 16-bit signed integers.
124    Short,
125    /// `NC_INT`: 32-bit signed integers.
126    Int,
127    /// `NC_FLOAT`: IEEE single precision.
128    Float,
129    /// `NC_DOUBLE`: IEEE double precision.
130    Double,
131}
132
133impl Type {
134    fn from_tag(tag: u32) -> Result<Self, NetCdfError> {
135        Ok(match tag {
136            1 => Type::Byte,
137            2 => Type::Char,
138            3 => Type::Short,
139            4 => Type::Int,
140            5 => Type::Float,
141            6 => Type::Double,
142            other => return Err(malformed(format!("unknown type tag {other}"))),
143        })
144    }
145
146    /// Bytes per value.
147    pub fn size(self) -> u64 {
148        match self {
149            Type::Byte | Type::Char => 1,
150            Type::Short => 2,
151            Type::Int | Type::Float => 4,
152            Type::Double => 8,
153        }
154    }
155}
156
157/// A block of values of one type, in row-major order.
158#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
159#[non_exhaustive]
160pub enum Values {
161    /// `NC_BYTE`, read as signed (the specification's default).
162    Byte(Vec<i8>),
163    /// `NC_CHAR`, as the bytes stored; the specification leaves their encoding open.
164    Char(Vec<u8>),
165    /// `NC_SHORT`.
166    Short(Vec<i16>),
167    /// `NC_INT`.
168    Int(Vec<i32>),
169    /// `NC_FLOAT`.
170    Float(Vec<f32>),
171    /// `NC_DOUBLE`.
172    Double(Vec<f64>),
173}
174
175impl Values {
176    fn decode(kind: Type, bytes: &[u8]) -> Values {
177        match kind {
178            Type::Byte => Values::Byte(bytes.iter().map(|&b| i8::from_be_bytes([b])).collect()),
179            Type::Char => Values::Char(bytes.to_vec()),
180            Type::Short => Values::Short(
181                bytes
182                    .as_chunks::<2>()
183                    .0
184                    .iter()
185                    .map(|&c| i16::from_be_bytes(c))
186                    .collect(),
187            ),
188            Type::Int => Values::Int(
189                bytes
190                    .as_chunks::<4>()
191                    .0
192                    .iter()
193                    .map(|&c| i32::from_be_bytes(c))
194                    .collect(),
195            ),
196            Type::Float => Values::Float(
197                bytes
198                    .as_chunks::<4>()
199                    .0
200                    .iter()
201                    .map(|&c| f32::from_be_bytes(c))
202                    .collect(),
203            ),
204            Type::Double => Values::Double(
205                bytes
206                    .as_chunks::<8>()
207                    .0
208                    .iter()
209                    .map(|&c| f64::from_be_bytes(c))
210                    .collect(),
211            ),
212        }
213    }
214
215    /// The type of the values.
216    pub fn kind(&self) -> Type {
217        match self {
218            Values::Byte(_) => Type::Byte,
219            Values::Char(_) => Type::Char,
220            Values::Short(_) => Type::Short,
221            Values::Int(_) => Type::Int,
222            Values::Float(_) => Type::Float,
223            Values::Double(_) => Type::Double,
224        }
225    }
226
227    /// The number of values.
228    pub fn len(&self) -> usize {
229        match self {
230            Values::Byte(v) => v.len(),
231            Values::Char(v) => v.len(),
232            Values::Short(v) => v.len(),
233            Values::Int(v) => v.len(),
234            Values::Float(v) => v.len(),
235            Values::Double(v) => v.len(),
236        }
237    }
238
239    /// Whether there are no values.
240    pub fn is_empty(&self) -> bool {
241        self.len() == 0
242    }
243
244    /// Value `index` as a number, exactly (every type converts to `f64` without rounding); `None`
245    /// past the end or for characters.
246    pub fn get(&self, index: usize) -> Option<f64> {
247        match self {
248            Values::Byte(v) => v.get(index).map(|&x| f64::from(x)),
249            Values::Char(_) => None,
250            Values::Short(v) => v.get(index).map(|&x| f64::from(x)),
251            Values::Int(v) => v.get(index).map(|&x| f64::from(x)),
252            Values::Float(v) => v.get(index).map(|&x| f64::from(x)),
253            Values::Double(v) => v.get(index).copied(),
254        }
255    }
256
257    /// Characters as text: `None` unless they are UTF-8. Trailing NUL bytes, which some writers
258    /// store after an attribute's text, are left off.
259    pub fn text(&self) -> Option<&str> {
260        match self {
261            Values::Char(bytes) => {
262                let end = bytes.iter().rposition(|&b| b != 0).map_or(0, |i| i + 1);
263                std::str::from_utf8(&bytes[..end]).ok()
264            }
265            _ => None,
266        }
267    }
268}
269
270/// A dimension.
271#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
272pub struct Dimension {
273    /// Its name.
274    pub name: String,
275    /// Its length; for the record dimension, the number of records.
276    pub length: u64,
277    /// Whether it is the record (unlimited) dimension.
278    pub is_record: bool,
279}
280
281/// A named attribute of the file or of a variable.
282#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
283pub struct Attribute {
284    /// Its name.
285    pub name: String,
286    /// Its values.
287    pub values: Values,
288}
289
290/// A variable, with all its values read.
291#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
292pub struct Variable {
293    /// Its name.
294    pub name: String,
295    /// Its dimensions' names, outermost first. Each is shared with every other variable along that
296    /// dimension, so a header that names one dimension many times costs no copy per axis.
297    pub dimensions: Vec<Arc<str>>,
298    /// Its length along each dimension (the record count for the record dimension).
299    pub shape: Vec<u64>,
300    /// Its attributes.
301    pub attributes: Vec<Attribute>,
302    /// Its values, in row-major order (last dimension varying fastest).
303    pub values: Values,
304}
305
306impl Variable {
307    /// The attribute called `name`.
308    pub fn attribute(&self, name: &str) -> Option<&Attribute> {
309        self.attributes.iter().find(|a| a.name == name)
310    }
311
312    /// The row-major position of the value at `index` (one entry per dimension), or `None` if the
313    /// index has the wrong rank or is out of range.
314    pub fn offset(&self, index: &[u64]) -> Option<usize> {
315        if index.len() != self.shape.len() {
316            return None;
317        }
318        let mut offset: u64 = 0;
319        for (&i, &n) in index.iter().zip(&self.shape) {
320            if i >= n {
321                return None;
322            }
323            offset = offset.checked_mul(n)?.checked_add(i)?;
324        }
325        usize::try_from(offset).ok()
326    }
327
328    /// How this variable's stored values map to physical ones, from its attributes (see the
329    /// module documentation).
330    ///
331    /// # Errors
332    ///
333    /// [`NetCdfError::Unsupported`] for character data, an `_Unsigned` attribute (whose values
334    /// this reader would read as signed), a packing or bound attribute that is not a single
335    /// number (two for `valid_range`), or `valid_range` given with `valid_min` or `valid_max`.
336    pub fn packing(&self) -> Result<Packing, NetCdfError> {
337        let unsupported = |reason: String| NetCdfError::Unsupported {
338            variable: self.name.clone(),
339            reason,
340        };
341        let own = self.values.kind();
342        if own == Type::Char {
343            return Err(unsupported("character data has no numeric values".into()));
344        }
345        if self.attribute("_Unsigned").is_some() {
346            return Err(unsupported("`_Unsigned` data is not read".into()));
347        }
348        let numbers = |name: &str| -> Result<Option<Vec<f64>>, NetCdfError> {
349            match self.attribute(name) {
350                None => Ok(None),
351                Some(a) if a.values.kind() == Type::Char => {
352                    Err(unsupported(format!("`{name}` is text")))
353                }
354                Some(a) => Ok(Some(
355                    (0..a.values.len())
356                        .filter_map(|i| a.values.get(i))
357                        .collect(),
358                )),
359            }
360        };
361        let single = |name: &str| -> Result<Option<f64>, NetCdfError> {
362            match numbers(name)? {
363                None => Ok(None),
364                Some(v) if v.len() == 1 => Ok(Some(v[0])),
365                Some(_) => Err(unsupported(format!("`{name}` is not a single number"))),
366            }
367        };
368        let scale_factor = single("scale_factor")?.unwrap_or(1.0);
369        let add_offset = single("add_offset")?.unwrap_or(0.0);
370        let explicit_fill = single("_FillValue")?;
371        // A byte variable with no `_FillValue` has no fill: the default's use is "not
372        // recommended" and every byte value is valid.
373        let fill = explicit_fill.or(match own {
374            Type::Byte | Type::Char => None,
375            Type::Short => Some(-32767.0),
376            Type::Int => Some(-2_147_483_647.0),
377            // FILL_FLOAT and FILL_DOUBLE, both 9.969209968386869e36.
378            Type::Float => Some(f64::from(f32::from_bits(0x7CF0_0000))),
379            Type::Double => Some(f64::from_bits(0x479E_0000_0000_0000)),
380        });
381        let mut missing: Vec<f64> = fill.into_iter().collect();
382        missing.extend(numbers("missing_value")?.unwrap_or_default());
383
384        let mut valid_min = single("valid_min")?;
385        let mut valid_max = single("valid_max")?;
386        if let Some(range) = numbers("valid_range")? {
387            if valid_min.is_some() || valid_max.is_some() {
388                return Err(unsupported(
389                    "`valid_range` is given with `valid_min` or `valid_max`".into(),
390                ));
391            }
392            let [low, high] = range[..] else {
393                return Err(unsupported("`valid_range` is not two numbers".into()));
394            };
395            valid_min = Some(low);
396            valid_max = Some(high);
397        }
398        // With no valid bounds, the fill bounds the valid range on its own side, one unit away
399        // for integers and two units in the last place for floating point.
400        if valid_min.is_none()
401            && valid_max.is_none()
402            && let Some(fill) = fill.filter(|f| !f.is_nan())
403            && (own != Type::Byte || explicit_fill.is_some())
404        {
405            // Stepped in the variable's own type, so a fill at the largest finite value still
406            // has a finite neighbour.
407            let toward_zero = |x: f64| match own {
408                Type::Float => {
409                    let x = x as f32;
410                    let step = |y: f32| {
411                        if fill > 0.0 {
412                            y.next_down()
413                        } else {
414                            y.next_up()
415                        }
416                    };
417                    f64::from(step(step(x)))
418                }
419                Type::Double => {
420                    let step = |y: f64| {
421                        if fill > 0.0 {
422                            y.next_down()
423                        } else {
424                            y.next_up()
425                        }
426                    };
427                    step(step(x))
428                }
429                _ => {
430                    if fill > 0.0 {
431                        x - 1.0
432                    } else {
433                        x + 1.0
434                    }
435                }
436            };
437            if fill > 0.0 {
438                valid_max = Some(toward_zero(fill));
439            } else {
440                valid_min = Some(toward_zero(fill));
441            }
442        }
443        Ok(Packing {
444            scale_factor,
445            add_offset,
446            missing,
447            valid_min,
448            valid_max,
449        })
450    }
451
452    /// The physical value at `index` (one entry per dimension): `Ok(None)` where it is missing.
453    ///
454    /// # Errors
455    ///
456    /// As [`Variable::packing`], and [`NetCdfError::Malformed`] if `index` is out of range.
457    pub fn unpacked(&self, index: &[u64]) -> Result<Option<f64>, NetCdfError> {
458        let packing = self.packing()?;
459        let stored = self
460            .offset(index)
461            .and_then(|i| self.values.get(i))
462            .ok_or_else(|| malformed(format!("index {index:?} is outside `{}`", self.name)))?;
463        Ok(packing.unpack(stored))
464    }
465}
466
467/// How a variable's stored values map to physical ones: [`Variable::packing`]. Every bound is in
468/// the stored (packed) values' domain, as the netCDF Users Guide's attribute conventions ask.
469#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
470pub struct Packing {
471    /// `scale_factor`, 1 if absent.
472    pub scale_factor: f64,
473    /// `add_offset`, 0 if absent.
474    pub add_offset: f64,
475    /// Stored values that mean "missing": the fill value (`_FillValue`, or the type's default
476    /// unless the type is byte) and every `missing_value`.
477    pub missing: Vec<f64>,
478    /// The least valid stored value: `valid_min`, the first of `valid_range`, or, with neither
479    /// bound given, just above a fill that is not positive.
480    pub valid_min: Option<f64>,
481    /// The greatest valid stored value: `valid_max`, the second of `valid_range`, or, with
482    /// neither bound given, just below a positive fill.
483    pub valid_max: Option<f64>,
484}
485
486impl Packing {
487    /// The physical value of stored value `stored`, `stored · scale_factor + add_offset`, or `None`
488    /// if it is missing or outside the valid range.
489    pub fn unpack(&self, stored: f64) -> Option<f64> {
490        if stored.is_nan()
491            || self.missing.contains(&stored)
492            || self.valid_min.is_some_and(|low| stored < low)
493            || self.valid_max.is_some_and(|high| stored > high)
494        {
495            return None;
496        }
497        Some(stored * self.scale_factor + self.add_offset)
498    }
499}
500
501/// A netCDF classic or 64-bit offset file, read whole.
502#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
503pub struct NetCdf {
504    /// Which of the two formats it is.
505    pub format: Format,
506    /// Its dimensions, in header order.
507    pub dimensions: Vec<Dimension>,
508    /// Its global attributes.
509    pub attributes: Vec<Attribute>,
510    /// Its variables, in header order.
511    pub variables: Vec<Variable>,
512}
513
514impl NetCdf {
515    /// Reads a file from its bytes.
516    ///
517    /// # Errors
518    ///
519    /// [`NetCdfError`]: a netCDF-4 or CDF-5 file, bytes that are not netCDF, a header that breaks
520    /// the grammar, values past the end of the bytes, or a streaming record count.
521    pub fn parse(bytes: &[u8]) -> Result<Self, NetCdfError> {
522        let format = match bytes.get(..4) {
523            Some(b"CDF\x01") => Format::Classic,
524            Some(b"CDF\x02") => Format::Offset64,
525            Some(b"CDF\x05") => return Err(NetCdfError::Cdf5),
526            Some([0x89, b'H', b'D', b'F']) => return Err(NetCdfError::NetCdf4),
527            _ => {
528                let head = bytes.iter().take(8).map(|b| format!("{b:02x}")).collect();
529                return Err(NetCdfError::NotNetCdf { head });
530            }
531        };
532        let mut header = Reader { bytes, position: 4 };
533        let numrecs = header.u32("the record count")?;
534        if numrecs == STREAMING {
535            return Err(NetCdfError::Streaming);
536        }
537        let numrecs = u64::from(numrecs);
538
539        // Names seen so far go in sets, so a header of many names costs no quadratic time.
540        let mut dimensions = Vec::new();
541        let mut seen = BTreeSet::new();
542        for _ in 0..header.list(NC_DIMENSION, "dimension")? {
543            let name = header.name()?;
544            if !seen.insert(name.clone()) {
545                return Err(malformed(format!("two dimensions are called `{name}`")));
546            }
547            let length = header.count("a dimension length")?;
548            let is_record = length == 0;
549            if is_record && dimensions.iter().any(|d: &Dimension| d.is_record) {
550                return Err(malformed("more than one record dimension"));
551            }
552            dimensions.push(Dimension {
553                name,
554                length: if is_record { numrecs } else { length },
555                is_record,
556            });
557        }
558        let attributes = header.attributes()?;
559
560        let mut layouts = Vec::new();
561        let mut seen = BTreeSet::new();
562        for _ in 0..header.list(NC_VARIABLE, "variable")? {
563            let name = header.name()?;
564            if !seen.insert(name.clone()) {
565                return Err(malformed(format!("two variables are called `{name}`")));
566            }
567            let rank = header.count("a variable's rank")?;
568            let mut dims = Vec::new();
569            for axis in 0..rank {
570                let id = usize::try_from(header.count("a dimension id")?)
571                    .map_err(|_| malformed("dimension id out of range"))?;
572                let Some(dimension) = dimensions.get(id) else {
573                    return Err(malformed(format!("variable `{name}` names dimension {id}")));
574                };
575                if dimension.is_record && axis != 0 {
576                    return Err(malformed(format!(
577                        "variable `{name}` has the record dimension other than first"
578                    )));
579                }
580                dims.push(id);
581            }
582            let variable_attributes = header.attributes()?;
583            let kind = Type::from_tag(header.u32("a variable's type")?)?;
584            let _vsize = header.u32("a variable's size")?;
585            let begin = match format {
586                Format::Classic => u64::from(header.u32("a variable's offset")?),
587                Format::Offset64 => header.u64("a variable's offset")?,
588            };
589            let is_record = dims.first().is_some_and(|&id| dimensions[id].is_record);
590            // Values per record for a record variable, or in all for any other.
591            let mut count: u64 = 1;
592            for &id in dims.iter().skip(usize::from(is_record)) {
593                count = count
594                    .checked_mul(dimensions[id].length)
595                    .ok_or_else(|| malformed(format!("variable `{name}` is too large")))?;
596            }
597            let slab = count
598                .checked_mul(kind.size())
599                .ok_or_else(|| malformed(format!("variable `{name}` is too large")))?;
600            layouts.push(Layout {
601                name,
602                dims,
603                attributes: variable_attributes,
604                kind,
605                begin,
606                is_record,
607                slab,
608            });
609        }
610
611        // The record size (Note on vsize): each record variable's slab padded to four bytes,
612        // except that a lone record variable is not padded (Note on padding; unpadded slabs of
613        // the wider types are multiples of four already).
614        let record_variables = layouts.iter().filter(|l| l.is_record).count();
615        let mut record_size: u64 = 0;
616        for layout in layouts.iter().filter(|l| l.is_record) {
617            let padded = if record_variables == 1 {
618                layout.slab
619            } else {
620                pad4(layout.slab)?
621            };
622            record_size = record_size
623                .checked_add(padded)
624                .ok_or_else(|| malformed("the record size is too large"))?;
625        }
626
627        // In a well-formed file no two variables share a byte, so together they hold no more than
628        // the file does. A header whose variables overlap could otherwise make a small file decode
629        // into far more memory than it takes.
630        let mut claimed: u64 = 0;
631        for layout in &layouts {
632            let bytes_held = if layout.is_record {
633                layout.slab.checked_mul(numrecs)
634            } else {
635                Some(layout.slab)
636            };
637            claimed = bytes_held
638                .and_then(|b| claimed.checked_add(b))
639                .ok_or_else(|| malformed("the variables' sizes overflow"))?;
640        }
641        if claimed > bytes.len() as u64 {
642            return Err(malformed(format!(
643                "the variables hold {claimed} bytes, more than the file's {}",
644                bytes.len()
645            )));
646        }
647
648        let names: Vec<Arc<str>> = dimensions
649            .iter()
650            .map(|d: &Dimension| Arc::from(d.name.as_str()))
651            .collect();
652        let mut variables = Vec::with_capacity(layouts.len());
653        for layout in layouts {
654            let values = if layout.is_record {
655                let mut data = Vec::new();
656                for record in 0..numrecs {
657                    let start = record
658                        .checked_mul(record_size)
659                        .and_then(|offset| offset.checked_add(layout.begin))
660                        .ok_or_else(|| malformed("a record offset is too large"))?;
661                    data.extend_from_slice(slice(bytes, start, layout.slab, &layout.name)?);
662                }
663                Values::decode(layout.kind, &data)
664            } else {
665                Values::decode(
666                    layout.kind,
667                    slice(bytes, layout.begin, layout.slab, &layout.name)?,
668                )
669            };
670            variables.push(Variable {
671                name: layout.name,
672                dimensions: layout
673                    .dims
674                    .iter()
675                    .map(|&id| Arc::clone(&names[id]))
676                    .collect(),
677                shape: layout
678                    .dims
679                    .iter()
680                    .map(|&id| dimensions[id].length)
681                    .collect(),
682                attributes: layout.attributes,
683                values,
684            });
685        }
686
687        Ok(NetCdf {
688            format,
689            dimensions,
690            attributes,
691            variables,
692        })
693    }
694
695    /// The variable called `name`.
696    pub fn variable(&self, name: &str) -> Option<&Variable> {
697        self.variables.iter().find(|v| v.name == name)
698    }
699
700    /// The dimension called `name`.
701    pub fn dimension(&self, name: &str) -> Option<&Dimension> {
702        self.dimensions.iter().find(|d| d.name == name)
703    }
704
705    /// The global attribute called `name`.
706    pub fn attribute(&self, name: &str) -> Option<&Attribute> {
707        self.attributes.iter().find(|a| a.name == name)
708    }
709}
710
711/// A variable's header entry, before its values are read.
712struct Layout {
713    name: String,
714    dims: Vec<usize>,
715    attributes: Vec<Attribute>,
716    kind: Type,
717    begin: u64,
718    is_record: bool,
719    /// Bytes of values per record (record variables) or in all (others), unpadded.
720    slab: u64,
721}
722
723/// `n` rounded up to a multiple of four, or an error if that overflows (a header can make a
724/// record variable's slab as large as `u64::MAX`).
725fn pad4(n: u64) -> Result<u64, NetCdfError> {
726    n.div_ceil(4)
727        .checked_mul(4)
728        .ok_or_else(|| malformed(format!("a size of {n} bytes is too large")))
729}
730
731/// `len` bytes of `bytes` from `start`, or [`NetCdfError::Truncated`].
732fn slice<'a>(bytes: &'a [u8], start: u64, len: u64, what: &str) -> Result<&'a [u8], NetCdfError> {
733    let total = bytes.len() as u64;
734    let truncated = || NetCdfError::Truncated {
735        what: what.to_owned(),
736        start,
737        end: start.saturating_add(len),
738        len: total,
739    };
740    let end = start.checked_add(len).ok_or_else(truncated)?;
741    if end > total {
742        return Err(truncated());
743    }
744    let (Ok(start), Ok(end)) = (usize::try_from(start), usize::try_from(end)) else {
745        return Err(truncated());
746    };
747    Ok(&bytes[start..end])
748}
749
750/// A cursor over the header.
751struct Reader<'a> {
752    bytes: &'a [u8],
753    position: u64,
754}
755
756impl Reader<'_> {
757    fn take(&mut self, len: u64, what: &str) -> Result<&[u8], NetCdfError> {
758        let taken = slice(self.bytes, self.position, len, what)?;
759        self.position += len;
760        Ok(taken)
761    }
762
763    fn u32(&mut self, what: &str) -> Result<u32, NetCdfError> {
764        let b = self.take(4, what)?;
765        Ok(u32::from_be_bytes([b[0], b[1], b[2], b[3]]))
766    }
767
768    fn u64(&mut self, what: &str) -> Result<u64, NetCdfError> {
769        let b = self.take(8, what)?;
770        Ok(u64::from_be_bytes([
771            b[0], b[1], b[2], b[3], b[4], b[5], b[6], b[7],
772        ]))
773    }
774
775    /// A signed count read as `NON_NEG`: the specification's integers are signed, so a count
776    /// with the top bit set is malformed.
777    fn count(&mut self, what: &str) -> Result<u64, NetCdfError> {
778        let n = self.u32(what)?;
779        if n > i32::MAX as u32 {
780            return Err(malformed(format!("{what} is negative")));
781        }
782        Ok(u64::from(n))
783    }
784
785    /// The length of a list tagged `tag`, or 0 if it is `ABSENT`.
786    fn list(&mut self, tag: u32, what: &str) -> Result<u64, NetCdfError> {
787        let found = self.u32(&format!("the {what} list's tag"))?;
788        let n = self.count(&format!("the {what} count"))?;
789        if found == 0 && n == 0 {
790            return Ok(0);
791        }
792        if found != tag {
793            return Err(malformed(format!(
794                "the {what} list has tag {found:#x}, not {tag:#x}"
795            )));
796        }
797        Ok(n)
798    }
799
800    fn name(&mut self) -> Result<String, NetCdfError> {
801        let n = self.count("a name's length")?;
802        let raw = self.take(pad4(n)?, "a name")?;
803        let text = std::str::from_utf8(&raw[..raw.len().min(n as usize)])
804            .map_err(|_| malformed("a name is not UTF-8"))?;
805        Ok(text.to_owned())
806    }
807
808    fn attributes(&mut self) -> Result<Vec<Attribute>, NetCdfError> {
809        let mut attributes = Vec::new();
810        let mut seen = BTreeSet::new();
811        for _ in 0..self.list(NC_ATTRIBUTE, "attribute")? {
812            let name = self.name()?;
813            if !seen.insert(name.clone()) {
814                return Err(malformed(format!("two attributes are called `{name}`")));
815            }
816            let kind = Type::from_tag(self.u32("an attribute's type")?)?;
817            let n = self.count("an attribute's length")?;
818            let size = n
819                .checked_mul(kind.size())
820                .ok_or_else(|| malformed(format!("attribute `{name}` is too large")))?;
821            let raw = self.take(pad4(size)?, &format!("attribute `{name}`"))?;
822            let values = Values::decode(kind, &raw[..size as usize]);
823            attributes.push(Attribute { name, values });
824        }
825        Ok(attributes)
826    }
827}
828
829#[cfg(test)]
830mod tests;