rm3100/capture.py

261 lines
12 KiB
Python
Raw Normal View History

"""Read an RM3100 capture and reconstruct everything the file does not store.
A capture holds only irreducible facts -- sample count, host clock, raw counts --
plus a header of configuration. This module turns that back into field values and
a time base, and is the single place either is derived.
Two time bases come out of `load()`:
elapsed_nominal N * dt_nominal, using the TMRC table rate
elapsed N * dt_true, using the rate actually observed
The chip's RC oscillator is accurate to about +/-7% (manual section 5.2.1) and
our unit runs ~6% slow, so `elapsed_nominal` is uniform but wrongly scaled.
`dt_true` comes from regressing the host clock on the sample index: the chip
supplies short-term regularity, the host supplies long-term rate calibration.
"""
import csv
import numpy as np
import rm3100
# Flags logger.py writes in the `warning` column, space separated. Duplicated
# rather than imported so that reading a capture never pulls in the USB stack.
WARN_MISSED = "MISSED" # placeholder row: no data, keeps the index contiguous
WARN_AMBIGUOUS = "AMBIGUOUS" # gap ending here was of uncertain length
class CaptureError(Exception):
pass
class Capture:
"""A loaded capture: raw counts, nanotesla axes, and two time bases."""
def __init__(self, path, meta, sample_index, system_time, counts,
missed=None, ambiguous=None):
self.path = path
self.missed = (np.zeros(len(sample_index), dtype=bool)
if missed is None else missed)
# A subset of missed: gaps whose length could not be counted
# confidently, so sample_index may have slipped across them.
self.ambiguous = (np.zeros(len(sample_index), dtype=bool)
if ambiguous is None else ambiguous)
self.meta = meta
self.sample_index = sample_index
self.system_time = system_time
self.counts = counts
self.cycle_count = int(meta["cycle_count"])
self.tesla_per_count = float(meta["tesla_per_count"])
self.lsb_nt = self.tesla_per_count * rm3100.NT_PER_TESLA
self.nominal_rate_hz = float(meta["nominal_rate_hz"])
self.dt_nominal = 1.0 / self.nominal_rate_hz
self.x, self.y, self.z = (counts[a] * self.lsb_nt for a in "xyz")
self.total = np.sqrt(self.x**2 + self.y**2 + self.z**2)
# Least-squares fit of host clock against grid coordinate, with both
# origins removed first. polyfit rescales the Vandermonde columns but
# never centres the ordinate, so fitting raw epoch seconds (~1.7e9) for
# a ~4 ms slope loses six significant digits to cancellation -- measured
# at 1.8e-6 of relative slope error against 3.8e-9 once centred.
# Subtracting a constant from each axis leaves the slope untouched and
# only moves the intercept, which nothing outside this fit uses.
index_0, time_0 = sample_index[0], system_time[0]
slope, intercept = np.polyfit(sample_index - index_0,
system_time - time_0, 1)
self.dt_true = float(slope)
# Error in the *rate*, so the sign matches the reported Hz: negative
# means the chip samples slower than the nominal table value.
#
# Against the *nominal* model, which no particular part obeys. Its own
# error is cycle-count dependent -- on one unit, +0.84% at cycle count
# 100 and -0.99% at 400, same supply -- so this is a diagnostic of the
# configuration, not a measurement of the oscillator, and multiplying a
# field by (1 + rate_error) is not a gain correction. calibrate.py does
# that against a measured per-unit reference.
self.rate_error = self.dt_nominal / self.dt_true - 1.0
self.residuals = (system_time - time_0) - (
intercept + slope * (sample_index - index_0))
self.elapsed_nominal = sample_index * self.dt_nominal
self.elapsed = sample_index * self.dt_true
# Two different things live in the timing error, and conflating them is
# misleading. Read latency is local: the scatter of one read interval
# about the next. Fit residual is global: how far the whole capture
# departs from a single straight line, which on a long run is dominated
# by the oscillator's rate drifting with temperature, not by the host.
self.read_jitter = float(np.diff(system_time).std()) if len(system_time) > 1 else 0.0
self.residual_sd = float(self.residuals.std())
self.residual_span = float(np.ptp(self.residuals))
# One sample period of accumulated error means the single slope is no
# longer describing the capture.
self.drift_limited = self.residual_sd > self.dt_true
@property
def true_rate_hz(self):
return 1.0 / self.dt_true
@property
def duration(self):
return float(self.elapsed[-1] - self.elapsed[0])
def axes(self):
"""(key, array) for the three axes plus the derived total."""
return [("x", self.x), ("y", self.y), ("z", self.z),
("total", self.total)]
def restrict(self, start=None, end=None):
"""Return a new Capture covering a window of drift-corrected seconds.
Re-fits on the subset, so a window's rate is its own rather than
inherited -- which is what makes it usable for spotting drift across a
long run by comparing windows.
"""
keep = np.ones(len(self.sample_index), dtype=bool)
if start is not None:
keep &= self.elapsed >= start
if end is not None:
keep &= self.elapsed <= end
if keep.sum() < 64:
raise CaptureError(
f"{self.path}: window [{start}, {end}] leaves "
f"{int(keep.sum())} samples, too few to analyse")
return Capture(self.path, self.meta, self.sample_index[keep],
self.system_time[keep],
{a: self.counts[a][keep] for a in "xyz"},
self.missed[keep], self.ambiguous[keep])
def summary(self):
lines = [
f"{self.path}: {len(self.sample_index):,} samples over "
f"{self.duration:.2f} s",
f" cycle count {self.cycle_count}, 1 LSB = {self.lsb_nt:.2f} nT",
f" rate {self.true_rate_hz:.3f} Hz measured vs "
f"{self.nominal_rate_hz:g} Hz nominal ({self.rate_error * 100:+.2f}%)",
f" read jitter {self.read_jitter * 1e3:.3f} ms sd | "
f"fit residual {self.residual_sd * 1e3:.1f} ms sd, "
f"{self.residual_span:.2f} s span",
f" Nyquist {self.true_rate_hz / 2:.2f} Hz",
]
if self.drift_limited:
lines += ["",
f" WARNING: the single-rate model does not fit this capture "
f"(residual sd {self.residual_sd * 1e3:.0f} ms",
f" against a {self.dt_true * 1e3:.1f} ms period)."]
if self.missed.any():
# Each lost measurement is an independent chance to insert one
# placeholder too many or too few, and the error accumulates in
# sample_index. That is the likelier cause here than thermal drift.
lines += [
f" With {int(self.missed.sum()):,} lost measurement(s) the likely cause is "
"miscounted placeholders:",
" how many grid points passed unseen can only be estimated, so the index",
" slips by about one per miss. The grid is exact only in a loss-free "
"capture.",
" Re-record at a lower rate rather than trusting this one for "
"spectral work.",
]
else:
lines += [
" No measurements were lost, so this is the chip's oscillator drifting --",
" expect thermal variation over a long run. Frequencies are scaled by an",
" average rate and will be smeared; analyse shorter windows",
" (capture.restrict) for spectral work.",
]
if self.missed.any():
n = int(self.missed.sum())
lines.insert(1, f" {n:,} lost measurement(s) "
f"({n / len(self.missed) * 100:.3f}%), interpolated")
if self.ambiguous.any():
lines.insert(1, f" {int(self.ambiguous.sum()):,} AMBIGUOUS gap(s) "
"-- length uncertain, index may have slipped")
if self.meta.get("note"):
lines.insert(1, f" note: {self.meta['note']}")
return "\n".join(lines)
def _parse_header(handle):
"""Consume leading '# key: value' lines, leaving the reader at the CSV."""
meta = {}
while True:
position = handle.tell()
line = handle.readline()
if not line:
break
if not line.startswith("#"):
handle.seek(position)
break
key, _, value = line[1:].partition(":")
if value:
meta[key.strip()] = value.strip()
return meta
def load(path):
"""Load a capture written by logger.py."""
with open(path, newline="") as handle:
meta = _parse_header(handle)
rows = list(csv.DictReader(handle))
if "rm3100_capture" not in meta:
raise CaptureError(
f"{path}: no capture header found. Files written before the header "
"format was introduced cannot be read -- re-record them.")
missing = {"cycle_count", "tesla_per_count", "nominal_rate_hz"} - meta.keys()
if missing:
raise CaptureError(f"{path}: header missing {', '.join(sorted(missing))}")
if len(rows) < 64:
raise CaptureError(f"{path}: only {len(rows)} samples, too few to analyse")
# logger.py writes "MISSED" in the count fields for a measurement it could
# not read, keeping sample_index contiguous so the chip-time grid stays
# valid across the gap.
if "warning" not in rows[0]:
raise CaptureError(
f"{path}: no 'warning' column. Captures predating it cannot be "
"read -- re-record them.")
flags = [set(r["warning"].split()) for r in rows]
missed = np.array([WARN_MISSED in f for f in flags])
# Independent of missed: a gap can be uncertain yet round to zero losses,
# in which case the flag rides on the real sample that ends it.
ambiguous = np.array([WARN_AMBIGUOUS in f for f in flags])
if missed.all():
raise CaptureError(f"{path}: every row is a lost measurement")
sample_index = np.array([int(r["sample_index"]) for r in rows], dtype=np.int64)
# logger.py keeps the index contiguous even across lost measurements -- it
# writes MISSED placeholders rather than skipping grid points, and stops the
# run outright if it ever cannot. So a gap here means the file was truncated
# mid-row, damaged, or hand-edited, not merely cut short.
gaps = np.diff(sample_index)
if np.any(gaps != 1):
bad = int(sample_index[np.argmax(gaps != 1)])
raise CaptureError(
f"{path}: sample_index is not contiguous (breaks after {bad}). "
"The chip grid is only valid for an unbroken index.")
# Lost measurements carry no data. They are linearly interpolated so the
# uniform grid the spectra depend on is preserved, and counted so the
# substitution is never silent.
counts = {}
good = ~missed
for a in "xyz":
# Every row parses as an integer -- placeholders carry zeros, and the
# warning column is what marks them as having no data.
v = np.array([int(r[f"{a}_raw"]) for r in rows], dtype=np.float64)
if missed.any():
v[missed] = np.interp(np.flatnonzero(missed), np.flatnonzero(good),
v[good])
counts[a] = v
return Capture(
path, meta, sample_index,
np.array([float(r["system_time_unix"]) for r in rows]),
counts, missed, ambiguous,
)