# geoML - machine learning models for geospatial data
# Copyright (C) 2026 Ítalo Gomes Gonçalves
#
# This program is free software: you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR a PARTICULAR PURPOSE. See the
# GNU General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with this program. If not, see <https://www.gnu.org/licenses/>.
"""
The same figures, for looking at rather than for printing.
`Interactive` answers every question `Explorer` does, with the same method
names and the same arguments, and hands back a plotly figure instead of a
matplotlib one:
.. code-block:: python
eda = geoml.plots.Interactive(point, continuous="Elements",
categorical="Rock")
eda.pairs().show()
eda.scene()
What that buys is the things a printed figure cannot do: zooming a panel of a
scatter matrix takes its whole row and column with it, so a cluster can be
followed across every pair at once; clicking a category in the legend takes it
out of every panel; and a three-dimensional scene can be turned around.
Every trace drawn from locations carries, in `customdata`, the row of the
container each of its points came from. That is what makes a `Dashboard` able
to link one figure's selection to another's -- see `dashboard`. Figures whose
points are not locations, such as the training curve, carry none, and a
simulated value is a realization rather than a place, so the simulated half of
`simulation_pairs` carries none either.
The numbers all come from `prepare`, the same functions `Explorer` reads, so
the two backends cannot drift into showing different things.
"""
import numpy as _np
import plotly.graph_objects as _go
from plotly.subplots import make_subplots as _subplots
import geoml.data as _data
import geoml.metrics as _gmet
import geoml.plots.base as _base
import geoml.plots.dashboard as _dash
import geoml.plots.prepare as _prep
import geoml.plots.style as _style
# What a point that fell outside the selection looks like. Plotly's own default
# is a light dimming, which on a crowded panel is not enough to tell the chosen
# points from the rest.
UNSELECTED = {"marker": {"opacity": 0.08}}
# the grey of the reference lines, matching the matplotlib figures
REFERENCE = "#4a4a4a"
INK = "#2b2b2b"
[docs]
class Interactive(_base.Selection):
"""
Exploratory figures for one data set, as plotly figures.
Takes its arguments from `base.Selection`: the container, a continuous and
a categorical variable, a model for the figures that need one, and the
colours to draw them in. Every method mirrors the one of the same name on
`Explorer`, down to the arguments, except that a figure is sized in pixels
(`height`, `width`) rather than in inches.
"""
# ------------------------------------------------------------------ #
# helpers
# ------------------------------------------------------------------ #
@staticmethod
def _finish(figure, title=None, height=None, width=None, **layout):
"""The package's look, applied to a figure on its way out."""
figure.update_layout(template=_style.TEMPLATE, height=height,
width=width, **layout)
if title is not None:
figure.update_layout(title_text=title)
return figure
@staticmethod
def _axis(row, column, columns):
"""Plotly's suffix for the axes of panel `(row, column)`, 1-based."""
index = (row - 1) * columns + column
return "" if index == 1 else str(index)
def _points(self, figure, x, y, rows, size, alpha, color, name,
legend, row, column):
"""One trace of locations, each carrying the row it came from."""
figure.add_trace(
_go.Scatter(
x=x, y=y, mode="markers",
customdata=rows,
marker={"size": size, "opacity": alpha, "color": color,
"line": {"width": 0}},
unselected=UNSELECTED,
name=name, legendgroup=name, showlegend=legend,
hovertemplate="%{x:.4g}, %{y:.4g}"
"<br>row %{customdata}<extra></extra>"),
row=row, col=column)
def _cells(self, figure, x, y, bins, log_counts, row, column):
"""Points counted into cells, as a heatmap.
Plotly's own `Histogram2d` bins for itself but paints every cell,
including the empty ones, and has no way to spread the colours over the
logarithm of a count. Counting first settles both: an empty cell is a
gap in the array and is left unpainted, and the colour can be taken
from whichever of the two the caller asked for.
No colour bar: these go in the panels of a matrix, where a bar beside
each would cost more room than the counts are worth. The hover says
how many points a cell holds, which is the question one actually asks
of a cell.
"""
counted = _prep.counts_2d(x, y, bins, log_counts)
figure.add_trace(
_go.Heatmap(
x=counted["x"], y=counted["y"], z=counted["z"],
# the count goes in `text`, not in `customdata`: a dashboard
# reads `customdata` as the row a point came from, and a cell
# is a number of points rather than one of them
text=counted["count"],
colorscale=self.cmap, showscale=False, hoverongaps=False,
hovertemplate="%{x:.4g}, %{y:.4g}"
"<br>%{text:d} points<extra></extra>"),
row=row, col=column)
def _density(self, figure, x, y, row, column):
"""Smoothed contours of where the points are."""
grid = _prep.density_grid(x, y)
if grid is None:
return
x_axis, y_axis, density = grid
figure.add_trace(
_go.Contour(
x=x_axis, y=y_axis, z=density, showscale=False,
colorscale=self.cmap, ncontours=7,
contours={"coloring": "lines"}, line={"width": 1.2},
hoverinfo="skip"),
row=row, col=column)
@staticmethod
def _line(figure, x, y, color, row=None, column=None, width=1.6,
name=None, legend=False, dash=None, hover=False):
"""A curve that is not made of locations: a fit, a reference, a mean."""
figure.add_trace(
_go.Scatter(x=x, y=y, mode="lines", name=name,
legendgroup=name, showlegend=legend,
line={"color": color, "width": width, "dash": dash},
hoverinfo=None if hover else "skip"),
row=row, col=column)
def _bars(self, figure, values, rows, edges, color, name, legend,
row, column, density=False):
"""A histogram, over bins chosen outside so panels can share them."""
figure.add_trace(
_go.Histogram(
x=values, customdata=rows,
xbins={"start": edges[0], "end": edges[-1],
"size": edges[1] - edges[0]},
autobinx=False,
histnorm="probability density" if density else "",
marker={"color": color}, opacity=0.65,
unselected={"marker": {"opacity": 0.15}},
name=name, legendgroup=name, showlegend=legend),
row=row, col=column)
# ------------------------------------------------------------------ #
# several at once
# ------------------------------------------------------------------ #
[docs]
def dashboard(self, figures=("scene", "histogram", "pairs"), **options):
"""
The named figures on one page, sharing a selection.
Each name is a method of this class, called with its own defaults.
That is the short way; for figures drawn with arguments of your own,
or captioned, or from more than one data set, build a `Dashboard`
directly -- it takes figures rather than names:
.. code-block:: python
geoml.plots.Dashboard(
[("Where the cadmium is", eda.scene(color="Cd")),
eda.pairs(log=True, upper="density")],
title="Jura", columns=2)
Parameters
----------
figures : sequence of str
Which figures, in the order they are to be laid out. Asking for
one this Interactive cannot draw raises whatever that figure
would have raised on its own -- there is no guessing here about
what a data set can support.
options
Passed to `Dashboard`: `title`, `columns`, `plotlyjs`, `hint`.
"""
options.setdefault(
"title", "%s — %s" % (type(self.data).__name__,
getattr(self.continuous, "name", None)
or getattr(self.categorical, "name", "")))
return _dash.Dashboard([getattr(self, name)() for name in figures],
**options)
# ------------------------------------------------------------------ #
# figures
# ------------------------------------------------------------------ #
[docs]
def histogram(self, bins=25, statistics=True, height=None,
width=None) -> "_go.Figure":
"""
The distribution of the continuous variable, one panel per component.
Split by category when there is one: the populations are drawn over
each other with a common set of bins, so their spreads can be compared
rather than only their shapes.
Parameters
----------
bins : int or sequence
How many bins, or where their edges are.
statistics : bool
Sum each panel up in a box: the count, the mean, the standard
deviation, the coefficient of variation, the skewness and the
kurtosis, beside the minimum, the quartiles, the median and the
maximum. They are of every measured value, the categories
pooled, and the kurtosis is the excess over a normal's. The box
sits over the half of the bins with the lower bars, and the axis
is raised so that none of those runs under it.
"""
var = self._require_continuous("histogram")
values, measured, labels = _prep.numeric_values(var)
self._check_measured(var, measured)
series = self._series()
rows, columns = _prep.grid_shape(len(labels))
# a panel here carries a title of its own as well as the tick labels
# of the one above, and both live in the gap between the two
figure = _subplots(rows=rows, cols=columns, subplot_titles=labels,
vertical_spacing=min(0.4 / rows, 0.2))
height = height or 100 + 250 * rows
# about how many pixels a panel takes, for the text plotly lays out
# in the page; a figure left to fill its page is taken as 900 wide
panel = (((width or 900) - 150) / columns * 0.85,
(height - 100) / rows * 0.75)
for i, label in enumerate(labels):
row, column = divmod(i, columns)
edges = _np.histogram_bin_edges(values[measured, i], bins=bins)
tallest = _np.zeros(len(edges) - 1)
for j, (name, mask) in enumerate(series):
keep = measured if mask is None else mask
self._bars(figure, values[keep, i], _np.where(keep)[0], edges,
self._color(j, name), name or var.name, i == 0,
row + 1, column + 1)
drawn = values[keep, i]
tallest = _np.maximum(tallest, _np.histogram(
drawn[_np.isfinite(drawn)], bins=edges)[0])
if statistics:
self._statistics(figure, values[measured, i], tallest, edges,
row + 1, column + 1, panel)
figure.update_yaxes(title_text="count", col=1)
return self._finish(
figure, title=var.name, height=height, width=width,
barmode="overlay",
legend={"title": {"text": getattr(self.categorical, "name", "")}})
[docs]
def pairs(self, kind="scatter", alpha=0.7, bins=60, log_counts=False,
log=False, principal_components=0, upper=None, size=5,
height=None, width=None) -> "_go.Figure":
"""
Every component against every other, with its distribution down the
diagonal.
Parameters
----------
kind : str
`"scatter"`, or `"hist2d"` to count the points into cells instead,
for when there are too many to draw one by one. Counts make a
single surface, so the categories are pooled -- and a counted cell
is not a location, so a matrix drawn this way takes no part in a
dashboard's linked selection.
alpha : float
How opaque each point is, for `kind="scatter"`.
bins : int
Cells along each axis, for `kind="hist2d"`.
log_counts : bool
Colour the cells by the logarithm of the count. Worth turning on
whenever a few cells hold most of the data, which is most of the
time with a skewed variable.
log : bool
Draw the data on a log scale: the centred log-ratio for a
composition, whose parts carry a constant sum, and an ordinary
logarithm otherwise.
principal_components : int
Draw this many principal components over the data, in the data's
own axes -- the reverse of `pca`. Each is a line through the mean,
one standard deviation of that component long either way.
upper : str
What to put in the upper triangle, which is otherwise left empty:
`"hist2d"` or `"density"` for where the mass is with the categories
pooled, or `"correlation"` for the coefficient alone.
"""
self._check_kind(kind)
var = self._require_vector("pairs")
values, measured, labels = _prep.numeric_values(var)
self._check_measured(var, measured)
series = self._series()
compositional = isinstance(var, _data.CompositionalVariable)
if log:
# only the measured rows: the gaps are padded with 1.0 to keep the
# array square, and that padding is not data to be transformed
transformed = values.copy()
transformed[measured] = _prep.logarithm(values[measured],
compositional)
values = transformed
labels = ["%s(%s)" % ("clr" if compositional else "log", label)
for label in labels]
analysis = None
if principal_components > 0:
if compositional and not log:
raise TypeError(
"the principal components of %r are those of its log-"
"ratios, which are not directions in the proportions "
"drawn here. Pass log=True to draw the log-ratios, which "
"is the space they belong to, or use `pca()` to put the "
"data on the components' own axes" % var.name)
analysis = _prep.principal_components(values[measured],
explained=1.0)
n = len(labels)
figure = self._matrix(values, labels, series, measured,
_np.arange(len(values)), kind=kind, alpha=alpha,
bins=bins, log_counts=log_counts, upper=upper,
size=size)
if analysis is not None:
for row in range(n):
for column in range(row):
self._axes_of_variation(figure, analysis, column, row,
principal_components)
title = "%s by %s" % (var.name, self.categorical.name) \
if self.categorical is not None else var.name
return self._finish(
figure, title=title, height=height or 80 + 170 * n,
width=width or 80 + 180 * n,
legend={"title": {"text": getattr(self.categorical, "name", "")}})
[docs]
def pca(self, explained=0.9, log=False, kind="scatter", alpha=0.7,
bins=60, log_counts=False, size=5, height=None, width=None) -> "_go.Figure":
"""
The same pairs plot, on principal components instead of measurements.
Only the components carrying `explained` of the variance are drawn.
Each panel also holds the loadings: an arrow per original column,
showing what it contributes to the two components on the axes. A
composition is opened up with the centred log-ratio first.
Parameters
----------
explained : float
Share of the total variance to reach, between 0 and 1.
log : bool
Take the components of the logarithms rather than of the
measurements. A composition is opened up either way.
kind, alpha, bins, log_counts
As in `pairs`.
"""
self._check_kind(kind)
var = self._require_vector("pca")
self._check_measured(var, _prep.numeric_values(var)[1])
analysis = _prep.component_analysis(var, explained, log)
scores = analysis["scores"]
ratio = analysis["ratio"]
n = analysis["n_components"]
labels = ["PC%d (%.1f%%)" % (i + 1, 100 * ratio[i]) for i in range(n)]
measured = analysis["measured"]
series = [(name, mask[measured]) for name, mask in self._series()] \
if self.categorical is not None else [(None, None)]
figure = self._matrix(
scores, labels, series, _np.ones(len(scores), dtype=bool),
_np.where(measured)[0], kind=kind, alpha=alpha, bins=bins,
log_counts=log_counts, upper=None, size=size)
for row in range(n):
for column in range(row):
self._loadings(figure, analysis["loadings"],
analysis["labels"], column, row, scores, n)
return self._finish(
figure,
title="%s — %.0f%% of the variance in %d components"
% (var.name, 100 * _np.sum(ratio[:n]), n),
height=height or 80 + 190 * n, width=width or 80 + 200 * n,
legend={"title": {"text": getattr(self.categorical, "name", "")}})
[docs]
def scene(self, color=None, clip=None, size=4, height=None, width=None) -> "_go.Figure":
"""
Where the data is, coloured by a variable.
The coordinates decide the drawing: a value against position in 1D, a
map in 2D, a scatter that can be turned around in 3D -- which is the
one plotly is worth reaching for on its own account.
A three-dimensional scene takes no part in a dashboard's linked
selection. Plotly has no box or lasso over a 3D scatter to select
with, and no per-point selection state on one to show a selection
made elsewhere. Several of them on one page do turn together, though.
Parameters
----------
color : str or list
The variable to colour by. Defaults to the continuous variable
held, or the categorical one if that is all there is.
A **list** of them puts a menu on the figure instead, one entry
per choice, and the ground is drawn once with the values swapped
over it. That is the answer to several variables over the same
body of rock: five grades become one scene with five entries on a
menu rather than five scenes, which is lighter, and it holds the
point cloud still while the variable changes -- the comparison one
is actually making. A name in the list may be a variable, one
component of a vector variable, or a vector variable standing for
all of its components, so `color=["Elements"]` names every grade
in it.
Only the locations measured in all of them are drawn; see
`prepare.color_choices`.
clip : pair of floats
Where to end the colour scale, as quantiles: `[0, 0.99]` for a
variable with a long right tail, which is most assays, and
`[0.01, 0.99]` to take both ends. Without it one value far from
the rest takes the whole scale and leaves everything else in a
single shade. Nothing is dropped -- the points beyond the ends
take the end colour, and the hover still reports what was
measured. A menu clips each of its choices by its own quantiles,
since they are different variables and share nothing but the
ground. Has no effect in 1D, where the value is an axis rather
than a colour.
"""
coordinates = _np.asarray(self.data.coordinates)
n_dim = coordinates.shape[1]
if n_dim > 3:
raise ValueError(
"a scene needs 1, 2 or 3 coordinates; this data has %d"
% n_dim)
labels = getattr(self.data, "coordinate_labels", None) \
or ["axis %d" % (i + 1) for i in range(n_dim)]
figure = _go.Figure()
if isinstance(color, (list, tuple)):
title, menu = self._scene_menu(figure, coordinates, n_dim, color,
size, clip)
else:
title, menu = self._scene_one(figure, coordinates, n_dim,
color, size, clip), None
layout: dict = {}
if n_dim == 3:
layout = {"scene": {
"xaxis": {"title": {"text": labels[0]}},
"yaxis": {"title": {"text": labels[1]}},
"zaxis": {"title": {"text": labels[2]}},
"aspectmode": "data"}}
else:
layout = {"xaxis": {"title": {"text": labels[0]}},
"yaxis": {"title": {"text": title if n_dim == 1
else labels[1]}}}
if n_dim == 2:
# a map: the two axes are the same thing, so the same scale
layout["yaxis"]["scaleanchor"] = "x"
if menu is not None:
layout["updatemenus"] = [menu]
# the menu sits over the figure, where the top margin is
layout["margin"] = {"l": 60, "r": 30, "t": 90, "b": 50}
return self._finish(figure, title=title,
height=height or 560, width=width,
legend={"title": {"text": title}}, **layout)
def _scene_one(self, figure, coordinates, n_dim, color, size, clip=None):
"""One variable over the coordinates, and what to call the figure."""
var = self._color_variable(color)
if self._is_categorical(var):
values, measured, names = _prep.category_values(var)
for j, (name, mask) in enumerate(_prep.groups(values, measured,
names)):
self._scene_trace(figure, coordinates[mask], n_dim, size,
rows=_np.where(mask)[0], name=name,
color=self._color(j, name))
else:
values, measured, _ = _prep.numeric_values(var)
self._scene_trace(figure, coordinates[measured], n_dim, size,
rows=_np.where(measured)[0], name=str(var.name),
values=values[measured, 0],
limits=_prep.color_limits(values[measured, 0],
clip))
return str(var.name)
def _scene_menu(self, figure, coordinates, n_dim, names, size, clip=None):
"""
One set of points, and a menu saying which values colour them.
The ground is drawn once and only the values change, which is both
lighter than a trace per choice -- a block model's coordinates are the
bulk of it, and there is no sense in carrying five copies -- and truer
to what is being asked: the cloud stays where it is while the variable
over it changes.
"""
values, rows, labels = _prep.color_choices(self.data, names)
# each choice by its own quantiles: they are different variables and
# share nothing but the ground they are drawn over
limits = [_prep.color_limits(values[:, i], clip)
for i in range(len(labels))]
self._scene_trace(figure, coordinates[rows], n_dim, size, rows=rows,
name=labels[0], values=values[:, 0],
limits=limits[0])
buttons = []
for i, label in enumerate(labels):
if n_dim == 1:
# in 1D the value is the vertical axis and carries no colour,
# so switching means moving the points rather than repainting
trace = {"y": [values[:, i]]}
layout = {"title": {"text": label},
"yaxis": {"title": {"text": label}}}
else:
low, high = limits[i] or (None, None)
trace = {"marker.color": [values[:, i]],
"marker.colorbar.title.text": label,
"marker.cauto": limits[i] is None,
"marker.cmin": low, "marker.cmax": high}
layout = {"title": {"text": label}}
buttons.append({"label": label, "method": "update",
"args": [trace, layout]})
return labels[0], {"buttons": buttons, "direction": "down",
"showactive": True, "x": 1.0, "xanchor": "right",
"y": 1.12, "yanchor": "top",
"bgcolor": "white", "bordercolor": "#d9d9d9",
"font": {"size": 10}}
def _scene_trace(self, figure, coordinates, n_dim, size, rows,
name=None, color=None, values=None, limits=None):
"""One scatter, whatever the number of coordinates."""
marker = {"size": size, "line": {"width": 0}}
if color is not None:
marker["color"] = color
elif n_dim > 1:
marker["color"] = values
marker["colorscale"] = self.cmap
marker["colorbar"] = {"title": {"text": name}, "thickness": 14}
if limits is not None:
marker["cmin"], marker["cmax"] = limits
if n_dim == 3:
# no customdata: a 3D scene cannot be brushed and cannot show a
# brush made elsewhere, so a row index on it would promise a link
# that is not there
figure.add_trace(_go.Scatter3d(
x=coordinates[:, 0], y=coordinates[:, 1], z=coordinates[:, 2],
mode="markers", marker=marker, name=name,
showlegend=color is not None))
return
if n_dim == 1:
# nothing to put on the other axis, so the value goes there --
# unless the colour is the whole message, and then it is a strip
position = {"x": coordinates[:, 0],
"y": values if values is not None
else _np.zeros(len(coordinates))}
else:
position = {"x": coordinates[:, 0], "y": coordinates[:, 1]}
figure.add_trace(_go.Scatter(
mode="markers", marker=marker, customdata=rows, name=name,
legendgroup=name, showlegend=color is not None,
unselected=UNSELECTED,
hovertemplate="%{x:.4g}, %{y:.4g}"
"<br>row %{customdata}<extra></extra>",
**position))
# ------------------------------------------------------------------ #
# figures that need the model
# ------------------------------------------------------------------ #
[docs]
def training_curve(self, window=None, height=None, width=None) -> "_go.Figure":
"""
The ELBO against the iteration it was measured at.
Every value is an estimate, from a sample of the latent variables and,
under `train_svi`, a sample of the data too, so the curve is noisy
whether or not training has settled. The running mean over it is the
line to read: flat means finished, still climbing means it is not.
Parameters
----------
window : int
Points to average over. Defaults to a fiftieth of the log.
"""
self._require_model("training_curve")
curve = _prep.training_curve(self.model, window)
figure = _go.Figure()
self._line(figure, curve["iteration"], curve["value"],
_style.color(0), width=1.0, name="ELBO", legend=True,
hover=True)
if len(curve["smooth"]) < len(curve["value"]):
self._line(figure, curve["smooth_iteration"], curve["smooth"],
_style.color(1), width=2.0, legend=True, hover=True,
name="mean of %d" % curve["window"])
return self._finish(
figure, title="Training", height=height or 420, width=width,
xaxis={"title": {"text": "iteration"}},
yaxis={"title": {"text": "ELBO"}},
hovermode="x unified")
[docs]
def simulation_pairs(self, kind="hist2d", bins=60, log_counts=False,
most=100000, margin=0.1, alpha=0.6, size=5,
height=None, width=None) -> "_go.Figure":
"""
What was measured against what was simulated.
The measurements fill the lower triangle and the simulations the upper,
in the same form and between the same limits, so the two halves of the
matrix can be read against each other: a simulation that reproduces the
data has an upper half that mirrors the lower one. Down the diagonal
the measured histogram carries the simulated density over it, which is
the same comparison one variable at a time.
Both halves are counted into cells by default. Simulations come in
location-by-realization blocks that run to millions of values, and a
scatter of that many points is a filled rectangle whatever its
transparency.
Parameters
----------
kind : str
`"hist2d"`, the default here, or `"scatter"`.
bins, log_counts, alpha, size
As in `pairs`.
most : int
About how many simulated values to draw per component.
margin : float
How far past the measured range to look, as a share of it.
Simulated values outside that window are left out.
"""
self._check_kind(kind)
var = self._require_continuous("simulation_pairs")
values, measured, labels = _prep.numeric_values(var)
self._check_measured(var, measured)
data = values[measured]
rows = _np.where(measured)[0]
sample = _prep.simulation_sample(var, most)
n = len(labels)
# The measured range with room around it, shared by a panel and its
# mirror -- without which the halves cannot be compared by eye at all.
limits = _prep.padded_range(data, margin)
inside = _np.column_stack(
[(sample[:, i] >= low) & (sample[:, i] <= high)
for i, (low, high) in enumerate(limits)])
figure = _subplots(rows=n, cols=n, shared_xaxes=True,
horizontal_spacing=0.02, vertical_spacing=0.02)
for row in range(n):
for column in range(n):
first = row == 0 and column == 0
if row == column:
edges = _np.histogram_bin_edges(
data[:, row], bins=25, range=limits[row])
self._bars(figure, data[:, row], rows, edges,
_style.color(0), "measured", first,
row + 1, column + 1, density=True)
curve = _prep.density_curve(sample[inside[:, row], row],
limits[row])
if curve is not None:
self._line(figure, curve[0], curve[1],
_style.color(1), row + 1, column + 1,
name="simulated", legend=first)
else:
measurements = column < row
source, identity = data, rows
if not measurements:
keep = inside[:, column] & inside[:, row]
# a simulated value is a realization, not a place
source, identity = sample[keep], None
self._pair(
figure, source[:, column], source[:, row], identity,
kind, alpha, size,
_prep.cells(len(source), bins), log_counts,
_style.color(0 if measurements else 1),
"measured" if measurements else "simulated",
False, row + 1, column + 1)
figure.update_yaxes(range=list(limits[row]),
row=row + 1, col=column + 1)
figure.update_xaxes(range=list(limits[column]),
row=row + 1, col=column + 1)
self._label_edges(figure, labels, n)
return self._finish(
figure,
title="%s: measured below the diagonal, simulated above"
% var.name,
height=height or 80 + 180 * n, width=width or 80 + 190 * n,
barmode="overlay")
[docs]
def prediction_scatter(self, component=None, kind="scatter", alpha=0.6,
bins=60, log_counts=False, trim=None, size=6,
height=None, width=None) -> "_go.Figure":
"""
What was predicted against what was measured.
A single variable is drawn with its two distributions along the sides,
which is where a bias shows that the scatter alone hides: the same
cloud can sit on the 1:1 line while the predicted values are packed
into a narrower range than the real ones. Several components are drawn
as a panel each; name one with `component` to get the margins for it.
Parameters
----------
component : str
One component of a vector variable, drawn on its own.
kind : str
`"scatter"`, or `"hist2d"` for a block model, where the points are
past counting.
alpha : float
How opaque each point is.
bins : int
Bins along each axis, for `kind="hist2d"`.
log_counts : bool
Colour the cells by the logarithm of the count.
trim : pair of floats
Leave the outliers out, as quantiles: `[0, 0.99]` for a variable
with a long right tail, which is most assays. Without it a few
values far from the rest set the limits and squeeze everything
else into a corner. The window runs from the lower quantile of
the measured or the predicted values, whichever is lower, to the
upper quantile of whichever is higher, and a location outside it
on either axis is left out of the panel and its margins. Each
panel is trimmed on its own, and counts in a corner how many it
left out.
"""
self._check_kind(kind)
var = self._require_continuous("prediction_scatter")
true, predicted, labels, rows = _prep.prediction_values(self.data,
var.name)
if component is not None:
if component not in labels:
raise KeyError("no component %r in %r; found %s"
% (component, var.name, ", ".join(labels)))
index = labels.index(component)
true, predicted = true[:, [index]], predicted[:, [index]]
labels = [component]
kept = [_prep.inside_trim(true[:, i], predicted[:, i], trim)
for i in range(len(labels))]
if len(labels) == 1:
return self._joint(true[kept[0], 0], predicted[kept[0], 0],
rows[kept[0]], labels[0], kind, alpha, bins,
log_counts, size, height, width,
left_out=int(_np.sum(~kept[0])))
panels, columns = _prep.grid_shape(len(labels))
figure = _subplots(rows=panels, cols=columns, subplot_titles=labels,
horizontal_spacing=0.08)
for i, label in enumerate(labels):
row, column = divmod(i, columns)
self._agreement(figure, true[kept[i], i], predicted[kept[i], i],
rows[kept[i]], kind, alpha, bins, log_counts, size,
row + 1, column + 1, legend=False,
left_out=int(_np.sum(~kept[i])))
figure.update_xaxes(title_text="measured", row=panels)
figure.update_yaxes(title_text="predicted", col=1)
return self._finish(
figure, title="%s: predicted against measured" % var.name,
height=height or 100 + 300 * panels, width=width,
showlegend=False)
[docs]
def accuracy(self, probabilities=None, height=None, width=None) -> "_go.Figure":
"""
Whether the simulated spread is the spread the errors actually have.
For each probability, the share of true values that fall inside the
interval holding that share of the simulations. On the 1:1 line the
model knows what it does not know; below it the intervals are too
narrow for the errors they have to cover, and above it the model is
hedging. Deutsch's goodness statistic sums that up in one number.
The intervals come from the model rather than from the container: what
is stored there is the ground, the likelihood noise having been
integrated out, and an assay is a measurement of the ground rather
than the ground itself. So this figure needs the model the selection
was built with.
"""
var = self._require_continuous("accuracy")
parts = _prep.continuous_parts(var)
# nothing measured means nothing to check against, said before the
# model is asked for anything
measured = [part.measurements.values.to_numpy().astype(float)
for part in parts]
for part, values in zip(parts, measured):
self._check_measured(part, ~_np.isnan(values))
samples = self._measurement_samples(var, "accuracy")
figure = _go.Figure()
self._line(figure, [0, 1], [0, 1], REFERENCE, width=1.0,
name="perfect", legend=True, dash="dash")
for i, part in enumerate(parts):
simulations = _np.asarray(samples[:, i, :], dtype=float)
keep = ~_np.isnan(measured[i])
nominal, observed = _gmet.coverage(
measured[i][keep], simulations[keep], probabilities)
figure.add_trace(_go.Scatter(
x=nominal, y=observed, mode="lines+markers",
marker={"size": 6},
line={"color": self._color(i, str(part.name)), "width": 1.8},
name="%s (G = %.2f)"
% (part.name, _gmet.goodness(nominal, observed)),
hovertemplate="nominal %{x:.2f}"
"<br>observed %{y:.2f}<extra></extra>"))
return self._finish(
figure, title="Accuracy", height=height or 520, width=width or 560,
xaxis={"title": {"text": "probability of the interval"}},
yaxis={"title": {"text": "share of true values inside it"},
"scaleanchor": "x"},
legend={"x": 0.02, "y": 0.98, "xanchor": "left",
"yanchor": "top"})
[docs]
def reliability(self, bins=10, height=None, width=None) -> "_go.Figure":
"""
Whether a claimed probability is the frequency it claims.
One curve per category: the locations binned by the probability
the model assigned to it, each bin's mean claim against the share
of its locations actually measured as that category. On the
diagonal a 70% claim is that category 70% of the time; below it
the model is overconfident, above it hedging. The legend carries
each curve's expected calibration error, the count-weighted mean
distance from the diagonal.
The same locations count as in `confusion_matrix`: a contact has
two measurements and no one truth, and locations missing either
the measurement or the prediction are left out.
Only honest on data the model has not seen: at a training
location the claim was fitted to its own outcome. The out-of-fold
container `models.cross_validate` fills is the honest input.
Parameters
----------
bins : int or sequence
How many bins, or where their edges are. A count gives
**equal-count** bins over the claimed probabilities; pass
explicit edges for equal width.
"""
var = self._require_categorical("reliability")
panels = _prep.reliability(self.data, var.name, bins=bins)
figure = _go.Figure()
self._line(figure, [0, 1], [0, 1], REFERENCE, width=1.0,
name="perfect", legend=True, dash="dash")
for i, panel in enumerate(panels):
figure.add_trace(_go.Scatter(
x=panel["claimed"], y=panel["observed"],
mode="lines+markers", marker={"size": 6},
line={"color": self._color(i, panel["label"]),
"width": 1.8},
# the counts ride in `text`: `customdata` is the dashboard's
# slot for container rows, and a bin is not a location
text=panel["count"],
name="%s (ECE = %.2f)" % (panel["label"], panel["ece"]),
hovertemplate="claimed %{x:.2f}<br>observed %{y:.2f}"
"<br>%{text} locations"
"<extra></extra>"))
return self._finish(
figure, title="Reliability",
height=height or 520, width=width or 560,
xaxis={"title": {"text": "claimed probability"}},
yaxis={"title": {"text": "share measured as the category"},
"scaleanchor": "x"},
legend={"x": 0.02, "y": 0.98, "xanchor": "left",
"yanchor": "top"})
[docs]
def confusion_matrix(self, height=None, width=None) -> "_go.Figure":
"""
What was measured against what the model called there, counted.
Rows are the measured categories, columns the predicted ones, so
the diagonal is agreement and each row reads as one category's
fate. The shading is each cell's share of its measured row --
categories are as unbalanced as rock types usually are, and raw
counts would light the dominant row and hide what happens to a
rare one -- and the counts are written in the cells.
Contacts do not count: a rock type variable carries two
measurements there, and neither alone is the truth the prediction
is measured against. Locations missing either the measurement or
the prediction are left out likewise.
Only honest on data the model has not seen: at a training location
the prediction interpolates its own measurement, and the diagonal
congratulates the model on remembering it. The out-of-fold
container `models.cross_validate` fills is the honest input.
"""
var = self._require_categorical("confusion_matrix")
table = _prep.confusion_matrix(self.data, var.name)
counts, labels = table["counts"], table["labels"]
figure = _go.Figure(_go.Heatmap(
z=table["share"], x=labels, y=labels,
text=counts, texttemplate="%{text}",
zmin=0.0, zmax=1.0, colorscale=_style.SEQUENTIAL,
colorbar={"title": "share of the<br>measured category"},
hovertemplate="measured %{y}<br>predicted %{x}"
"<br>count %{text}<br>share %{z:.2f}"
"<extra></extra>"))
return self._finish(
figure,
title="%s: %d locations, %.0f%% agreement"
% (var.name, int(counts.sum()),
100.0 * table["agreement"]),
height=height, width=width,
xaxis={"title": {"text": "predicted"}},
# reversed, so the matrix reads top-down as the printed one does
yaxis={"title": {"text": "measured"},
"autorange": "reversed"})
[docs]
def swath(self, predicted, axis=0, bins=12, where=None, weights=None,
quantiles=(0.05, 0.95), height=None,
width=None) -> "_go.Figure":
"""
The data's mean against the model's, slab by slab along one axis.
The check that localizes conditional bias instead of aggregating it
away. The data's means are declustered and the model's run only
over the ground `where` names -- the reach `assign_from_data`
writes -- so the comparison describes the deposit rather than the
drilling; with simulations, the band between two quantiles of the
realizations' slab means is drawn. Draws the continuous variable
when one was given, else the categorical one as stacked shares.
A swath draws slab aggregates, not locations, so like `variogram`
it carries no row index for the dashboard to link on.
Parameters
----------
predicted, axis, bins, where, weights, quantiles
As in `Explorer.swath`.
"""
if self.continuous is not None:
return self._continuous_swath(predicted, axis, bins, where,
weights, quantiles, height, width)
var = self._require_categorical("swath")
result = _prep.categorical_swath(self.data, predicted, var.name,
axis, bins, weights, where)
figure = _subplots(rows=2, cols=1, shared_xaxes=True,
row_heights=[0.75, 0.25], vertical_spacing=0.05)
for k, label in enumerate(result["labels"]):
color = self._color(k, label)
for side, share, offset in (
("data", result["data_share"][:, k], "data"),
("model", result["model_share"][:, k], "model")):
figure.add_trace(_go.Bar(
x=result["centre"], y=_np.nan_to_num(share),
width=0.42 * (result["hi"] - result["lo"]),
offsetgroup=offset, legendgroup=label, name=label,
showlegend=side == "data",
marker={"color": color,
"pattern": {"shape": "/" if side == "model"
else ""}},
hovertemplate="%s, %s: %%{y:.2f}<extra></extra>"
% (label, side)), row=1, col=1)
figure.add_trace(_go.Bar(
x=result["centre"], y=result["data_count"],
width=0.9 * (result["hi"] - result["lo"]),
marker={"color": self._color(3, "data"), "opacity": 0.5},
name="samples", showlegend=False,
hovertemplate="%{y} samples<extra></extra>"), row=2, col=1)
figure.update_layout(barmode="stack")
figure.update_yaxes(title_text="share (data | model, hatched)",
range=[0, 1], row=1, col=1)
figure.update_yaxes(title_text="samples", row=2, col=1)
figure.update_xaxes(title_text=result["axis"], row=2, col=1)
return self._finish(
figure, title="%s: %sshares along %s"
% (var.name, "declustered " if result["declustered"] else "",
result["axis"]),
height=height or 480, width=width)
[docs]
def proportions(self, predicted, where=None, weights=None, height=None,
width=None) -> "_go.Figure":
"""
The data's category shares against the model's, one bar pair each.
The whole-model reading of the categorical `swath`: the data's
declustered share of each category beside the model's expected
share over the ground `where` names, each block at its own volume.
Like `swath` it draws aggregates, not locations, so it carries no
row index for the dashboard to link on.
Parameters
----------
predicted, where, weights
As in `Explorer.proportions`.
"""
var = self._require_categorical("proportions")
result = _prep.proportions(self.data, predicted, var.name, weights,
where)
colors = [self._color(k, label)
for k, label in enumerate(result["labels"])]
figure = _go.Figure()
for side, share, pattern in (("data", result["data_share"], ""),
("model", result["model_share"], "/")):
figure.add_trace(_go.Bar(
x=result["labels"], y=share, name=side,
marker={"color": colors, "pattern": {"shape": pattern}},
hovertemplate="%%{x}, %s: %%{y:.3f}<extra></extra>" % side))
figure.update_layout(barmode="group")
figure.update_yaxes(title_text="share (data | model, hatched)")
figure.update_xaxes(title_text="%d samples, %d model locations"
% (result["data_count"], result["model_count"]))
return self._finish(
figure, title="%s: %sshares, data against the model"
% (var.name, "declustered " if result["declustered"] else ""),
height=height or 420, width=width)
def _continuous_swath(self, predicted, axis, bins, where, weights,
quantiles, height, width):
var = self._require_continuous("swath")
panels = _prep.swath(self.data, predicted, var.name, axis=axis,
bins=bins, weights=weights, where=where,
quantiles=quantiles)
rows, columns = _prep.grid_shape(len(panels))
titles = []
for row in range(rows):
titles += [panels[row * columns + c]["label"]
if row * columns + c < len(panels) else ""
for c in range(columns)]
titles += [""] * columns
figure = _subplots(
rows=2 * rows, cols=columns, shared_xaxes=True,
row_heights=[0.75, 0.25] * rows, vertical_spacing=0.05,
horizontal_spacing=0.1, subplot_titles=titles,
specs=[[{"secondary_y": r % 2 == 1} for _ in range(columns)]
for r in range(2 * rows)])
for i, panel in enumerate(panels):
row, column = divmod(i, columns)
self._swath(figure, panel, 2 * row + 1, column + 1, quantiles,
legend=i == 0)
figure.update_xaxes(title_text=panel["axis"], row=2 * row + 2,
col=column + 1)
return self._finish(
figure, title="%s: declustered data against the model along %s"
% (var.name, panels[0]["axis"]),
height=height or 150 + 380 * rows, width=width,
legend={"orientation": "h", "x": 0.5, "xanchor": "center",
"y": 1.0, "yanchor": "bottom"})
def _swath(self, figure, panel, row, column, quantiles, legend):
"""One component of `swath`: means and band above, support below."""
x, model = _prep.step_path(panel["lo"], panel["hi"],
panel["model_mean"])
_, data = _prep.step_path(panel["lo"], panel["hi"],
panel["data_mean"])
model_color, data_color = self._color(0, "model"), \
self._color(3, "data")
if panel["band_lo"] is not None:
_, low = _prep.step_path(panel["lo"], panel["hi"],
panel["band_lo"])
_, high = _prep.step_path(panel["lo"], panel["hi"],
panel["band_hi"])
figure.add_trace(_go.Scatter(
x=x, y=low, mode="lines", line={"width": 0},
showlegend=False, hoverinfo="skip"), row=row, col=column)
figure.add_trace(_go.Scatter(
x=x, y=high, mode="lines", line={"width": 0},
fill="tonexty", fillcolor="rgba(31, 111, 139, 0.2)",
name="model: %.0f-%.0f%% of realizations"
% (100 * quantiles[0], 100 * quantiles[1]),
showlegend=legend, hoverinfo="skip"), row=row, col=column)
figure.add_trace(_go.Scatter(
x=x, y=model, mode="lines",
line={"color": model_color, "width": 2.0},
name="model mean", showlegend=legend,
hovertemplate="model %{y:.4g}<extra></extra>"),
row=row, col=column)
figure.add_trace(_go.Scatter(
x=x, y=data, mode="lines",
line={"color": data_color, "width": 1.4},
name="data mean", showlegend=legend,
hovertemplate="data %{y:.4g}<extra></extra>"),
row=row, col=column)
figure.add_trace(_go.Scatter(
x=panel["centre"], y=panel["data_mean"], mode="markers",
marker={"size": 6, "color": data_color},
customdata=_np.stack([panel["data_count"],
panel["data_weight"]], axis=1),
showlegend=False,
hovertemplate="data %{y:.4g}<br>%{customdata[0]} samples, "
"%{customdata[1]:.1f} effective<extra></extra>"),
row=row, col=column)
figure.update_yaxes(title_text=panel["label"], row=row, col=column)
figure.add_trace(_go.Bar(
x=panel["centre"], y=panel["data_count"],
width=0.9 * (panel["hi"] - panel["lo"]),
marker={"color": data_color, "opacity": 0.5},
name="samples", showlegend=False,
hovertemplate="%{y} samples<extra></extra>"),
row=row + 1, col=column)
_, count = _prep.step_path(panel["lo"], panel["hi"],
panel["model_count"])
figure.add_trace(_go.Scatter(
x=x, y=count, mode="lines",
line={"color": model_color, "width": 1.0},
name="model cells", showlegend=False,
hovertemplate="%{y} cells<extra></extra>"),
row=row + 1, col=column, secondary_y=True)
figure.update_yaxes(title_text="samples", row=row + 1, col=column,
secondary_y=False)
figure.update_yaxes(title_text="cells", row=row + 1, col=column,
secondary_y=True, showgrid=False)
[docs]
def spread_check(self, bins=8, height=None, width=None) -> "_go.Figure":
"""
Whether the noise the model fitted is the noise the data has.
A residual holds two things at once -- how wrong the model was about
the ground, and how far the assay fell from the ground -- so it is
read against the two together. The band is the noise, the line the
whole claim, the points what the errors actually did. On the line is
calibrated, below it is hedging, above it is over-confident.
The level axis is what says which term is at fault. A warping bends,
so the noise grows with the value while the model's own uncertainty
does not: a shortfall widening with the grade is the noise, a flat one
is the posterior.
Only honest on data the model has not seen.
Parameters
----------
bins : int or sequence
How many bins, or where their edges are. A count gives
**equal-count** bins; pass `np.linspace(...)` for equal width.
"""
var = self._require_continuous("spread_check")
panels = _prep.spread_check(self.data, var.name, bins=bins)
labels = [panel["label"] for panel in panels]
rows, columns = _prep.grid_shape(len(panels))
figure = _subplots(rows=rows, cols=columns, subplot_titles=labels,
horizontal_spacing=0.1)
for i, panel in enumerate(panels):
row, column = divmod(i, columns)
self._spread(figure, panel, row + 1, column + 1, legend=i == 0)
figure.update_xaxes(title_text="predicted value", row=rows)
figure.update_yaxes(title_text="spread", col=1, rangemode="tozero")
return self._finish(
figure, title="%s: claimed spread against observed" % var.name,
height=height or 120 + 320 * rows, width=width,
# above the panels rather than beside them, as in `Explorer`: the
# entries are long and a column of them squeezes every panel
legend={"orientation": "h", "x": 0.5, "xanchor": "center",
"y": 1.0, "yanchor": "bottom"})
def _spread(self, figure, panel, row, column, legend):
"""One component of `spread_check`."""
x, noise = _prep.step_path(panel["lo"], panel["hi"], panel["noise"])
_, total = _prep.step_path(panel["lo"], panel["hi"], panel["total"])
claimed = self._color(0, "noise")
figure.add_trace(_go.Scatter(
x=x, y=noise, mode="lines", fill="tozeroy",
line={"color": claimed, "width": 0.8},
fillcolor="rgba(31, 111, 139, 0.25)",
name="claimed: measurement noise", showlegend=legend,
hovertemplate="noise %{y:.4g}<extra></extra>"), row=row, col=column)
figure.add_trace(_go.Scatter(
x=x, y=total, mode="lines", line={"color": claimed, "width": 1.8},
name="claimed: noise and uncertainty", showlegend=legend,
hovertemplate="claimed %{y:.4g}<extra></extra>"),
row=row, col=column)
figure.add_trace(_go.Scatter(
x=panel["centre"], y=panel["observed"], mode="markers",
marker={"size": 7, "color": self._color(3, "observed")},
error_y={"type": "data", "array": panel["observed_error"]},
error_x={"type": "data",
"array": panel["hi"] - panel["centre"],
"arrayminus": panel["centre"] - panel["lo"]},
customdata=_np.stack([panel["count"],
panel["observed"] / panel["total"]], axis=1),
name="observed: rms residual", showlegend=legend,
hovertemplate="observed %{y:.4g}<br>%{customdata[0]} samples"
"<br>observed/claimed %{customdata[1]:.2f}"
"<extra></extra>"), row=row, col=column)
[docs]
def variogram(self, n_lags=15, max_lag=None, direction=None,
tolerance=45.0, residuals=False, decluster=True,
height=None, width=None) -> "_go.Figure":
"""
The data's spatial structure, against the fan the simulations make.
The experimental semivariogram of the measurements, with one thin
curve per realization on the same pairs. A model that learned the
spatial structure scatters its fan *around* the data's curve; a
kernel too smooth sags below it at short lags, and a nugget fitted
into the range lifts it there. Neither shows in `accuracy` or
`spread_check`, which judge one location at a time.
The measurements carry the likelihood noise and the realizations do
not, so the fan is raised by what an independent error at each
location adds to a semivariogram, taken from `noise_variance`.
Without that the two curves are not the same quantity and every
model looks over-smooth by a nugget.
With `residuals=True` it is the variogram of `measured - predicted`
and the fan is dropped: structure left in the residuals is structure
the model missed -- honest on cross-validated predictions
(`models.cross_validate`).
Each panel's title carries `VS`, the variogram score of
:func:`geoml.metrics.variogram_score` over the same locations and
weights: the eye's verdict on the fan as one number, for comparing
two models without squinting. Lower is better, but only against
another model on the same data -- the score keeps a bias the curves
are corrected for, and never reaches zero.
Parameters
----------
n_lags : int
Number of equal-width lag bins.
max_lag : float, optional
Longest separation considered; half the bounding-box diagonal by
default.
direction : array-like, optional
Direction vector for a directional variogram; omnidirectional
when absent.
tolerance : float
Angular tolerance around `direction`, in degrees.
residuals : bool
Variogram of the residuals instead, without the fan.
decluster : bool or float
Weight pairs by cell-declustering weights, so that the curve
estimates the field's variogram rather than the sampling's.
`True` chooses the cell size, a number fixes it, `False` leaves
the pairs raw.
"""
var = self._require_continuous("variogram")
panels = _prep.variogram(
self.data, var.name, n_lags=n_lags, max_lag=max_lag,
direction=direction, tolerance=tolerance, residuals=residuals,
decluster=decluster)
labels = [panel["label"] if panel["score"] is None
else "%s (VS = %.3g)" % (panel["label"], panel["score"])
for panel in panels]
rows, columns = _prep.grid_shape(len(panels))
figure = _subplots(rows=rows, cols=columns, subplot_titles=labels,
horizontal_spacing=0.1)
for i, panel in enumerate(panels):
row, column = divmod(i, columns)
self._variogram_panel(figure, panel, row + 1, column + 1,
legend=i == 0)
figure.update_xaxes(title_text="lag distance", row=rows)
figure.update_yaxes(title_text="semivariance", col=1,
rangemode="tozero")
what = "residual variogram" if residuals else \
"variogram and simulation fan"
return self._finish(
figure, title="%s: %s" % (var.name, what),
height=height or 120 + 320 * rows, width=width,
legend={"orientation": "h", "x": 0.5, "xanchor": "center",
"y": 1.0, "yanchor": "bottom"})
def _variogram_panel(self, figure, panel, row, column, legend):
"""One component of `variogram`."""
fan = panel["realizations"]
if fan is not None:
for r in range(fan.shape[0]):
figure.add_trace(_go.Scatter(
x=panel["lag"], y=fan[r], mode="lines",
line={"color": self._color(0, "realizations"),
"width": 0.7},
opacity=0.25, name="realizations",
legendgroup="realizations",
showlegend=legend and r == 0,
hoverinfo="skip"), row=row, col=column)
figure.add_trace(_go.Scatter(
x=[panel["lag"][0], panel["lag"][-1]],
y=[panel["sill"], panel["sill"]], mode="lines",
line={"color": self._color(2, "sill"), "width": 1.0,
"dash": "dash"},
name="data variance", showlegend=legend,
hovertemplate="variance %{y:.4g}<extra></extra>"),
row=row, col=column)
figure.add_trace(_go.Scatter(
x=panel["lag"], y=panel["data"], mode="lines+markers",
marker={"size": 6, "color": self._color(3, "data")},
line={"color": self._color(3, "data"), "width": 1.6},
customdata=panel["count"],
name="data", showlegend=legend,
hovertemplate="lag %{x:.4g}<br>semivariance %{y:.4g}"
"<br>%{customdata} pairs<extra></extra>"),
row=row, col=column)
[docs]
def grade_tonnage(self, component=None, density=None, cutoffs=30,
max_uncertainty=None, log_mass=False, height=None,
width=None) -> "_go.Figure":
"""
How much material clears each cut-off, and how good it is.
Tonnage falls and grade rises as the cut-off climbs, and where the two
cross is the question the curve is drawn to answer. Simulations are
carried through one by one and drawn as a family, with the median over
them picked out: the spread between the thin lines is what the model
does not know about the answer.
Two scales share one frame, so each gets gridlines in the colour of its
own curve -- a reader following the grade curve to a grey line would
otherwise read a tonnage off it.
Parameters
----------
component : str
Which grade, when the variable is a vector one.
density : float or str
A number, a metadata column, or a `ContinuousVariable` -- and in
that last case its simulations are matched with the grade's, one to
one. Without a density the curve is in volume.
cutoffs : int or array-like
The grades to cut at, or how many of them to spread evenly across
the range of the data.
max_uncertainty : float
Leave out the blocks the model doubts more than this, reading the
column named when the Interactive was built.
log_mass : bool
Put the tonnage on a logarithmic scale. Most of a deposit clears
the low cut-offs, so on a linear axis the high ones are a flat
line along the bottom and the spread between the realizations
there -- which is where the decision usually is -- cannot be seen.
A cut-off that nothing clears has no logarithm and drops out.
"""
var = self._require_continuous("grade_tonnage")
name = var.name
if component is not None:
name = component
elif var.length > 1:
raise ValueError(
"%r holds %d components and a cut-off applies to one grade; "
"name one with component= (%s)"
% (var.name, var.length, _prep.component_names(var)))
curves = _prep.grade_tonnage(
self.data, name, density, cutoffs,
uncertainty=self.uncertainty, max_uncertainty=max_uncertainty)
cutoff = curves["cutoff"]
many = curves["tonnage"].shape[1] > 1
figure = _go.Figure()
for key, index, axis in (("tonnage", 0, "y"), ("grade", 1, "y2")):
if many:
# every realization in one trace, cut apart by the gaps a NaN
# leaves: a hundred traces of a hundred points each is a
# hundred legend entries and a far heavier page
x, y = self._family(cutoff, curves[key])
figure.add_trace(_go.Scatter(
x=x, y=y, mode="lines", yaxis=axis, showlegend=False,
line={"color": _style.color(index), "width": 0.6},
opacity=0.3, hoverinfo="skip"))
figure.add_trace(_go.Scatter(
x=cutoff, y=_np.median(curves[key], axis=1), mode="lines",
yaxis=axis, name=key,
line={"color": _style.color(index), "width": 2.5},
hovertemplate="cut-off %{x:.4g}<br>" + key +
" %{y:.4g}<extra></extra>"))
title = "Grade and tonnage"
if curves["kept"] < curves["total"]:
# an uncertainty handed over as values has no name to give
named = self.uncertainty \
if isinstance(self.uncertainty, str) else "uncertainty"
title += " (%d of %d blocks, %s ≤ %g)" % (
curves["kept"], curves["total"], named, max_uncertainty)
graded = name if curves["unit"] is None \
else "%s, %s" % (name, curves["unit"])
in_units = "" if curves["unit"] is None \
else " (%s)" % curves["unit"]
return self._finish(
figure, title=title, height=height or 470, width=width,
xaxis={"title": {"text": "cut-off grade (%s)" % graded}},
# only the tonnage: the grade axis spans one order of magnitude at
# most and a log scale would say nothing
yaxis={"title": {"text": curves["extent"] + " above the cut-off",
"font": {"color": _style.color(0)}},
"tickfont": {"color": _style.color(0)},
"type": "log" if log_mass else "linear",
"gridcolor": self._faint(_style.color(0))},
yaxis2={"title": {"text": "mean grade above the cut-off" + in_units,
"font": {"color": _style.color(1)}},
"tickfont": {"color": _style.color(1)},
"gridcolor": self._faint(_style.color(1)),
"overlaying": "y", "side": "right", "showgrid": True},
hovermode="x unified",
legend={"x": 0.98, "y": 0.5, "xanchor": "right"})
@staticmethod
def _family(x, curves):
"""A bundle of curves as one trace, separated by gaps.
A NaN breaks a plotly line, so every realization can be laid end to end
in a single trace instead of costing one of its own.
"""
n_cutoffs, n_curves = curves.shape
gap = _np.full([1, n_curves], _np.nan)
stacked = _np.vstack([curves, gap])
repeated = _np.concatenate([_np.asarray(x, dtype=float), [_np.nan]])
return (_np.tile(repeated, n_curves),
stacked.T.reshape(-1))
@staticmethod
def _faint(color):
"""A hex colour, watered down to something a gridline can be."""
red, green, blue = (int(color[i:i + 2], 16) for i in (1, 3, 5))
return "rgba(%d,%d,%d,0.25)" % (red, green, blue)
[docs]
def dispersion_by_support(self, component=None, kind="box", alpha=0.2,
most=5000, size=5, height=None,
width=None) -> "_go.Figure":
"""
How much the ground varies inside a block, against the block's size.
Only a `BlockSet3D` has blocks of several sizes. Every block is merged
into its parent, level by level up to the coarsest, so each size the
lattice has holds a distribution: the within-block standard deviation
of every block of that size, the finest on the left. The line joins
each size's root mean square, the dispersion of the ground within
blocks of that size. A parent is put together from the blocks inside
it, realization by realization, and never predicted.
A block the refinement left whole reads its dispersion off its own
sub-blocks, one position per child, and one put together from its
descendants off all of theirs. Fewer positions see less of the
ground, so at one size a block left whole reads lower than a split
block over the same ground; `kind="jitter"` colours every block by
how many times the refinement split it, which is where that shows.
Each size's label gives how many blocks it holds and the share of
the volume they cover: the fine sizes exist only where the
refinement went, so the distributions are of different ground.
A block put together from others is a row of no container, so the
points carry no row for a dashboard to link on.
Parameters
----------
component : str
One component of a vector variable, drawn on its own.
kind : str
`"box"`, `"violin"` or `"jitter"`, the last one point per block,
coloured by how many times the refinement split it.
alpha : float
How opaque each point is, for `kind="jitter"`. Low by default:
a size can hold thousands of blocks, and where they pile up is
what there is to see.
most : int
About how many blocks of each size `kind="jitter"` draws, taken
by striding through them; the box and the violin use them all.
size : float
Point size, for `kind="jitter"`.
See Also
--------
geoml.plots.prepare.dispersion_by_support : the numbers drawn here.
"""
self._check_kind(kind, ("box", "violin", "jitter"))
var = self._require_continuous("dispersion_by_support")
panels = _prep.dispersion_by_support(self.data, var.name,
component=component)
rows, columns = _prep.grid_shape(len(panels))
figure = _subplots(rows=rows, cols=columns,
subplot_titles=[panel["label"] for panel in panels],
horizontal_spacing=0.08)
for i, panel in enumerate(panels):
row, column = divmod(i, columns)
self._support(figure, panel, kind, most, size, alpha, row + 1,
column + 1, legend=i == 0)
return self._finish(
figure, title="%s: dispersion by block size" % var.name,
height=height or 140 + 380 * rows, width=width)
def _support(self, figure, panel, kind, most, size, alpha, row, column,
legend):
"""One component of `dispersion_by_support`."""
sizes = panel["sizes"]
position = _np.arange(len(sizes), dtype=float)
filled = [i for i, entry in enumerate(sizes) if entry["count"]]
colour = _style.color(0)
# a strip is drawn per depth below, not per size
for i in ([] if kind == "jitter" else filled):
values = sizes[i]["deviation"]
common = {"x": _np.full(len(values), position[i]), "y": values,
"name": _prep.support_tick(sizes[i])[0],
"showlegend": False}
if kind == "box":
trace = _go.Box(width=0.5, boxpoints="outliers",
marker={"color": colour, "size": 3},
line={"color": colour},
fillcolor=self._faint(colour), **common)
else:
trace = _go.Violin(width=0.7, points=False,
line={"color": colour},
fillcolor=self._faint(colour),
meanline={"visible": False},
box={"visible": True}, **common)
figure.add_trace(trace, row=row, col=column)
if kind == "jitter":
# the blocks split least are the fewest at the coarse sizes and
# the ones the colours are for, so they are drawn last, on top;
# the legend keeps them first
for depth, x, y in reversed(_prep.support_strip(sizes, most)):
label = _prep.split_label(depth)
colour = self._color(depth, label)
figure.add_trace(_go.Scatter(
x=x, y=y, mode="markers",
marker={"size": size, "opacity": alpha,
"color": colour, "line": {"width": 0}},
name=label, legendgroup=label, showlegend=False,
hovertemplate="%{y:.4g}<extra></extra>"),
row=row, col=column)
if legend:
# the points are faint on purpose, and a legend draws
# them as they are; an empty trace in the same group
# carries the key at full strength, and clicking it
# still hides the strip
figure.add_trace(_go.Scatter(
x=[None], y=[None], mode="markers",
marker={"size": 8, "color": colour},
name=label, legendgroup=label, legendrank=depth,
hoverinfo="skip"), row=row, col=column)
self._line(figure, position[filled],
[sizes[i]["rms"] for i in filled], INK, row, column,
width=1.4, name="root mean square", legend=legend)
figure.update_xaxes(
tickvals=position,
ticktext=["<br>".join(_prep.support_tick(entry))
for entry in sizes],
range=[-0.6, len(sizes) - 0.4], title_text="block size",
row=row, col=column)
# from zero, so a change with size reads at its true scale
figure.update_yaxes(title_text="within-block standard deviation",
rangemode="tozero", row=row, col=column)
[docs]
def volume_dispersion(self, shells: "_data.MeshSet", kind: str = "box",
relative: bool = False, alpha: float = 0.2,
size: float = 6, height=None,
width=None) -> "_go.Figure":
"""
How much the realizations' meshes vary in volume, against the
prediction's.
For every cut-off or category of a `MeshSet` built with its
realizations, the distribution of the realizations' mesh volumes,
with the prediction's marked. The prediction is smoother than any
realization, so its mesh tends to hold less volume at a high cut-off
and more at a low one; how far it sits from the middle of the
distribution is how far one mesh misreports the volume, and the
spread is what no single mesh can show. The figure reads what the
set measured as it was made, and loads no mesh.
A mesh volume is a number about a whole realization, a row of no
container, so the points carry no row for a dashboard to link on.
Parameters
----------
shells
A `MeshSet` built with `simulations=True`.
kind : str
`"box"`, `"violin"` or `"jitter"`, the last one point per
realization.
relative : bool
Whether to divide every volume by the prediction's.
alpha : float
How opaque each point is, for `kind="jitter"`.
size : float
Point size, for `kind="jitter"`.
See Also
--------
geoml.plots.prepare.volume_dispersion : the numbers drawn here.
geoml.data.MeshSet.volume_dispersion : the same, as a table.
"""
self._check_kind(kind, ("box", "violin", "jitter"))
panel = _prep.volume_dispersion(shells, relative=relative)
values = panel["values"]
position = _np.arange(len(values), dtype=float)
colour = _style.color(0)
figure = _go.Figure()
shown = False
for i, held in enumerate(values):
if held.size == 0:
continue
common = {"x": _np.full(held.size, position[i]), "y": held,
"name": "realizations", "legendgroup": "realizations",
"showlegend": not shown}
if kind == "box":
figure.add_trace(_go.Box(
width=0.5, boxpoints="outliers",
marker={"color": colour, "size": 3},
line={"color": colour}, fillcolor=self._faint(colour),
**common))
elif kind == "violin":
figure.add_trace(_go.Violin(
width=0.7, points=False, line={"color": colour},
fillcolor=self._faint(colour),
meanline={"visible": False}, box={"visible": True},
**common))
else:
common["x"] = position[i] + _prep.jitter(held.size)
figure.add_trace(_go.Scatter(
mode="markers",
marker={"size": size, "opacity": alpha, "color": colour,
"line": {"width": 0}},
hovertemplate="%{y:.6g}<extra></extra>", **common))
shown = True
figure.add_trace(_go.Scatter(
x=position, y=panel["prediction"], mode="markers",
marker={"symbol": "diamond", "size": 9, "color": INK},
name="prediction", hovertemplate="%{y:.6g}<extra></extra>"))
if relative:
figure.add_hline(y=1.0, line={"color": INK, "width": 1,
"dash": "dot"})
figure.update_xaxes(tickvals=position, ticktext=panel["labels"],
range=[-0.6, len(values) - 0.4],
title_text=panel["keys"])
# from zero, so a spread reads at its true scale
figure.update_yaxes(title_text=panel["axis"], rangemode="tozero")
return self._finish(figure, title=panel["title"],
height=height or 460, width=width)
[docs]
def connectivity(self, shells: "_data.MeshSet", height=None,
width=None) -> "_go.Figure":
"""
Whether the ground above each cut-off holds together.
Two panels for a `MeshSet`: the share of each mesh's volume in its
largest piece, and how many pieces it is in. Read along the
cut-offs the first is a connectivity curve -- where it drops, the
ground breaks into pods -- and where the set holds realizations
their P10 to P90 is drawn as a band, their median dashed.
Parameters
----------
shells
The set.
See Also
--------
geoml.plots.prepare.connectivity : the numbers drawn here.
"""
panel = _prep.connectivity(shells)
x = panel["x"]
mode = "lines+markers" if panel["numeric"] else "markers"
figure = _subplots(rows=1, cols=2, horizontal_spacing=0.12,
subplot_titles=["largest piece's share",
"pieces"])
colour = _style.color(0)
if panel["band"] is not None:
low, high = panel["band"]
if panel["numeric"]:
figure.add_trace(_go.Scatter(
x=_np.concatenate([x, x[::-1]]),
y=_np.concatenate([high, low[::-1]]), fill="toself",
fillcolor=self._faint(colour), line={"width": 0},
name="realizations, P10–P90", hoverinfo="skip"),
row=1, col=1)
else:
figure.add_trace(_go.Scatter(
x=x, y=(low + high) / 2, mode="markers",
error_y={"type": "data", "symmetric": False,
"array": high - (low + high) / 2,
"arrayminus": (low + high) / 2 - low,
"color": colour},
marker={"size": 1, "color": colour},
name="realizations, P10–P90"), row=1, col=1)
for column, median in ((1, panel["median"]),
(2, panel["pieces_median"])):
figure.add_trace(_go.Scatter(
x=x, y=median, mode=mode,
line={"color": colour, "dash": "dash"},
marker={"color": colour, "symbol": "line-ew-open"},
name="realizations, P50", legendgroup="median",
showlegend=column == 1), row=1, col=column)
for column, series in ((1, panel["largest"]), (2, panel["pieces"])):
figure.add_trace(_go.Scatter(
x=x, y=series, mode=mode, line={"color": INK},
marker={"color": INK}, name="prediction",
legendgroup="prediction", showlegend=column == 1),
row=1, col=column)
figure.update_yaxes(range=[0.0, 1.05], row=1, col=1)
figure.update_yaxes(rangemode="tozero", row=1, col=2)
for column in (1, 2):
if panel["numeric"]:
figure.update_xaxes(title_text=panel["keys"], row=1,
col=column)
else:
figure.update_xaxes(tickvals=x, ticktext=panel["labels"],
title_text=panel["keys"], row=1,
col=column)
return self._finish(figure, title=panel["title"],
height=height or 440, width=width)
[docs]
def section(self, shells: "_data.MeshSet", axis, value: float,
component: "str | None" = None,
resolution: "float | None" = None, height=None,
width=None) -> "_go.Figure":
"""
Every mesh of a set where it crosses a plane, over the model.
The lines each mesh draws on a plane across one axis, a colour per
cut-off or category, over the prediction of the continuous variable
this selection names -- a grade under its own shells, or under a
rock model's contacts. Without a continuous variable, the lines
alone.
Parameters
----------
shells
The set.
axis
The coordinate held fixed, by index or by label.
value
Where along it the plane sits.
component : str
For a vector variable, the component to draw beneath.
resolution : float
The spacing the prediction is sampled at on the plane.
See Also
--------
geoml.plots.prepare.mesh_section : the numbers drawn here.
geoml.data.MeshSet.section : the lines, as arrays.
"""
beneath = self.continuous
if beneath is not None and component is not None:
beneath = beneath.components[component]
panel = _prep.mesh_section(shells, axis, value, variable=beneath,
resolution=resolution)
figure = _go.Figure()
image = panel["image"]
if image is not None:
rows, columns = image["values"].shape
u0, u1, v0, v1 = image["extent"]
figure.add_trace(_go.Heatmap(
z=image["values"],
x=_np.linspace(u0, u1, columns + 1)[:-1]
+ (u1 - u0) / columns / 2,
y=_np.linspace(v0, v1, rows + 1)[:-1]
+ (v1 - v0) / rows / 2,
colorscale=self.cmap, hoverongaps=False,
colorbar={"title": {"text": image["label"]}},
name=image["label"]))
for i, (label, lines) in enumerate(panel["lines"].items()):
if not lines:
continue
# one trace per mesh, its polylines joined through gaps
u = _np.concatenate([_np.append(line[:, 0], _np.nan)
for line in lines])
v = _np.concatenate([_np.append(line[:, 1], _np.nan)
for line in lines])
figure.add_trace(_go.Scatter(
x=u, y=v, mode="lines", name=label,
line={"color": self._color(i, label), "width": 2},
hoverinfo="name"))
figure.update_xaxes(title_text=panel["axes"][0])
figure.update_yaxes(title_text=panel["axes"][1], scaleanchor="x",
scaleratio=1)
return self._finish(figure, title=panel["title"],
height=height or 560, width=width,
legend_title_text=panel["keys"])
# ------------------------------------------------------------------ #
# the matrix
# ------------------------------------------------------------------ #
def _matrix(self, values, labels, series, measured, rows, kind="scatter",
alpha=0.7, bins=60, log_counts=False, upper=None, size=5,
density=False):
"""A scatter matrix: pairs below the diagonal, distributions along it.
The upper triangle is the lower one transposed, so it is left out
unless something else is asked for it -- half the ink for all of the
information.
"""
if kind == "hist2d" and len(series) > 1:
# counts make one surface, and several laid over each other read
# as none of them, so the categories are pooled here
series = [(None, None)]
n = len(labels)
figure = _subplots(rows=n, cols=n, shared_xaxes=True,
horizontal_spacing=0.02, vertical_spacing=0.02)
# a category belongs in the legend once, not once per panel; the rest
# of its traces join it by `legendgroup`, so clicking the entry still
# takes the category out of the whole matrix
named = False
for row in range(n):
for column in range(n):
if column > row:
if upper is None:
continue
self._upper(figure, values[measured, column],
values[measured, row], upper, bins,
log_counts, row, column, n)
continue
for j, (name, mask) in enumerate(series):
keep = measured if mask is None else mask
color = self._color(j, name)
legend = name is not None and not named
if row == column:
edges = _np.histogram_bin_edges(values[keep, row],
bins=20)
self._bars(figure, values[keep, row], rows[keep],
edges, color, name, legend,
row + 1, column + 1, density=density)
else:
self._pair(figure, values[keep, column],
values[keep, row], rows[keep], kind, alpha,
size, bins, log_counts, color, name,
legend, row + 1, column + 1)
named = True
self._label_edges(figure, labels, n)
self._match_rows(figure, n, upper)
return self._finish(figure, barmode="overlay")
def _pair(self, figure, x, y, rows, kind, alpha, size, bins, log_counts,
color, name, legend, row, column):
"""One off-diagonal panel: a cloud of points, or counts in cells."""
if kind == "hist2d":
self._cells(figure, x, y, bins, log_counts, row, column)
else:
self._points(figure, x, y, rows, size, alpha, color, name,
legend, row, column)
def _upper(self, figure, x, y, upper, bins, log_counts, row, column, n):
"""
The upper triangle: the same pair, told the other way.
The lower triangle answers "which category is this point"; up here the
categories are pooled on purpose, so the question becomes "where does
the data sit, all of it together".
"""
if upper == "hist2d":
self._cells(figure, x, y, bins, log_counts, row + 1, column + 1)
elif upper == "density":
self._density(figure, x, y, row + 1, column + 1)
elif upper == "correlation":
# An axis carrying no trace is dropped on the way to the drawn
# figure, and an annotation pinned to it goes with it -- so the
# number needs a panel to sit in before it can be written there.
figure.add_trace(
_go.Scatter(x=[None], y=[None], mode="markers",
showlegend=False, hoverinfo="skip"),
row=row + 1, col=column + 1)
self._correlation(figure, x, y, column, row, n)
figure.update_xaxes(showticklabels=False, showgrid=False,
row=row + 1, col=column + 1)
figure.update_yaxes(showticklabels=False, showgrid=False,
row=row + 1, col=column + 1)
else:
raise ValueError(
"upper must be 'hist2d', 'density' or 'correlation'; got %r"
% upper)
@staticmethod
def _statistics(figure, values, tallest, edges, row, column, panel):
"""A panel's summary statistics, in the corner its bars leave free.
`panel` is about how many pixels the panel takes across and down.
Plotly lays the text out in the page, so the share of the panel the
box covers is worked out from the font: a monospace character about
0.6 of the size across, a line about 1.3 of it down, and the frame's
padding around them.
"""
lines = _prep.statistics_lines(_prep.summary_statistics(values))
side = _prep.statistics_side(tallest)
font = 9
across = (0.6 * font * max(len(line) for line in lines) + 8) \
/ panel[0]
down = (1.3 * font * len(lines) + 8) / panel[1]
span = float(edges[-1] - edges[0])
if side == "right":
low, high = edges[-1] - (across + 0.02) * span, edges[-1]
else:
low, high = edges[0], edges[0] + (across + 0.02) * span
top = _prep.statistics_top(tallest, edges, low, high, down + 0.02)
# the columns are lined up with spaces, which a label collapses
# unless they cannot break
text = "<br>".join(line.replace(" ", "\u00a0") for line in lines)
figure.add_annotation(
xref="x domain", yref="y domain", row=row, col=column,
x=0.98 if side == "right" else 0.02, y=0.98, xanchor=side,
yanchor="top", align="left", text=text, showarrow=False,
font={"size": font, "family": "Courier New, monospace",
"color": INK},
bgcolor="rgba(255,255,255,0.8)", bordercolor=INK,
borderwidth=0.6, borderpad=3)
figure.update_yaxes(range=[0.0, top], row=row, col=column)
def _correlation(self, figure, x, y, column, row, n, corner=False):
"""What the eye is being asked about, as a number."""
correlation = _np.corrcoef(x, y)[0, 1]
suffix = self._axis(row + 1, column + 1, n)
place = {"x": 0.05, "y": 0.95, "xanchor": "left", "yanchor": "top",
"text": "r = %.2f" % correlation, "font": {"size": 9}} \
if corner else \
{"x": 0.5, "y": 0.5, "xanchor": "center", "yanchor": "middle",
"text": "%.2f" % correlation,
# the stronger it is, the larger it reads
"font": {"size": 10 + 14 * abs(correlation),
"color": _style.color(0) if correlation >= 0
else _style.color(3)}}
figure.add_annotation(
xref="x%s domain" % suffix, yref="y%s domain" % suffix,
showarrow=False, bgcolor="rgba(255,255,255,0.5)",
bordercolor=INK if corner else None,
borderwidth=0.6 if corner else 0, borderpad=2, **place)
def _normal(self, figure, values, row):
"""The normal of the same mean and spread, over the histogram."""
curve = _prep.normal_curve(values, _np.min(values), _np.max(values))
if curve is None:
return
self._line(figure, curve[0], curve[1], INK, row + 1, row + 1,
width=1.2)
def _axes_of_variation(self, figure, analysis, x_column, y_column, count):
"""
The principal components, drawn back onto the data they came from.
A component is a direction in the space of the measurements, so in a
panel showing two of them it is the pair of entries belonging to those
two. Each line runs one standard deviation of that component either
side of the mean, which puts the components in the data's own units:
the first is longest because it carries the most variance, and a line
lying flat along an axis means that measurement moves on its own. The
sign of a component means nothing, hence a line rather than an arrow.
"""
mean = analysis["mean"]
loadings = analysis["loadings"]
deviations = _np.sqrt(_np.maximum(analysis["eigenvalues"], 0.0))
for k in range(min(count, loadings.shape[1])):
step = loadings[:, k] * deviations[k]
x = [mean[x_column] - step[x_column],
mean[x_column] + step[x_column]]
y = [mean[y_column] - step[y_column],
mean[y_column] + step[y_column]]
figure.add_trace(
_go.Scatter(x=x, y=y, mode="lines+text",
text=["", "PC%d" % (k + 1)],
textposition="top right",
textfont={"size": 8, "color": INK},
line={"color": INK, "width": 1.4},
opacity=0.85, showlegend=False,
hovertemplate="PC%d<extra></extra>" % (k + 1)),
row=y_column + 1, col=x_column + 1)
def _loadings(self, figure, loadings, labels, x_component, y_component,
scores, n):
"""An arrow per original column, over the cloud of scores.
Each axis is stretched to its own component's spread. A single scale
for the whole figure would suit the first component, which carries most
of the variance, and send every arrow off the edge of the others.
Two annotations per column, not one. Plotly draws an annotation's text
at the *tail* of its arrow and points the head at the position given,
so a single annotation carrying both would pile every name on the
origin the arrows all start from. The arrow is drawn with no text, and
the name is placed at the head, which is where the eye goes looking
for it.
"""
x_reach = _np.max(_np.abs(scores[:, x_component])) or 1.0
y_reach = _np.max(_np.abs(scores[:, y_component])) or 1.0
suffix = self._axis(y_component + 1, x_component + 1, n)
axes = {"xref": "x%s" % suffix, "yref": "y%s" % suffix}
for i, label in enumerate(labels):
x = loadings[i, x_component] * x_reach
y = loadings[i, y_component] * y_reach
figure.add_annotation(
x=x, y=y, ax=0, ay=0, text="",
axref="x%s" % suffix, ayref="y%s" % suffix,
showarrow=True, arrowhead=2, arrowsize=1.2, arrowwidth=0.9,
arrowcolor=INK, opacity=0.9, **axes)
figure.add_annotation(
x=x, y=y, text=label, showarrow=False,
font={"size": 8, "color": INK},
bgcolor="rgba(255,255,255,0.5)", bordercolor=INK,
borderwidth=0.6, borderpad=2,
xanchor="left", yanchor="bottom", **axes)
@staticmethod
def _label_edges(figure, labels, n):
"""
The names, written once along the bottom and once down the side.
A panel away from the edges keeps no tick labels of its own: its scales
are the ones already written along the edges of the matrix, and
repeating them in every cell is what makes a scatter matrix unreadable.
The x axes were shared down each column when the figure was made, which
already clears theirs.
"""
for i, label in enumerate(labels):
figure.update_xaxes(title_text=label, row=n, col=i + 1)
figure.update_yaxes(title_text=label, row=i + 1, col=1)
for row in range(1, n + 1):
for column in range(2, n + 1):
figure.update_yaxes(showticklabels=False, row=row, col=column)
def _match_rows(self, figure, n, upper):
"""
Zooming one panel takes its whole row and column with it.
`make_subplots` was asked to share the x axis down each column, which
is what puts one variable on one scale wherever it is drawn
horizontally. The same has to hold across a row -- except for the
diagonal, whose vertical axis is a count and not the variable at all,
and which would be flattened out of existence by being matched to it.
"""
for row in range(1, n + 1):
columns = [c for c in range(1, n + 1)
if c != row and (c < row or upper is not None)]
if len(columns) < 2:
continue
reference = "y%s" % self._axis(row, columns[0], n)
for column in columns[1:]:
figure.update_yaxes(matches=reference, row=row, col=column)
# ------------------------------------------------------------------ #
# predicted against measured
# ------------------------------------------------------------------ #
def _agreement(self, figure, true, predicted, rows, kind, alpha, bins,
log_counts, size, row, column, legend=True, left_out=0):
"""Predicted against measured, with the line they would sit on."""
self._pair(figure, true, predicted, rows, kind, alpha, size, bins,
log_counts, _style.color(0), "measured", legend, row,
column)
low = min(_np.min(true), _np.min(predicted))
high = max(_np.max(true), _np.max(predicted))
margin = 0.05 * (high - low) if high > low else 1.0
self._line(figure, [low, high], [low, high], REFERENCE, row, column,
width=1.0, dash="dash")
# the same limits on both axes, so the line runs at 45 degrees and a
# cloud leaning off it is leaning visibly
window = [low - margin, high + margin]
figure.update_xaxes(range=window, row=row, col=column)
figure.update_yaxes(range=window, row=row, col=column)
if left_out:
# a trimmed panel is not showing everything, and says so -- in
# the corner a smoothing model leaves empty, since it never
# gives the highest measurements the lowest predictions
figure.add_annotation(
xref="x domain", yref="y domain", row=row, col=column,
x=0.95, y=0.05, xanchor="right", yanchor="bottom",
text="outliers left out: %d of %d"
% (left_out, len(true) + left_out),
font={"size": 9}, showarrow=False,
bgcolor="rgba(255,255,255,0.5)", bordercolor=INK,
borderwidth=0.6, borderpad=2)
def _joint(self, true, predicted, rows, label, kind, alpha, bins,
log_counts, size, height, width, left_out=0):
"""One variable, with the two distributions along the sides."""
figure = _subplots(
rows=2, cols=2, column_widths=[0.82, 0.18],
row_heights=[0.18, 0.82], shared_xaxes=True, shared_yaxes=True,
horizontal_spacing=0.02, vertical_spacing=0.02)
self._agreement(figure, true, predicted, rows, kind, alpha, bins,
log_counts, size, 2, 1, legend=False,
left_out=left_out)
self._bars(figure, true, rows,
_np.histogram_bin_edges(true, bins=25),
_style.color(0), "measured", False, 1, 1)
figure.add_trace(
_go.Histogram(
y=predicted, customdata=rows, nbinsy=25,
marker={"color": _style.color(0)}, opacity=0.65,
unselected={"marker": {"opacity": 0.15}},
showlegend=False),
row=2, col=2)
figure.update_xaxes(title_text="measured", row=2, col=1)
figure.update_yaxes(title_text="predicted", row=2, col=1)
figure.update_yaxes(title_text="count", row=1, col=1)
figure.update_xaxes(title_text="count", row=2, col=2)
return self._finish(figure, title=label, height=height or 620,
width=width or 640, showlegend=False,
barmode="overlay")