diff --git a/.gitmodules b/.gitmodules index f6d55e3..285aa68 100644 --- a/.gitmodules +++ b/.gitmodules @@ -1,3 +1,3 @@ [submodule "x4i3_tools"] path = x4i3_tools - url = git@github.com:afedynitch/x4i3_tools.git + url = https://github.com/afedynitch/x4i3_tools.git diff --git a/examples/examples_2023_release/ang_frame_conversion_and_json.ipynb b/examples/examples_2023_release/ang_frame_conversion_and_json.ipynb index ea5d134..a13a45d 100644 --- a/examples/examples_2023_release/ang_frame_conversion_and_json.ipynb +++ b/examples/examples_2023_release/ang_frame_conversion_and_json.ipynb @@ -206,6 +206,8 @@ "name": "stdout", "output_type": "stream", "text": [ + "Found subentry 21782031 with the following columns:\n", + "['EN', 'EN-RSL-FW', 'E-LVL-MIN', 'E-LVL-MAX', 'ANG', 'DATA', 'ERR-T']\n", "Found subentry 21782032 with the following columns:\n", "['EN', 'EN-RSL-FW', 'ANG', 'DATA', 'ERR-T']\n" ] @@ -257,8 +259,8 @@ { "data": { "text/plain": [ - "array([ 70.23, 85.24, 90.24, 95.24, 105.2 , 115.2 , 125.2 , 135.2 ,\n", - " 145.1 , 155.1 ])" + "array([ 20.08, 30.12, 40.15, 50.18, 60.21, 70.23, 80.24, 85.24,\n", + " 90.24, 95.24, 105.2 , 115.2 , 125.2 , 135.2 , 145.1 , 155.1 ])" ] }, "execution_count": 7, diff --git a/examples/examples_2023_release/dataset_curation_tutorial.ipynb b/examples/examples_2023_release/dataset_curation_tutorial.ipynb index 1ca15b6..e4c8bed 100644 --- a/examples/examples_2023_release/dataset_curation_tutorial.ipynb +++ b/examples/examples_2023_release/dataset_curation_tutorial.ipynb @@ -785,6 +785,24 @@ " (DATA-ERR).Data-Point Reader Uncertainty.\n", " (ANG-ERR).Data-Point Reader Uncertainty.\n", "Entry: O0253\n", + "O0253017 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253018 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253019 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253020 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253021 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253022 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253023 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253024 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253025 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", "O0253026 : Ambiguous statistical error labels:\n", "ERR-1, ERR-2, DATA-ERR\n", "ERR-ANALYS (ERR-1) Uncertainty of Corrections on Carbon and Oxygen\n", @@ -802,6 +820,15 @@ " Absolute Uncertainties are Approximately 5%.\n", " (ANG-ERR).Data-Point Reader Uncertainty.\n", "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", "Entry: O0302\n", "O0302004 : Ambiguous statistical error labels:\n", "DATA-ERR1, DATA-ERR2, ERR-DIG\n", @@ -993,6 +1020,32 @@ " DETERMINATIONOF THE SUPERFICIAL OF TARGET.\n", "\n", "Entry: O0253\n", + "O0253003 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253004 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253005 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253006 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253007 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253008 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253009 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253010 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253011 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253012 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253013 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253014 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", + "O0253015 : Ambiguous statistical error labels:\n", + "ERR-1, ERR-2, DATA-ERR\n", "O0253016 : Ambiguous statistical error labels:\n", "ERR-1, ERR-2, DATA-ERR\n", "ERR-ANALYS (ERR-1) Uncertainty of Corrections on Carbon and Oxygen\n", @@ -1010,6 +1063,19 @@ " Absolute Uncertainties are Approximately 5%.\n", " (ANG-ERR).Data-Point Reader Uncertainty.\n", "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", + "ERR-ANALYS (DATA-ERR).Data-Point Reader Uncertainty.\n", "Entry: O0382\n", "O0382002 : Ambiguous statistical error labels:\n", "DATA-ERR, ERR-T\n", diff --git a/src/exfor_tools/curate.py b/src/exfor_tools/curate.py index 9dfd54c..569194a 100644 --- a/src/exfor_tools/curate.py +++ b/src/exfor_tools/curate.py @@ -405,9 +405,15 @@ def cross_reference_entry_systematic_err( def print_failed_parses(failed_parses): for k, v in failed_parses.items(): print(f"Entry: {k}") - print(v.failed_parses[k][0], " : ", v.failed_parses[k][1]) + # report every failed subentry, not just the last one recorded per entry + failures = getattr(v, "failed_parses_by_subentry", None) or dict( + [v.failed_parses[k]] + ) + for subentry, e in failures.items(): + print(subentry, " : ", e) print(v.err_analysis) - print(v.subentry_err_analysis[v.failed_parses[k][0]]) + for subentry in failures: + print(v.subentry_err_analysis[subentry]) def plot_measurements( diff --git a/src/exfor_tools/distribution.py b/src/exfor_tools/distribution.py index 4644d13..416e8d8 100644 --- a/src/exfor_tools/distribution.py +++ b/src/exfor_tools/distribution.py @@ -237,9 +237,11 @@ def determine_error_categories( statistical_err = [] for i, label in enumerate(statistical_err_labels): - if label in y_err_labels: - index = y_err_labels.index(label) - statistical_err.append(y_errs[index]) + # a label may occur more than once (e.g. two DATA-ERR columns, one in + # per-cent and one absolute); take every occurrence + for index, candidate in enumerate(y_err_labels): + if candidate == label: + statistical_err.append(y_errs[index]) if statistical_err == []: statistical_err = [np.zeros((rows))] diff --git a/src/exfor_tools/exfor_entry.py b/src/exfor_tools/exfor_entry.py index 82d8bd8..4e6dcbc 100644 --- a/src/exfor_tools/exfor_entry.py +++ b/src/exfor_tools/exfor_entry.py @@ -173,6 +173,10 @@ def __init__( self.subentries = [key[1] for key in entry_datasets.keys()] self.measurements = [] self.failed_parses = {} + # failed_parses is keyed by entry and so retains only one failure per entry, + # which loses information when several subentries of one entry fail. Keep the + # full record alongside it, keyed by subentry. + self.failed_parses_by_subentry = {} for key, data_set in entry_datasets.items(): if not isinstance(data_set.reaction[0], X4Reaction): @@ -205,6 +209,7 @@ def __init__( self.measurements.append(m) for subentry, e in failed_parses.items(): self.failed_parses[key[0]] = (subentry, e) + self.failed_parses_by_subentry[subentry] = e def bibtex(self): if self.meta is None: diff --git a/src/exfor_tools/parsing.py b/src/exfor_tools/parsing.py index 1420d6d..4fdfb71 100644 --- a/src/exfor_tools/parsing.py +++ b/src/exfor_tools/parsing.py @@ -1,3 +1,4 @@ +from collections import defaultdict from functools import reduce import numpy as np @@ -25,12 +26,35 @@ # these are the supported quantities at the moment quantity_matches = { - "dXS/dA": [["DA"], ["PAR", "DA"]], + # EXFOR qualifies the differential cross section in several ways that leave it a + # differential cross section. Per the EXFOR dictionary: "AV" is an average (the + # subentry carries EN-MEAN), "DERIV" is derived data, "DI" is the direct-interaction + # part, and "MSC" flags an approximate reaction code whose meaning is given in the + # entry text. "EXL" appears on data whose reaction string is plain elastic + # scattering. All are matched; the reaction match still requires the right target, + # projectile and process, and for level-resolved data the excitation-energy filter + # still selects the channel. + "dXS/dA": [ + ["DA"], + ["PAR", "DA"], + ["DA", "AV"], + ["PAR", "DA", "AV"], + ["DA", "DERIV"], + ["EXL", "DA"], + ["DI", "DA"], + ["DI", "DA", "MSC"], + ], "dXS/dRuth": [["DA", "RTH"], ["DA", "RTH/REL"]], - "Ay": [["POL/DA", "ANA"]], + # EXFOR tabulates the analyzing power under several codes. "POL/DA,ANA" is the + # modern one; a bare "POL/DA" is the outgoing-particle polarization, which equals + # the analyzing power for elastic scattering by time-reversal invariance; and + # "POL/DA,ASY" is the measured asymmetry, which is the analyzing power once the + # beam polarization is divided out. Older entries use the latter two. + "Ay": [["POL/DA", "ANA"], ["POL/DA"], ["POL/DA", "ASY"]], "Q": [["POL/DA", "SRF"]], "XS": [ ["SIG"], + ["SIG", "AV"], ], } quantities = list(quantity_matches.keys()) @@ -108,6 +132,7 @@ def parse_energy_dependent_xs(data_set, data_error_columns=["DATA-ERR"]): # parse errors xs_err = [] + seen_labels = defaultdict(int) for label in data_error_columns: # parse error column err_parser = X4ColumnParser( @@ -122,12 +147,18 @@ def parse_energy_dependent_xs(data_set, data_error_columns=["DATA-ERR"]): raise ValueError(f"Subentry does not have a column called {label}") else: iyerr = [idx for idx, value in enumerate(data_set.labels) if value == label] - if len(iyerr) > 1: + # A label may legitimately repeat: some subentries carry two DATA-ERR + # columns, one in per-cent and one in absolute units, each mostly null, + # which together make up the uncertainty. Take them in order, one per + # occurrence of the label in data_error_columns, rather than refusing. + occurrence = seen_labels[label] + seen_labels[label] += 1 + if occurrence >= len(iyerr): raise ValueError( - f"Expected only one {label} column, found {len(iyerr)}" + f"Requested {occurrence + 1} {label} columns, found {len(iyerr)}" ) - err = err_parser.getColumn(iyerr[0], data_set) + err = err_parser.getColumn(iyerr[occurrence], data_set) err_units = err[1] err_data = np.nan_to_num(np.array(err[2:], dtype=np.float64)) # convert to same units as data @@ -167,6 +198,7 @@ def parse_differential_data( # parse errors xs_err = [] + seen_labels = defaultdict(int) for label in data_error_columns: # parse error column err_parser = X4ColumnParser( @@ -181,12 +213,18 @@ def parse_differential_data( raise ValueError(f"Subentry does not have a column called {label}") else: iyerr = [idx for idx, value in enumerate(data_set.labels) if value == label] - if len(iyerr) > 1: + # A label may legitimately repeat: some subentries carry two DATA-ERR + # columns, one in per-cent and one in absolute units, each mostly null, + # which together make up the uncertainty. Take them in order, one per + # occurrence of the label in data_error_columns, rather than refusing. + occurrence = seen_labels[label] + seen_labels[label] += 1 + if occurrence >= len(iyerr): raise ValueError( - f"Expected only one {label} column, found {len(iyerr)}" + f"Requested {occurrence + 1} {label} columns, found {len(iyerr)}" ) - err = err_parser.getColumn(iyerr[0], data_set) + err = err_parser.getColumn(iyerr[occurrence], data_set) if np.all([x is None for x in err]): continue diff --git a/src/exfor_tools/reaction.py b/src/exfor_tools/reaction.py index fc924a8..da46c20 100644 --- a/src/exfor_tools/reaction.py +++ b/src/exfor_tools/reaction.py @@ -149,6 +149,30 @@ def query_for_reaction(reaction: Reaction, quantity: str): return entries +#: Columns by which a data set states which residual excitation it covers. Some +#: resolve it to a single level -- E-LVL, E-EXC, LVL-NUMB -- and some only bound it, +#: EXFOR spelling the bound E-LVL-MAX, E-EXC-MAX or E-EXC-MX-A. All are matched here, +#: so the fragments below are deliberately prefixes. +EXCITATION_COLUMN_FRAGMENTS = ("E-LVL", "E-EXC", "LVL-NUMB") + + +def specifies_excitation(subentry) -> bool: + """Whether a data set states the residual excitation it covers. + + True both when the excitation is resolved to one level and when it is merely + bounded, in which case the data set is summed over every level below the bound -- + the ground state plus whatever low-lying levels the experiment could not separate. + The two are not the same thing, and this predicate deliberately does not + distinguish them: the published nucleon-nucleus corpora count both as elastic, + with bounds running from 30 keV on 93Nb up to 800 keV. A caller that needs the + ground state alone must check for a resolved column itself. + """ + return any( + any(fragment in label for fragment in EXCITATION_COLUMN_FRAGMENTS) + for label in subentry.labels + ) + + def is_match(reaction: Reaction, subentry, vocal=False): """Checks if the reaction matches a given subentry. @@ -177,8 +201,24 @@ def is_match(reaction: Reaction, subentry, vocal=False): if isinstance(product, str): if reaction.process is None: return False - if product != reaction.process.upper(): - return False + process = reaction.process.upper() + if product != process: + # The EXFOR dictionary defines SCT as "Total scattering (elastic + + # inelastic)". Summed over everything, that is not elastic scattering and + # must not satisfy a query for it. But many measurements are written as + # (n,SCT) against an excitation column rather than being split into (n,EL) + # and (n,INL): either resolved to a level, whose ground state *is* elastic, + # or bounded above by a few tens to a few hundred keV, which is the ground + # state plus the low-lying levels the experiment could not separate. Both + # match, leaving the excitation-energy filter to select the channel where + # the data set resolves one -- so a caller wanting only the elastic channel + # must pass elastic_only=True, which forces Ex_range to (0, 0). A data set + # that only bounds its excitation has no column for that filter to act on + # and is admitted whole; see specifies_excitation. + if not ( + process == "EL" and product == "SCT" and specifies_excitation(subentry) + ): + return False else: product = (product.getA(), product.getZ()) if product != reaction.product: @@ -191,6 +231,12 @@ def is_match(reaction: Reaction, subentry, vocal=False): subentry.reaction[0].residual.getA(), subentry.reaction[0].residual.getZ(), ) + # the residual of a natural target carries the same -3000 sentinel as the + # target, and must be normalized the same way, or no elastic scattering data + # set on a natural target will ever match + if residual[0] == -3000: + residual = (0, residual[1]) + if reaction.residual is None and reaction.process.upper() in [ "EL", "INL",