454 catalog_ref: pd.DataFrame,
455 catalog_target: pd.DataFrame,
456 select_ref: np.array =
None,
457 select_target: np.array =
None,
458 logger: logging.Logger =
None,
459 logging_n_rows: int =
None,
466 catalog_ref : `pandas.DataFrame`
467 A reference catalog to match in order of a given column (i.e. greedily).
468 catalog_target : `pandas.DataFrame`
469 A target catalog for matching sources from `catalog_ref`. Must contain measurements with errors.
470 select_ref : `numpy.array`
471 A boolean array of the same length as `catalog_ref` selecting the sources that can be matched.
472 select_target : `numpy.array`
473 A boolean array of the same length as `catalog_target` selecting the sources that can be matched.
474 logger : `logging.Logger`
475 A Logger for logging.
476 logging_n_rows : `int`
477 The number of sources to match before printing a log message.
479 Additional keyword arguments to pass to `format_catalogs`.
483 catalog_out_ref : `pandas.DataFrame`
484 A catalog of identical length to `catalog_ref`, containing match information for rows selected by
485 `select_ref` (including the matching row index in `catalog_target`).
486 catalog_out_target : `pandas.DataFrame`
487 A catalog of identical length to `catalog_target`, containing the indices of matching rows in
489 exceptions : `dict` [`int`, `Exception`]
490 A dictionary keyed by `catalog_target` row number of the first exception caught when matching.
493 logger = logger_default
505 ref, target = config.coord_format.format_catalogs(
506 catalog_ref=catalog_ref, catalog_target=catalog_target,
507 select_ref=select_ref, select_target=select_target,
517 catalog_ref.loc[ref.extras.select, config.column_ref_order]
518 if config.column_ref_order
is not None else
519 np.nansum(catalog_ref.loc[ref.extras.select, config.columns_ref_flux], axis=1)
521 order = np.argsort(column_order
if config.order_ascending
else -column_order)
523 n_ref_select = len(ref.extras.indices)
525 match_dist_max = config.match_dist_max
526 coords_spherical = config.coord_format.coords_spherical
528 match_dist_max = np.radians(match_dist_max / 3600.)
531 func_convert = _radec_to_xyz
if coords_spherical
else np.vstack
532 vec_ref, vec_target = (
533 func_convert(cat.coord1[cat.extras.select], cat.coord2[cat.extras.select])
534 for cat
in (ref, target)
538 logger.info(
'Generating cKDTree with match_n_max=%d', config.match_n_max)
539 tree_obj = cKDTree(vec_target)
541 scores, idxs_target_select = tree_obj.query(
543 distance_upper_bound=match_dist_max,
544 k=config.match_n_max,
547 n_target_select = len(target.extras.indices)
548 n_matches = np.sum(idxs_target_select != n_target_select, axis=1)
549 n_matched_max = np.sum(n_matches == config.match_n_max)
550 if n_matched_max > 0:
552 '%d/%d (%.2f%%) selected true objects have n_matches=n_match_max(%d)',
553 n_matched_max, n_ref_select, 100.*n_matched_max/n_ref_select, config.match_n_max
557 target_row_match = np.full(target.extras.n, np.nan, dtype=np.int64)
558 ref_candidate_match = np.zeros(ref.extras.n, dtype=bool)
559 ref_row_match = np.full(ref.extras.n, np.nan, dtype=np.int64)
560 ref_match_count = np.zeros(ref.extras.n, dtype=np.int32)
561 ref_match_meas_finite = np.zeros(ref.extras.n, dtype=np.int32)
562 ref_chisq = np.full(ref.extras.n, np.nan, dtype=float)
565 idx_orig_ref, idx_orig_target = (np.argwhere(cat.extras.select)
for cat
in (ref, target))
568 columns_convert = config.coord_format.coords_ref_to_convert
569 if columns_convert
is None:
571 data_ref = ref.catalog[
572 [columns_convert.get(column, column)
for column
in config.columns_ref_meas]
573 ].iloc[ref.extras.indices[order]]
574 data_target = target.catalog[config.columns_target_meas][target.extras.select]
575 errors_target = target.catalog[config.columns_target_err][target.extras.select]
579 matched_target = {n_target_select, }
581 t_begin = time.process_time()
583 logger.info(
'Matching n_indices=%d/%d', len(order), len(ref.catalog))
584 for index_n, index_row_select
in enumerate(order):
585 index_row = idx_orig_ref[index_row_select]
586 ref_candidate_match[index_row] =
True
587 found = idxs_target_select[index_row_select, :]
592 found = [x
for x
in found
if x
not in matched_target]
597 (data_target.iloc[found].values - data_ref.iloc[index_n].values)
598 / errors_target.iloc[found].values
600 finite = np.isfinite(chi)
601 n_finite = np.sum(finite, axis=1)
603 chisq_good = n_finite >= config.match_n_finite_min
604 if np.any(chisq_good):
606 chisq_sum = np.zeros(n_found, dtype=float)
607 chisq_sum[chisq_good] = np.nansum(chi[chisq_good, :] ** 2, axis=1)
608 idx_chisq_min = np.nanargmin(chisq_sum / n_finite)
609 ref_match_meas_finite[index_row] = n_finite[idx_chisq_min]
610 ref_match_count[index_row] = len(chisq_good)
611 ref_chisq[index_row] = chisq_sum[idx_chisq_min]
612 idx_match_select = found[idx_chisq_min]
613 row_target = target.extras.indices[idx_match_select]
614 ref_row_match[index_row] = row_target
616 target_row_match[row_target] = index_row
617 matched_target.add(idx_match_select)
618 except Exception
as error:
621 exceptions[index_row] = error
623 if logging_n_rows
and ((index_n + 1) % logging_n_rows == 0):
624 t_elapsed = time.process_time() - t_begin
626 'Processed %d/%d in %.2fs at sort value=%.3f',
627 index_n + 1, n_ref_select, t_elapsed, column_order[order[index_n]],
631 'match_candidate': ref_candidate_match,
632 'match_row': ref_row_match,
633 'match_count': ref_match_count,
634 'match_chisq': ref_chisq,
635 'match_n_chisq_finite': ref_match_meas_finite,
638 'match_candidate': target.extras.select
if target.extras.select
is not None else (
639 np.ones(target.extras.n, dtype=bool)),
640 'match_row': target_row_match,
643 for (columns, out_original, out_matched, in_original, in_matched, matches, name_cat)
in (
663 matched = matches >= 0
664 idx_matched = matches[matched]
665 logger.info(
'Matched %d/%d %s sources', np.sum(matched), len(matched), name_cat)
667 for column
in columns:
668 values = in_original.catalog[column]
669 out_original[column] = values
670 dtype = in_original.catalog[column].dtype
674 types = list(
set((type(x)
for x
in values)))
676 raise RuntimeError(f
'Column {column} dtype={dtype} has multiple types={types}')
683 dtype = f
'<U{max(len(x) for x in values)}'
685 column_match = np.full(in_matched.extras.n, value_fill, dtype=dtype)
686 column_match[matched] = in_original.catalog[column][idx_matched]
687 out_matched[f
'match_{column}'] = column_match
689 catalog_out_ref = pd.DataFrame(data_ref)
690 catalog_out_target = pd.DataFrame(data_target)
692 return catalog_out_ref, catalog_out_target, exceptions