Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
19 commits
Select commit Hold shift + click to select a range
ce36a0f
update for deploy only on version tags;
nickjhathaway Jun 4, 2026
1d22c18
default in the argparse was still mhap_id but we changed the default …
nickjhathaway Jun 5, 2026
374e14e
small fixes for mutation of input data for filter_pmo_by_target_ids;
nickjhathaway Jun 29, 2026
6ad954b
update the hashes because now reads_by_stage is now properly exported…
nickjhathaway Jun 29, 2026
5ea5a2e
fix for when exporting a header for bed locations didn't add a newlin…
nickjhathaway Jun 29, 2026
c71bbfd
update so pseudocigar is properly constructed during the building fun…
nickjhathaway Jun 29, 2026
aa1c22f
start and end were flipped when setting location for forward primer l…
nickjhathaway Jun 29, 2026
fb80099
fix for the change in the schema to qpcr name in the schema which was…
nickjhathaway Jun 29, 2026
e45234f
two bugs in combine_pmos fixed;
nickjhathaway Jun 29, 2026
8b32cda
fixed a left over old schema version where all fields were copies ove…
nickjhathaway Jun 29, 2026
6bb0564
small fixes for handling minimum format; added more robust unittests …
nickjhathaway Jun 29, 2026
0ced495
Merge pull request #80 from PlasmoGenEpi/hotfix/update_man_deploy_git…
nickjhathaway Jul 21, 2026
e7aa09d
Merge pull request #81 from PlasmoGenEpi/hotfix/update_output_name_in…
nickjhathaway Jul 21, 2026
f50e531
Merge pull request #83 from PlasmoGenEpi/hotfix/fix_small_bugs
nickjhathaway Jul 21, 2026
cc821c1
Merge pull request #86 from PlasmoGenEpi/hotfix/update_combine_pmos
nickjhathaway Jul 21, 2026
1b8bf8a
Merge pull request #82 from PlasmoGenEpi/hotfix/fix_filtering_pmos
nickjhathaway Jul 21, 2026
804fce5
Merge pull request #87 from PlasmoGenEpi/hotfix/various_bug_fixes
nickjhathaway Jul 21, 2026
3361846
change ps_chrom_col to pseduocigar_chrom_col
nickjhathaway Jul 21, 2026
ddef2d0
Merge pull request #85 from PlasmoGenEpi/hotfix/fix_pseudocigar_handling
nickjhathaway Jul 21, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 0 additions & 1 deletion .github/workflows/docs.yml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,6 @@ on:
push:
tags:
- 'v*.*.*'
- 'test*'
workflow_dispatch:

permissions:
Expand Down
12 changes: 6 additions & 6 deletions src/pmotools/pmo_builder/metatable_to_pmo.py
Original file line number Diff line number Diff line change
Expand Up @@ -157,7 +157,7 @@ def library_sample_info_table_to_pmo(
library_prep_plate_position_col,
meta_json,
copy_contents,
"specimen_name",
"library_sample_name",
"library_prep_plate_info",
)
meta_json = add_parasite_density_info(
Expand All @@ -166,7 +166,7 @@ def library_sample_info_table_to_pmo(
meta_json,
copy_contents,
"library_sample_name",
entry_name="parasite_density_info",
entry_name="qpcr_parasite_density_info",
)
# listify columns that contain values that could be list, are delimited by the argument list_values_library_values_delimiter
primitives = (int, float, str, bool, complex)
Expand Down Expand Up @@ -494,7 +494,7 @@ def add_plate_info(
plate_position_col,
meta_json,
df,
specimen_name_col,
match_col,
entry_name="plate_info",
):
if all(
Expand Down Expand Up @@ -531,7 +531,7 @@ def add_plate_info(
) from e

for row in meta_json:
content_row = df[df[specimen_name_col] == row[specimen_name_col]]
content_row = df[df[match_col] == row[match_col]]
plate_name_val = content_row[plate_name_col].iloc[0] if plate_name_col else None
plate_row_val = (
content_row[plate_row_col].iloc[0].upper() if plate_row_col else None
Expand Down Expand Up @@ -560,7 +560,7 @@ def add_parasite_density_info(
parasite_density_method_col,
meta_json,
df,
specimen_name_col,
match_col,
entry_name,
):
density_method_pairs = []
Expand Down Expand Up @@ -608,7 +608,7 @@ def add_parasite_density_info(

# Add parasite density info to meta_json
for row in meta_json:
content_row = df[df[specimen_name_col] == row[specimen_name_col]]
content_row = df[df[match_col] == row[match_col]]
density_infos = []
for density_col, method_col in density_method_pairs:
density_val = content_row[density_col].iloc[0] if density_col else None
Expand Down
101 changes: 100 additions & 1 deletion src/pmotools/pmo_builder/mhap_table_to_pmo.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,13 @@ def mhap_table_to_pmo(
masking_delim: str = ",",
microhaplotype_name_col: str | None = None,
pseudocigar_col: str | None = None,
pseudocigar_chrom_col: str | None = None,
pseudocigar_start_col: str | None = None,
pseudocigar_end_col: str | None = None,
pseudocigar_ref_seq_col: str | None = None,
pseudocigar_strand_col: str | None = None,
pseudocigar_genome_id: int | None = None,
pseudocigar_generation_description_col: str | None = None,
quality_col: str | None = None,
additional_representative_mhap_cols: list | None = None,
additional_mhap_detected_cols: list | None = None,
Expand Down Expand Up @@ -73,6 +80,20 @@ def mhap_table_to_pmo(
:type microhaplotype_name_col: str, optional
:param pseudocigar_col: the name of the column containing a pseudocigar for the microhaplotype
:type pseudocigar_col: str, optional
:param pseudocigar_chrom_col: column for the pseudocigar ref_loc chromosome; defaults to chrom_col
:type pseudocigar_chrom_col: str, optional
:param pseudocigar_start_col: column for the pseudocigar ref_loc start (required if pseudocigar_col is set)
:type pseudocigar_start_col: str, optional
:param pseudocigar_end_col: column for the pseudocigar ref_loc end (required if pseudocigar_col is set)
:type pseudocigar_end_col: str, optional
:param pseudocigar_ref_seq_col: optional column for the pseudocigar ref_loc reference sequence
:type pseudocigar_ref_seq_col: str, optional
:param pseudocigar_strand_col: optional column for the pseudocigar ref_loc strand
:type pseudocigar_strand_col: str, optional
:param pseudocigar_genome_id: genome id for the pseudocigar ref_loc; defaults to genome_id
:type pseudocigar_genome_id: int, optional
:param pseudocigar_generation_description_col: optional column describing how the pseudocigar was generated
:type pseudocigar_generation_description_col: str, optional
:param quality_col: the name of the column containing the ANSI FASTQ per-base quality score for this sequence
:type quality_col: str, optional
:param additional_representative_mhap_cols: additional columns to add to the representative microhaplotypes table
Expand Down Expand Up @@ -100,6 +121,13 @@ def mhap_table_to_pmo(
masking_delim=masking_delim,
microhaplotype_name_col=microhaplotype_name_col,
pseudocigar_col=pseudocigar_col,
pseudocigar_chrom_col=pseudocigar_chrom_col,
pseudocigar_start_col=pseudocigar_start_col,
pseudocigar_end_col=pseudocigar_end_col,
pseudocigar_ref_seq_col=pseudocigar_ref_seq_col,
pseudocigar_strand_col=pseudocigar_strand_col,
pseudocigar_genome_id=pseudocigar_genome_id,
pseudocigar_generation_description_col=pseudocigar_generation_description_col,
quality_col=quality_col,
additional_representative_mhap_cols=additional_representative_mhap_cols,
)
Expand Down Expand Up @@ -160,6 +188,13 @@ def create_representative_microhaplotype_dict(
masking_delim: str = ",",
microhaplotype_name_col: str | None = None,
pseudocigar_col: str | None = None,
pseudocigar_chrom_col: str | None = None,
pseudocigar_start_col: str | None = None,
pseudocigar_end_col: str | None = None,
pseudocigar_ref_seq_col: str | None = None,
pseudocigar_strand_col: str | None = None,
pseudocigar_genome_id: int | None = None,
pseudocigar_generation_description_col: str | None = None,
quality_col: str | None = None,
additional_representative_mhap_cols: list[str] | None = None,
):
Expand Down Expand Up @@ -198,6 +233,20 @@ def create_representative_microhaplotype_dict(
:type microhaplotype_name_col: str, optional
:param pseudocigar_col: the name of the column containing a pseudocigar for the microhaplotype
:type pseudocigar_col: str, optional
:param pseudocigar_chrom_col: column for the pseudocigar ref_loc chromosome; defaults to chrom_col
:type pseudocigar_chrom_col: str, optional
:param pseudocigar_start_col: column for the pseudocigar ref_loc start (required if pseudocigar_col is set)
:type pseudocigar_start_col: str, optional
:param pseudocigar_end_col: column for the pseudocigar ref_loc end (required if pseudocigar_col is set)
:type pseudocigar_end_col: str, optional
:param pseudocigar_ref_seq_col: optional column for the pseudocigar ref_loc reference sequence
:type pseudocigar_ref_seq_col: str, optional
:param pseudocigar_strand_col: optional column for the pseudocigar ref_loc strand
:type pseudocigar_strand_col: str, optional
:param pseudocigar_genome_id: genome id for the pseudocigar ref_loc; defaults to genome_id
:type pseudocigar_genome_id: int, optional
:param pseudocigar_generation_description_col: optional column describing how the pseudocigar was generated
:type pseudocigar_generation_description_col: str, optional
:param quality_col: the name of the column containing the ANSI FASTQ per-base quality score for this sequence
:type quality_col: str, optional
:param additional_representative_mhap_cols: additional columns to add to the representative microhaplotypes table
Expand All @@ -211,6 +260,23 @@ def create_representative_microhaplotype_dict(
microhaplotype_table, additional_representative_mhap_cols
)

# the pseudocigar ref_loc shares the chromosome / genome with the
# microhaplotype location by default, but each may use its own column
pseudocigar_chrom_col = (
pseudocigar_chrom_col if pseudocigar_chrom_col else chrom_col
)
ps_genome_id = (
pseudocigar_genome_id if pseudocigar_genome_id is not None else genome_id
)
if pseudocigar_col and not (
pseudocigar_chrom_col and pseudocigar_start_col and pseudocigar_end_col
):
raise ValueError(
"pseudocigar_col is set, so a Pseudocigar ref_loc must be "
"constructable: set a chromosome (pseudocigar_chrom_col or "
"chrom_col), pseudocigar_start_col, and pseudocigar_end_col."
)

def get_if_present(row, col):
return row[col] if col and pd.notna(row[col]) else None

Expand Down Expand Up @@ -263,6 +329,12 @@ def warn_if_duplicated_seqs(df, target_col, seq_col):
alt_annotations_col,
microhaplotype_name_col,
pseudocigar_col,
pseudocigar_chrom_col,
pseudocigar_start_col,
pseudocigar_end_col,
pseudocigar_ref_seq_col,
pseudocigar_strand_col,
pseudocigar_generation_description_col,
quality_col,
]
masking_cols = [
Expand Down Expand Up @@ -319,7 +391,34 @@ def warn_if_duplicated_seqs(df, target_col, seq_col):
if val := get_if_present(row, microhaplotype_name_col):
mhap["microhaplotype_name"] = val
if val := get_if_present(row, pseudocigar_col):
mhap["pseudo_cigar"] = val
if not (
pd.notna(row[pseudocigar_chrom_col])
and pd.notna(row[pseudocigar_start_col])
and pd.notna(row[pseudocigar_end_col])
):
raise ValueError(
f"pseudocigar present for target {target}, seq "
f"{row[seq_col]} but its ref_loc chrom/start/end "
"is missing"
)
ref_loc = {
"genome_id": ps_genome_id,
"chrom": row[pseudocigar_chrom_col],
"start": row[pseudocigar_start_col],
"end": row[pseudocigar_end_col],
}
if pseudocigar_ref_seq_col and pd.notna(row[pseudocigar_ref_seq_col]):
ref_loc["ref_seq"] = row[pseudocigar_ref_seq_col]
if pseudocigar_strand_col and pd.notna(row[pseudocigar_strand_col]):
ref_loc["strand"] = row[pseudocigar_strand_col]
pseudocigar = {"pseudocigar_seq": val, "ref_loc": ref_loc}
if pseudocigar_generation_description_col and pd.notna(
row[pseudocigar_generation_description_col]
):
pseudocigar["pseudocigar_generation_description"] = row[
pseudocigar_generation_description_col
]
mhap["pseudocigar"] = pseudocigar
if val := get_if_present(row, quality_col):
mhap["quality"] = val
if additional_representative_mhap_cols:
Expand Down
4 changes: 2 additions & 2 deletions src/pmotools/pmo_builder/panel_information_to_pmo.py
Original file line number Diff line number Diff line change
Expand Up @@ -332,8 +332,8 @@ def build_target_info_dict(
fwd_primer_dict["location"] = {
"genome_id": genome_id,
"chrom": row[chrom_col],
"end": int(row[forward_primers_start_col]),
"start": int(row[forward_primers_end_col]),
"start": int(row[forward_primers_start_col]),
"end": int(row[forward_primers_end_col]),
}
if strand_col and pd.notna(row[strand_col]):
fwd_primer_dict["location"]["strand"] = row[strand_col]
Expand Down
26 changes: 15 additions & 11 deletions src/pmotools/pmo_engine/pmo_exporter.py
Original file line number Diff line number Diff line change
Expand Up @@ -121,10 +121,14 @@ def export_specimen_meta_table(pmodata, separator: str = ",") -> pd.DataFrame:
for specimen in pmodata["specimen_info"]:
export_row = {}
for key, value in specimen.items():
if "project_id" == key:
if "project_id" == key and "project_info" in pmodata:
export_row["project_name"] = pmodata["project_info"][value][
"project_name"
]
elif "project_id" == key:
# project_info is optional as of schema v1.1.0; keep the raw id
# rather than failing to resolve a name that isn't available
export_row[key] = value
elif PMOExporter._is_primitive(value):
export_row[key] = value
elif PMOExporter._is_primitive_list(value):
Expand All @@ -145,10 +149,14 @@ def export_library_sample_meta_table(pmodata, separator: str = ",") -> pd.DataFr
for library_sample in pmodata["library_sample_info"]:
export_row = {}
for key, value in library_sample.items():
if "sequencing_info_id" == key:
if "sequencing_info_id" == key and "sequencing_info" in pmodata:
export_row["sequencing_info_name"] = pmodata["sequencing_info"][
value
]["sequencing_info_name"]
elif "sequencing_info_id" == key:
# sequencing_info is optional as of schema v1.1.0; keep the raw id
# rather than failing to resolve a name that isn't available
export_row[key] = value
elif "specimen_id" == key:
export_row["specimen_name"] = pmodata["specimen_info"][value][
"specimen_name"
Expand Down Expand Up @@ -451,6 +459,7 @@ def write_bed_locs(bed_locs: list[BedLoc], fnp, add_header: bool = False):
]
)
)
f.write("\n")
for bed_loc in bed_locs:
f.write(
"\t".join(
Expand Down Expand Up @@ -540,11 +549,10 @@ def extract_panels_insert_bed_loc(
:param sort_output: whether to sort output by genomic location
:return: a list of target inserts, with named tuples with fields: chrom, start, end, name, score, strand, ref_seq, extra_info
"""
bed_loc_out = {}
bed_loc_out = []
if select_panel_ids is None:
select_panel_ids = list(range(len(pmodata["panel_info"])))
for panel_id in select_panel_ids:
bed_loc_out_per_panel = []
for reaction_id in range(len(pmodata["panel_info"][panel_id]["reactions"])):
for target_id in pmodata["panel_info"][panel_id]["reactions"][
reaction_id
Expand Down Expand Up @@ -589,7 +597,7 @@ def extract_panels_insert_bed_loc(
if "ref_seq" not in tar["insert_location"]
else tar["insert_location"]["ref_seq"]
)
bed_loc_out_per_panel.append(
bed_loc_out.append(
BedLoc(
tar["insert_location"]["chrom"],
tar["insert_location"]["start"],
Expand All @@ -602,12 +610,8 @@ def extract_panels_insert_bed_loc(
extra_info,
)
)
if sort_output:
return sorted(
bed_loc_out_per_panel,
key=lambda bed: (bed.chrom, bed.start, bed.end),
)
bed_loc_out[panel_id] = bed_loc_out_per_panel
if sort_output:
return sorted(bed_loc_out, key=lambda bed: (bed.chrom, bed.start, bed.end))
return bed_loc_out

@staticmethod
Expand Down
Loading
Loading