Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
93 changes: 93 additions & 0 deletions .github/workflows/crisprdecode-collaborator.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,93 @@
name: CRISPRDecode collaborator tests

on:
push:
branches:
- "test/crisprdecode-*"
workflow_dispatch:

permissions:
contents: read

concurrency:
group: crisprdecode-collaborator-${{ github.ref }}
cancel-in-progress: true

jobs:
python:
name: Python validation and assignment
runs-on: ubuntu-24.04
timeout-minutes: 10
steps:
- uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5
- uses: actions/setup-python@e797f83bcb11b83ae66e0230d6156d7c80228e7c # v6
with:
python-version: "3.12.8"
- name: Run regression tests
shell: bash
env:
PYTHONDONTWRITEBYTECODE: "1"
run: |
set -euo pipefail
mkdir -p test-artifacts
git rev-parse HEAD > test-artifacts/commit.txt
python --version > test-artifacts/python.txt
python -m unittest discover -s tests/crisprdecode -v 2>&1 | tee test-artifacts/python-tests.log
- name: Save Python results
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: crisprdecode-python-${{ github.sha }}
path: test-artifacts/
retention-days: 14

nextflow:
name: Nextflow ${{ matrix.suite }}
runs-on: ubuntu-24.04
timeout-minutes: 25
strategy:
fail-fast: false
matrix:
include:
- suite: decoding
test_path: subworkflows/local/crisprdecode_paired_guide/tests/main.nf.test
- suite: screening-routing
test_path: tests/main_screening_crisprdecode.nf.test
env:
NXF_VER: "25.04.0"
NXF_ANSI_LOG: "false"
NFT_WORKDIR: ".nf-test"
steps:
- uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5
- uses: nf-core/setup-nextflow@6c2e22b4d901f0c42ca66c5069f8026df026d165 # v2
with:
version: "25.04.0"
java-version: "17"
- uses: nf-core/setup-nf-test@f2db585128ae681918440f6c074dededeceadce1 # v1
with:
version: "0.9.3"
- name: Run selected nf-tests with Docker
shell: bash
env:
TEST_PATH: ${{ matrix.test_path }}
run: |
set -euo pipefail
mkdir -p test-artifacts
git rev-parse HEAD > test-artifacts/commit.txt
nextflow -version > test-artifacts/nextflow.txt
nf-test version > test-artifacts/nf-test.txt
docker --version > test-artifacts/docker.txt
nf-test test "$TEST_PATH" --profile=+docker --ci --verbose --tap=test-artifacts/tests.tap 2>&1 | tee test-artifacts/nf-test.log
- name: Save nf-test results
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: crisprdecode-${{ matrix.suite }}-${{ github.sha }}
include-hidden-files: true
path: |
test-artifacts/
.nf-test.log
.nextflow.log
.nf-test/**/*.log
.nf-test/**/output/**
retention-days: 14
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

### Added

- Add exact paired-guide construct counting with CRISPRDecode, including conservative ambiguity handling, QC, MultiQC integration and synthetic nf-tests ([#278](https://github.com/nf-core/crisprseq/issues/278))
- Add an acknowlodgement to Hitselection method ([#222]https://github.com/nf-core/crisprseq/pull/222)
- Added parameters specifying additional arguments for the cutadapt command line ([#237](https://github.com/nf-core/crisprseq/pull/237))
- Template update to 3.2.1 ([#244](https://github.com/nf-core/crisprseq/pull/244))
Expand Down
42 changes: 42 additions & 0 deletions assets/multiqc_config.yml
Original file line number Diff line number Diff line change
Expand Up @@ -192,6 +192,45 @@ custom_data:
max: 1
min: 0

# CRISPRDecode paired-guide assignment summary
crisprdecode_assignment:
id: "crisprdecode_assignment"
section_name: "CRISPRDecode paired-guide assignment"
plot_type: "table"
file_format: "tsv"
description: |
Exact paired-guide construct assignment statistics per sample. Uniquely assigned,
ambiguous, unassigned and extraction-failed reads sum to the total read count.
pconfig:
id: "crisprdecode_assignment"
namespace: "CRISPRDecode paired-guide assignment"
table_title: "CRISPRDecode paired-guide assignment"
headers:
total_reads:
title: "Total reads"
description: "Total synchronized read pairs"
format: "{:,.0f}"
extracted_reads:
title: "Extracted reads"
description: "Read pairs for which both guide elements were extracted"
format: "{:,.0f}"
unique_reads:
title: "Unique"
description: "Read pairs uniquely assigned to one construct"
format: "{:,.0f}"
ambiguous_reads:
title: "Ambiguous"
description: "Read pairs matching a signature shared by multiple constructs"
format: "{:,.0f}"
unassigned_reads:
title: "Unassigned"
description: "Extracted signatures absent from the construct library"
format: "{:,.0f}"
extraction_failed_reads:
title: "Extraction failed"
description: "Read pairs from which one or both guide elements could not be extracted"
format: "{:,.0f}"

sp:
edition_plot:
fn: "*_edits.csv"
Expand All @@ -203,6 +242,8 @@ sp:
fn: "*.cutadapt.log"
mageck_count:
fn: "*.countsummary.txt"
crisprdecode_assignment:
fn: "assignment_summary.tsv"

# Define the order of sections
module_order:
Expand All @@ -217,6 +258,7 @@ custom_content:
- read_processing
- edition_plot
- indel_qc_plot
- crisprdecode_assignment

# Order of table columns
table_columns_placement:
Expand Down
121 changes: 121 additions & 0 deletions bin/crisprdecode_aggregate.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,121 @@
#!/usr/bin/env python3
"""Aggregate per-sample paired-guide counts into crisprseq outputs."""

from __future__ import annotations

import argparse
import csv
from pathlib import Path


def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("--library", required=True, type=Path)
parser.add_argument("--sample-counts", required=True, nargs="+", type=Path)
parser.add_argument("--sample-summaries", required=True, nargs="+", type=Path)
parser.add_argument("--count-matrix", required=True, type=Path)
parser.add_argument("--assignment-summary", required=True, type=Path)
parser.add_argument("--library-recovery", required=True, type=Path)
return parser.parse_args()


def read_tsv(path: Path) -> list[dict[str, str]]:
with path.open(newline="") as handle:
return list(csv.DictReader(handle, delimiter="\t"))


def main() -> None:
args = parse_args()
library = read_tsv(args.library)
if not library:
raise SystemExit("ERROR: validated construct library is empty")

counts_by_sample: dict[str, dict[str, int]] = {}
target_by_construct = {row["construct_id"]: row["target_id"] for row in library}
for path in args.sample_counts:
rows = read_tsv(path)
if not rows:
raise SystemExit(f"ERROR: sample count file is empty: {path}")
samples = {row["sample"] for row in rows}
if len(samples) != 1:
raise SystemExit(
f"ERROR: expected one sample in {path}, found {sorted(samples)}"
)
sample = samples.pop()
if sample in counts_by_sample:
raise SystemExit(f"ERROR: duplicate sample count file for '{sample}'")
counts_by_sample[sample] = {
row["construct_id"]: int(row["count"]) for row in rows
}

samples = sorted(counts_by_sample)
with args.count_matrix.open("w", newline="") as handle:
writer = csv.writer(handle, delimiter="\t", lineterminator="\n")
writer.writerow(["sgRNA", "Gene", *samples])
for row in library:
construct_id = row["construct_id"]
writer.writerow(
[
construct_id,
target_by_construct[construct_id],
*[
counts_by_sample[sample].get(construct_id, 0)
for sample in samples
],
]
)

summaries: list[dict[str, str]] = []
for path in args.sample_summaries:
rows = read_tsv(path)
if len(rows) != 1:
raise SystemExit(f"ERROR: expected exactly one summary row in {path}")
summaries.extend(rows)
summary_fields = [
"sample",
"total_reads",
"extracted_reads",
"unique_reads",
"ambiguous_reads",
"unassigned_reads",
"extraction_failed_reads",
]
with args.assignment_summary.open("w", newline="") as handle:
writer = csv.DictWriter(
handle, fieldnames=summary_fields, delimiter="\t", lineterminator="\n"
)
writer.writeheader()
writer.writerows(sorted(summaries, key=lambda row: row["sample"]))

recovery_fields = [
"construct_id",
"target_id",
"assignability_class",
"duplicate_signature_size",
"total_count",
"zero_count",
]
with args.library_recovery.open("w", newline="") as handle:
writer = csv.DictWriter(
handle, fieldnames=recovery_fields, delimiter="\t", lineterminator="\n"
)
writer.writeheader()
for row in library:
construct_id = row["construct_id"]
total_count = sum(
counts_by_sample[sample].get(construct_id, 0) for sample in samples
)
writer.writerow(
{
"construct_id": construct_id,
"target_id": row["target_id"],
"assignability_class": row["assignability_class"],
"duplicate_signature_size": row["duplicate_signature_size"],
"total_count": total_count,
"zero_count": str(total_count == 0).lower(),
}
)


if __name__ == "__main__":
main()
Loading