{
  "dataset": "FinaleMe processed ultra-low-pass plasma cfDNA WGS fragment-coordinate cohort",
  "zenodo_record": "7779198",
  "doi": "10.5281/zenodo.7779198",
  "paper_doi": "10.1038/s41467-024-47196-6",
  "paper_pmcid": "PMC10981715",
  "license": "CC BY 4.0 for the Zenodo record",
  "official_record_url": "https://zenodo.org/records/7779198",
  "paper_url": "https://doi.org/10.1038/s41467-024-47196-6",
  "label_source": {
    "supplement_endpoint": "https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10981715/supplementaryFiles",
    "workbook": "41467_2024_47196_MOESM4_ESM.xlsx",
    "table_1": "WGS library ID, patient ID, tumor type, sample name, and source BAM filename",
    "table_2": "WGS library ID, batch number, and flowcell",
    "manifest": "real_data/manifests/finaleme_samples.csv"
  },
  "cohort": {
    "sample_records": 77,
    "distinct_patient_id_groups": 42,
    "cancer_sample_records": 65,
    "noncancer_sample_records": 12,
    "cancer_patient_id_groups": 30,
    "noncancer_patient_id_groups": 12,
    "prostate_samples": 43,
    "breast_samples": 22,
    "healthy_samples": 12,
    "analysis_boundary": "The cancer samples are castration-resistant prostate and metastatic breast disease, not an early-detection screening cohort. Split and resample only by patient_id."
  },
  "archive": {
    "name": "ulp_wgs_frag.tar",
    "path": "real_data/raw/finaleme_zenodo_7779198/ulp_wgs_frag.tar",
    "url": "https://zenodo.org/api/records/7779198/files/ulp_wgs_frag.tar/content",
    "bytes": 1894809600,
    "md5": "a98744886e648d8b02d1c914dd4548b0",
    "sha256_local": "d06852eb556822c1c7653972c312c18d151eeced236920ce4e9ed2f6c6c0e6dc",
    "status": "downloaded; official MD5 matched; locally SHA-256 hashed; safely extracted"
  },
  "safe_extraction": {
    "member_count": 77,
    "unique_member_count": 77,
    "regular_file_count": 77,
    "non_regular_entries": 0,
    "absolute_or_drive_qualified_paths": 0,
    "dot_or_dotdot_components": 0,
    "resolved_paths_outside_destination": 0,
    "members_missing_from_label_table": 0,
    "unexpected_members_vs_label_table": 0,
    "destination": "real_data/raw/finaleme_zenodo_7779198/frag_wgs",
    "extracted_files": 77,
    "extracted_compressed_bytes": 1894740349
  },
  "compression_validation": {
    "format": "BGZF / concatenated gzip members",
    "files_read_to_eof": 77,
    "files_failed": 0,
    "total_uncompressed_bytes": 6977345245,
    "minimum_uncompressed_file_bytes": 18198890,
    "maximum_uncompressed_file_bytes": 559736715,
    "validator": "Python 3.14 gzip reader with concatenated-member support; every stream was read to EOF, exercising gzip CRC/trailer checks",
    "reader_warning": "On the available Windows/.NET runtime, a basic GZipStream stopped after the first BGZF member and returned only 65,280 uncompressed bytes per file. Use a BGZF-aware or concatenated-gzip-aware reader."
  },
  "schema_audit": {
    "records_examined": 250308232,
    "malformed_records": 0,
    "blank_records": 0,
    "coordinate_start_decreases_within_contig": 0,
    "revisited_contig_blocks": 0,
    "columns": [
      "chrom",
      "start",
      "end",
      "name_placeholder",
      "mapq",
      "strand"
    ],
    "header": "none",
    "name_placeholder": ". for all 250,308,232 records",
    "reference_build": "b37 / human_g1k_v37 (GRCh37 primary coordinates with GL contigs, MT, and NC_007605)",
    "coordinate_system": "zero-based, half-open BED coordinates",
    "chromosome_naming": "no chr prefix",
    "observed_contigs": "1-22, X, Y, MT, GL000191.1-GL000249.1, and NC_007605",
    "strand_plus_records": 125173021,
    "strand_minus_records": 125135211,
    "other_strand_records": 0,
    "mapq_minimum": 0,
    "mapq_maximum": 70,
    "mapq_below_30_records": 18269157,
    "mapq_at_least_30_records": 232039075,
    "mapq_at_least_30_fraction": 0.9270133593,
    "fragment_length_minimum": 1,
    "fragment_length_maximum": 248759267,
    "length_warning": "The files are not prefiltered to a cfDNA insert-size range and contain very long genomic spans. Apply an explicit, locked length filter and canonical-contig policy before feature extraction.",
    "first_three_records": [
      "1\\t10021\\t10148\\t.\\t27\\t-",
      "1\\t10326\\t10455\\t.\\t3\\t-",
      "1\\t10457\\t10622\\t.\\t60\\t+"
    ]
  },
  "mapq_provenance": {
    "archive_observation": "Column 5 is numeric and behaves as fragment MAPQ; column 4 is the BED name placeholder.",
    "documented_related_format": "FinaleToolkit documents block-gzipped fragment files as chrom/start/stop/mapq/strand.",
    "documented_conversion": "The published FinaleToolkit BAM-to-fragment recipe takes BEDPE score column 8. Bedtools documents the default BEDPE score as the minimum mapping quality of the two mates.",
    "qualification": "The Zenodo archive has no header or archive-specific conversion README proving that exact recipe for these 77 files. Treat column 5 as fragment MAPQ; minimum-mate MAPQ is strongly consistent with the related documented recipe but should not be overstated as archive-specific fact.",
    "finaletoolkit_format_url": "https://pypi.org/project/FinaleToolkit/0.5.1/",
    "bedtools_bedpe_url": "https://bedtools.readthedocs.io/en/latest/content/tools/bamtobed.html"
  },
  "dedup_provenance": {
    "state": "unknown",
    "reason": "The BED6 files contain no duplicate flag or unique record identifier, and the Zenodo archive does not document duplicate marking/removal for this conversion. Supplementary source BAM names end in .calmd.bam; that filename alone does not establish duplicate removal.",
    "required_reporting": "Do not claim duplicate removal was reproduced or verified. Preserve duplicate state as unknown in QC."
  },
  "parser_compatibility": {
    "project_parser": "Compatible. All 77 files are detected by mced_lpwgs.features as headerless BED6 (detector label crag_bed6), Python gzip handles their concatenated BGZF members, chromosome names are normalized, and duplicate=None is preserved.",
    "project_parser_naming_note": "The detector label crag_bed6 is generic/misleading for FinaleMe but does not change parsing semantics.",
    "external_finaletoolkit": "Not directly established. FinaleToolkit documentation expects five-column BED3+2 plus a Tabix index; this archive has six columns and no .tbi files. Drop the placeholder name column, BGZF-compress if rewritten, and create Tabix indexes before using tools that enforce FinaleToolkit's five-column indexed contract.",
    "basic_dotnet_gzip": "Incompatible without explicit multi-member handling; it silently stopped after one BGZF block in this environment."
  },
  "leakage_and_batch_warnings": [
    "Use patient_id as the mandatory split group: 77 records correspond to only 42 apparent donor groups.",
    "Healthy samples are concentrated in batches 21 (7), 10 (3), 8 (1), and 17 (1); batches 21 and 17 contain no cancer samples.",
    "Report batch-only and flowcell-only baselines plus leave-batch/leave-flowcell-out sensitivity analyses.",
    "Do not interpret this advanced-disease cohort as prospective MCED sensitivity or specificity evidence."
  ]
}
