diff --git a/coderbuild/build_all.py b/coderbuild/build_all.py index 067dd5dc..d98f338c 100644 --- a/coderbuild/build_all.py +++ b/coderbuild/build_all.py @@ -430,17 +430,17 @@ def get_latest_commit_hash(owner, repo, branch='main'): decompress_file(file) ### These should be done before schema checking. - sample_mapping_command = ['python3', 'scripts/map_improve_sample_ids.py', '--local_dir', "/tmp", '--version', args.version] + sample_mapping_command = ['python3', 'coderbuild/utils/map_improve_sample_ids.py', '--local_dir', "/tmp", '--version', args.version] run_docker_upload_cmd(sample_mapping_command, 'all_files_dir', 'Map_Samples', args.version) - drug_mapping_command = ['python3', 'scripts/map_improve_drug_ids.py', '--local_dir', "/tmp", '--version', args.version] + drug_mapping_command = ['python3', 'coderbuild/utils/map_improve_drug_ids.py', '--local_dir', "/tmp", '--version', args.version] run_docker_upload_cmd(drug_mapping_command, 'all_files_dir', 'Map_Drugs', args.version) - drug_mapping_command_2 = ['python3', 'scripts/align_drug_descriptors.py', '--local_dir', "/tmp", '--version', args.version] + drug_mapping_command_2 = ['python3', 'coderbuild/utils/align_drug_descriptors.py', '--local_dir', "/tmp", '--version', args.version] run_docker_upload_cmd(drug_mapping_command_2, 'all_files_dir', 'Align_Drug_Descriptors', args.version) # Run schema checker - This will always run if uploading data. - schema_check_command = ['python3', 'scripts/check_schema.py', '--datasets'] + datasets + schema_check_command = ['python3', 'coderbuild/utils/check_schema.py', '--datasets'] + datasets run_docker_upload_cmd(schema_check_command, 'all_files_dir', 'validate', args.version) print("Validation complete. Proceeding with file compression/decompression adjustments") @@ -457,7 +457,7 @@ def get_latest_commit_hash(owner, repo, branch='main'): ### Upload to Figshare using Docker if args.figshare and args.version and figshare_token: - figshare_command = ['python3', 'scripts/push_to_figshare.py', '--directory', "/tmp", '--title', f"CODERData{args.version}", '--token', os.getenv('FIGSHARE_TOKEN'), '--project_id', '189342', '--version', args.version, '--publish'] + figshare_command = ['python3', 'coderbuild/utils/push_to_figshare.py', '--directory', "/tmp", '--title', f"CODERData{args.version}", '--token', os.getenv('FIGSHARE_TOKEN'), '--project_id', '189342', '--version', args.version, '--publish'] run_docker_upload_cmd(figshare_command, 'all_files_dir', 'Figshare', args.version) ### Push changes to GitHub using Docker diff --git a/coderbuild/build_dataset.py b/coderbuild/build_dataset.py index 7c5ed10c..930c0e2b 100644 --- a/coderbuild/build_dataset.py +++ b/coderbuild/build_dataset.py @@ -268,7 +268,7 @@ def run_schema_checker(dataset): decompress_file(os.path.join('local', all_files_dir, file)) # Run schema checker - schema_check_command = ['python3', 'scripts/check_schema.py', '--datasets'] + datasets + schema_check_command = ['python3', 'coderbuild/utils/check_schema.py', '--datasets'] + datasets run_docker_validate_cmd(schema_check_command, all_files_dir, 'Validation') def main(): diff --git a/coderbuild/docker/Dockerfile.upload b/coderbuild/docker/Dockerfile.upload index e08bb0c3..e8cf61eb 100644 --- a/coderbuild/docker/Dockerfile.upload +++ b/coderbuild/docker/Dockerfile.upload @@ -25,4 +25,5 @@ RUN apt-get update && \ # WORKDIR /usr/src/app/coderdata ADD schema schema -ADD scripts scripts \ No newline at end of file +ADD scripts scripts +ADD coderbuild/utils coderbuild/utils \ No newline at end of file diff --git a/scripts/align_drug_descriptors.py b/coderbuild/utils/align_drug_descriptors.py similarity index 100% rename from scripts/align_drug_descriptors.py rename to coderbuild/utils/align_drug_descriptors.py diff --git a/scripts/check_schema.py b/coderbuild/utils/check_schema.py similarity index 100% rename from scripts/check_schema.py rename to coderbuild/utils/check_schema.py diff --git a/scripts/map_improve_drug_ids.py b/coderbuild/utils/map_improve_drug_ids.py similarity index 100% rename from scripts/map_improve_drug_ids.py rename to coderbuild/utils/map_improve_drug_ids.py diff --git a/scripts/map_improve_sample_ids.py b/coderbuild/utils/map_improve_sample_ids.py similarity index 100% rename from scripts/map_improve_sample_ids.py rename to coderbuild/utils/map_improve_sample_ids.py diff --git a/scripts/push_to_figshare.py b/coderbuild/utils/push_to_figshare.py similarity index 100% rename from scripts/push_to_figshare.py rename to coderbuild/utils/push_to_figshare.py diff --git a/scripts/assign_improve_ids.py b/scripts/assign_improve_ids.py deleted file mode 100755 index 66ee658c..00000000 --- a/scripts/assign_improve_ids.py +++ /dev/null @@ -1,76 +0,0 @@ -import pandas as pd -import argparse - - -def determine_delimiter(filename): - """ - Determine the delimiter based on the file extension. - """ - if filename.endswith('.csv'): - return ',' - elif filename.endswith('.tsv'): - return '\t' - else: - raise ValueError(f"Unsupported file type for file {filename}. Only .csv and .tsv are supported.") - - -def add_improve_id(previous, new, sample_col, new_name): - """ - Add Improve IDs to a new samples.tsv file. - - previous: previous sample.tsv filepath - new: new sample.tsv filepath - sample_col: name of column to map improve_sample_id to in the new samples.tsv - new_name: name/save the new samples.tsv instead of overwriting it (unless you want to). - - Example Use: - add_improve_id("path_to_previous_samples.tsv", "path_to_new_samples.tsv") - """ - # Determine delimiter for the input files - previous_delimiter = determine_delimiter(previous) - new_delimiter = determine_delimiter(new) - - # Read the previous file - previous_df = pd.read_csv(previous, sep=previous_delimiter) - # Extract the maximum value of improve_sample_id from the previous file - max_id = previous_df['improve_sample_id'].max() - - # Read the new file - new_df = pd.read_csv(new, sep=new_delimiter) - - # Ceate improve_sample_id column filled with NaN values - if 'improve_sample_id' not in new_df.columns: - new_df['improve_sample_id'] = float('nan') - - # Extract unique sample_ids from the new dataframe where improve_sample_id is NaN - unique_sample_ids = new_df[new_df['improve_sample_id'].isna()][sample_col].unique() - - # Create a mapping of sample_id to improve_sample_id - id_map = {} - for sample_id in unique_sample_ids: - max_id += 1 - id_map[sample_id] = max_id - - ## Apply the mapping to the new dataframe - new_df.loc[new_df['improve_sample_id'].isna(), 'improve_sample_id'] = new_df[sample_col].map(id_map) - - # Reorder the columns to place 'improve_sample_id' to the right of 'sample_col' - col_idx = new_df.columns.get_loc(sample_col) - improve_col = new_df.pop('improve_sample_id') # Remove and get the column - new_df.insert(col_idx + 1, 'improve_sample_id', improve_col) # Insert it back right after sample_col - - # Save the updated version of the new dataframe with the same delimiter as the input 'new' file - new_df.to_csv(new_name, sep=new_delimiter, index=False) - return - -if __name__ == '__main__': - parser = argparse.ArgumentParser(description="Assign improve_sample_id values to a new samples file based on a previous samples file.") - - parser.add_argument("-p", "--previous", type=str, required=True, help="Path to the previous samples.tsv file with existing improve_sample_id values.") - parser.add_argument("-n", "--new", type=str, required=True, help="Path to the new samples.tsv file that needs improve_sample_id values assigned.") - parser.add_argument("-s", "--sample_col", type=str, required=True, help="Name of column to map improve_sample_id to in the new samples.tsv.") - parser.add_argument("-o", "--new_name", type=str, required=True, help="Name/save the new samples.tsv instead of overwriting it. You could name it the same thing if you want it overwritten though.") - - args = parser.parse_args() - - add_improve_id(args.previous, args.new, args.sample_col, args.new_name) diff --git a/scripts/figshare_pull.py b/scripts/figshare_pull.py deleted file mode 100755 index 5e82449e..00000000 --- a/scripts/figshare_pull.py +++ /dev/null @@ -1,111 +0,0 @@ -import requests -import wget -import os -import shutil -import gzip -import pandas as pd -import re -import argparse - -def retrieve_figshare_data(url): - """ - Download data from FigShare url. - - Returns: a list of newly unpacked files. - """ - files_0 = os.listdir() - wget.download(url) - files_1 = os.listdir() - figdir = str(next(iter((set(files_1) - set(files_0))))) - shutil.unpack_archive(figdir) - files_2 = os.listdir() - files_3 = list(set(files_2) - set(files_1)) - files_4 = os.listdir() - files_5 = list(set(files_4) - set(files_1)) - return files_5 - -def merger(*data_types, directory=".", outname="Merged_Data.csv", no_duplicate=True, drop_na=False): - """ - Merge datasets by data types. - - Args: - data_types: Type of data sets to merge. (Example: "transcriptomics", "mutations") - directory: Directory where the data resides. - outname: Name of the output file. - no_duplicate: Drop duplicate rows if set to True. Default is True. - drop_na: Drop rows with NA values if set to True. Default is False. - - Example Usage: - python figshare_pull.py transcriptomics mutations copy_number methylation proteomics -d . -o Merged_Data.csv --url "https://figshare.com/ndownloader/articles/22822286?private_link=525f7777039f4610ef47" --no_duplicate --drop_na - - Returns: - DataFrame: Merged dataset. - """ - - def get_prefix_number(filename): - """Extract prefix number or return 0 if none exists.""" - match = re.match(r"(\d+)", filename) - return int(match.group(1)) if match else 0 - - dfs = {} - primary_data_type = None - - for data_type in data_types: - files = [f for f in os.listdir(directory) if data_type in f and f.endswith(('.csv', '.tsv', '.csv.gz', '.tsv.gz'))] - - if not files: - print(f"No files found for data type: {data_type}. This data type will not be included.") - continue - - selected_file = max(files, key=get_prefix_number) - print(f"Selected file for {data_type}: {selected_file}. Proceeding with merge.") - - if not primary_data_type: - primary_data_type = data_type - - path = os.path.join(directory, selected_file) - compression = 'gzip' if selected_file.endswith('.gz') else None - delimiter = "\t" if selected_file.endswith((".tsv", ".tsv.gz")) else "," - chunk_iter = pd.read_csv(path, sep=delimiter, compression=compression, chunksize=10**5, low_memory=False) - df_parts = [chunk for chunk in chunk_iter] - dfs[data_type] = pd.concat(df_parts, ignore_index=True) - - if not primary_data_type: - print("No suitable data found for any specified data type.") - return - - merged_df = dfs[primary_data_type] - - for data_type in data_types[1:]: - if data_type in dfs: - merge_cols = ["improve_sample_id", "entrez_id"] - if "source" in merged_df.columns and "source" in dfs[data_type].columns: - merge_cols.append("source") - if "study" in merged_df.columns and "study" in dfs[data_type].columns: - merge_cols.append("study") - merged_df = merged_df.merge(dfs[data_type], on=merge_cols, how="outer") - if no_duplicate: - merged_df.drop_duplicates(inplace=True) - if drop_na: - merged_df.dropna(inplace=True) - - merged_df.to_csv(outname, index=False) - return merged_df - - -if __name__ == "__main__": - parser = argparse.ArgumentParser(description="Merge datasets by data types from specified directory.") - parser.add_argument("data_types", nargs="+", help="Types of data sets to merge. (Example: 'transcriptomics', 'mutations')") - parser.add_argument("-d", "--directory", default=".", help="Directory where the data resides.") - parser.add_argument("-o", "--outname", help="Name of the output file.") - parser.add_argument("-u", "--url", help="URL to figshare.") - parser.add_argument("--no_duplicate", action="store_true", help="Drop duplicate rows.") - parser.add_argument("--drop_na", action="store_true", help="Drop rows with NA values.") - - args = parser.parse_args() -# cell_line_data = "https://figshare.com/ndownloader/articles/22822286?private_link=525f7777039f4610ef47" -# cell_line_files = retrieve_figshare_data(cell_line_data) - retrieve_figshare_data(args.url) - print("\n") - merger(*args.data_types, directory=args.directory, outname=args.outname, no_duplicate=args.no_duplicate, drop_na=args.drop_na) - diff --git a/scripts/update_version.py b/scripts/update_version.py deleted file mode 100644 index 4dd8c6e3..00000000 --- a/scripts/update_version.py +++ /dev/null @@ -1,14 +0,0 @@ -import sys - -def update_version(file_path, new_version): - with open(file_path, 'r') as file: - lines = file.readlines() - - with open(file_path, 'w') as file: - for line in lines: - if "version=" in line: - line = f"version='{new_version}'\n" - file.write(line) - -if __name__ == "__main__": - update_version('setup.py', sys.argv[1]) \ No newline at end of file