#!/usr/bin/env python3 """ Download, process, and upload all RDF datasets to HuggingFace Hub. WORKFLOW: For each dataset, the script executes three phases: Step 1: Download - Downloads raw RDF from public source URL - Extracts to usable RDF format - Progress bars with download speed Step 2: Decompress - Decompresses RDF files (if necessary) Step 3: Convert - Parses RDF file (Turtle, N-Triples, RDF/XML, etc.) - Converts to HuggingFace dataset format - Preserves all semantic information (lossless) - Creates dataset card (README.md) with metadata Step 4: Upload - Uploads to HuggingFace Hub under CleverThis organization - Creates/updates dataset repository - Uploads dataset card with comprehensive documentation - Makes dataset publicly accessible DATASET CATEGORIES: small < 1 GB (WordNet, Schema.org, Gene Ontology) medium 1-10 GB (YAGO 4.5, DisGeNET, GeoNames) large 10-100 GB (DBpedia, UniProt, DBLP) xlarge > 100 GB (Wikidata, BabelNet, Bio2RDF) DIRECTORY STRUCTURE: dataset_processing/ ├── downloads/ # Raw RDF downloads │ ├── wordnet/ │ │ ├── english-wordnet-2024.ttl.gz │ │ └── english-wordnet-2024.ttl │ └── yago-4.5/ │ └── yago-4.5.0.2-tiny.ttl │ └── hf_datasets/ # Converted HuggingFace datasets ├── wordnet/ │ ├── data/ │ ├── dataset_info.json │ └── README.md └── yago-4.5/ ├── data/ ├── dataset_info.json └── README.md """ from __future__ import annotations import argparse import os import shutil import subprocess import sys import time from datetime import datetime from pathlib import Path # Import our modules from dataset_registry import DATASET_REGISTRY, DatasetInfo, get_dataset_config from rich.console import Console from rich.table import Table console = Console() def parse_args() -> tuple[argparse.Namespace, argparse.ArgumentParser]: """Parse command-line arguments.""" parser = argparse.ArgumentParser( description="Download, process, and upload all RDF datasets to HuggingFace Hub." ) parser.add_argument( "--dataset", "-d", type=str, action="append", help="Specific dataset(s) to process (can be specified multiple times)", ) parser.add_argument( "--category", "-c", choices=["small", "medium", "large", "xlarge", "all"], default=None, help="Process datasets by category", ) parser.add_argument( "--base-dir", type=Path, default=Path("./dataset_processing"), help="Base directory for work", ) parser.add_argument( "--skip-download", action="store_true", help="Skip download step" ) parser.add_argument( "--skip-convert", action="store_true", help="Skip conversion step" ) parser.add_argument("--skip-upload", action="store_true", help="Skip upload step") parser.add_argument( "--dry-run", action="store_true", help="Show what would be done without doing it", ) parser.add_argument( "--list", action="store_true", help="List datasets that would be processed" ) parser.add_argument( "--rm", action="store_true", dest="remove_downloaded", help="Remove mode: wipe all downloads at start, " "delete each dataset after processing", ) parser.add_argument( "--parallel", "-p", type=int, default=1, help="Number of datasets to process in parallel (default: 1, sequential)", ) parser.add_argument( "--repeat", action="store_true", default=3, help="Repeat download/upload attempts", ) return parser.parse_args(), parser def get_datasets_to_process(args: argparse.Namespace) -> list[str]: """ Determine which datasets to process. Args: args: A list of arguments parsed by argparse Returns: Which datasets to process. """ if args.dataset: datasets_to_process = args.dataset elif args.category == "all": datasets_to_process = list(DATASET_REGISTRY.keys()) elif args.category: datasets_to_process = [ d_id for d_id, d_info in DATASET_REGISTRY.items() if d_info.category == args.category ] else: datasets_to_process = [] return datasets_to_process def list_datasets(datasets_to_process: list[str]) -> None: """Print a list of datasets that would be processed.""" console.print("\n[bold cyan]Datasets to Process:[/bold cyan]\n") table = Table(show_header=True) table.add_column("ID", style="cyan") table.add_column("Name", style="green") table.add_column("Category", style="yellow") table.add_column("Size", style="blue") table.add_column("Available", style="magenta") if len(datasets_to_process) == 0: list_datasets = list(DATASET_REGISTRY.keys()) else: list_datasets = datasets_to_process for dataset_id in list_datasets: info = DATASET_REGISTRY[dataset_id] available = "[green]✓[/green]" if info.available else "[red]✗[/red]" table.add_row( dataset_id, info.name, info.category, f"{info.size_gb} GB", available ) console.print(table) available_count = sum( 1 for d_id in datasets_to_process if DATASET_REGISTRY[d_id].available ) console.print( f"\n[bold]Total: {len(datasets_to_process)} datasets " f"({available_count} available)[/bold]" ) def get_most_recent_file(download_dir: Path) -> Path | None: """ Get the file with the most recent creation date in a directory. Args: download_dir: Path object pointing to the directory to search Returns: Path to the most recent file, or None if no files found """ # Get all files (not directories) in the directory and subdirectories files = [f for f in download_dir.rglob("*") if f.is_file()] if not files: return None # Find the file with the most recent creation time most_recent_file = max(files, key=lambda f: f.stat().st_ctime) return most_recent_file def get_dataset_info(dataset_id: str) -> DatasetInfo | None: """ Gets the dataset info from the dataset registry. Args: dataset_id: An identifier for the dataset Returns: The dataset info, or None if not found. """ try: dataset_info = DATASET_REGISTRY.get(dataset_id) if not dataset_info: console.print(f"[red]Dataset '{dataset_id}' not found in registry[/red]") console.print( "[yellow]Available datasets: " f"{', '.join(list(DATASET_REGISTRY.keys())[:5])}...[/yellow]" ) return None except Exception as e: console.print(f"[red]Error accessing dataset registry: {e}[/red]") return None return dataset_info def download( dataset_id: str, base_dir: Path, skip_download: bool = False, dry_run: bool = False, ) -> tuple[bool, bool, Path | None]: """ This function downloads the dataset from the source URL. It is responsible only for downloading the dataset and checking that the dataset looks reasonable. At the end of the function, the (possibly compressed) dataset should be available in the downloads directory. Returns: Whether the download was successful. Whether there should be another attempt to download the dataset. The name of the downloaded file, if successful. """ dataset_info = get_dataset_info(dataset_id) rdf_file = None if not dataset_info: return (False, False, None) # Check if dataset is available if not dataset_info.available: console.print(f"\n[bold yellow]⊘ Skipping '{dataset_info.name}'[/bold yellow]") console.print("[yellow]Status:[/yellow] Dataset currently unavailable") console.print(f"[yellow]Reason:[/yellow] {dataset_info.notes}") console.print(f"[dim]URL: {dataset_info.url}[/dim]") return (False, False, None) # Create directories if necessary console.print( f"\n[bold cyan]{'[DRY RUN] ' if dry_run else ''}" f"Processing: {dataset_info.name}[/bold cyan]" ) # Setup directories download_dir = base_dir / "downloads" / dataset_id try: download_dir.mkdir(parents=True, exist_ok=True) except OSError as e: console.print(f"[red]Error creating directories: {e}[/red]") console.print(f"[yellow]Download dir: {download_dir}[/yellow]") return (False, False, None) # Download the dataset if not skip_download: path = str(Path(__file__).parent / "rdf_dataset_downloader.py") console.print("\n[yellow]Step 1: Downloading raw RDF dataset...[/yellow]") if dry_run: console.print( f"[dim]Would run: {path} " f"{dataset_id} -o {download_dir}[/dim]" ) return (True, False, None) else: console.print( f"[dim]Running: {path} {dataset_id} -o {download_dir}[/dim]" ) try: result = subprocess.run( [ sys.executable, path, dataset_id, "-o", str(download_dir), ], capture_output=False, timeout=3600 * 24, # 24 hour timeout ) if result.returncode != 0: console.print( f"[red]Download failed for {dataset_id} " f"(exit code: {result.returncode})[/red]" ) console.print( "[yellow]Try running manually: " "python scripts/rdf_dataset_downloader.py " f"{dataset_id} -o {download_dir}[/yellow]" ) return (False, False, None) except subprocess.TimeoutExpired: console.print( f"[red]Download timeout for {dataset_id} (exceeded 24 hours)[/red]" ) console.print( "[yellow]Large datasets may need more time. " "Try running manually using rdf_dataset_downloader.py " "with --force[/yellow]" ) return (False, True, None) except FileNotFoundError: console.print( "[red]Download script not found: rdf_dataset_downloader.py[/red]" ) console.print( "[yellow]Make sure you're running " "from the correct directory[/yellow]" ) return (False, False, None) except Exception as e: console.print(f"[red]Unexpected error during download: {e}[/red]") return (False, False, None) # Check the dataset. # Find the downloaded file (only if we need it for conversion) # This code assumes that the most recent file is the one we want. rdf_file = get_most_recent_file(download_dir) if not skip_download else None if rdf_file is None: console.print(f"[red]No RDF file found in {download_dir}[/red]") return (False, False, None) if str(rdf_file).endswith(".bz2"): # Verify that the file is not corrupted console.print( "[yellow]Found a BZ2 file, checking integrity...[/yellow]" ) # TEST 1: Check the size of the file file_size_gb = rdf_file.stat().st_size / (1024**3) if dataset_info.compressed_size_gb: expected_gb = dataset_info.compressed_size_gb if file_size_gb < expected_gb * 0.9: console.print( f"[red]Incomplete BZ2 file: {rdf_file.name}[/red]" ) console.print( f"[yellow]File size: {file_size_gb:.3f} GB, " f"Expected: ~{expected_gb:.2f} GB[/yellow]" ) console.print( "[red]File is only " f"{(file_size_gb / expected_gb * 100):.1f}% " "of expected size[/red]" ) console.print( "[yellow]Removing incomplete file " "and re-downloading...[/yellow]" ) rdf_file.unlink() return (False, True, None) # Re-download. # TEST 2: Check the integrity of the file try: # Test the bz2 file integrity test_result = subprocess.run( ["bunzip2", "-t", str(rdf_file)], capture_output=True, text=True ) if test_result.returncode != 0: console.print(f"[red]Corrupted BZ2 file: {rdf_file.name}[/red]") console.print( "[yellow]File size: " f"{rdf_file.stat().st_size / (1024**2):.2f} MB[/yellow]" ) console.print( "[yellow]Removing corrupted file " "and attempting re-download...[/yellow]" ) rdf_file.unlink() return (False, True, None) # Re-download. except FileNotFoundError: console.print("[red]bunzip2 not found: please install it[/red]") return (False, False, None) except Exception as e: if ( "end-of-stream marker" in str(e).lower() or "compressed file ended" in str(e).lower() ): console.print( "[yellow]The file is incomplete. " "Attempting to re-download...[/yellow]" ) rdf_file.unlink() return (False, True, None) console.print(f"[red]Unexpected error checking BZ2 file: {e}[/red]") return (False, False, None) else: console.print("[dim]Skipping download[/dim]") return (True, False, rdf_file) def decompress(rdf_file: Path) -> bool: """ This function decompresses the dataset if necessary. At the end of the function, the uncompressid dataset should be available in the downloads directory. Returns: Whether the decompression was successful. """ decompress_result = None if str(rdf_file).endswith(".bz2"): decompress_result = subprocess.run( ["bunzip2", "-d", "-f", str(rdf_file)], capture_output=True, text=True, ) if str(rdf_file).endswith(".gz"): decompress_result = subprocess.run( ["gunzip", "-f", str(rdf_file)], capture_output=True, text=True ) if str(rdf_file).endswith(".zip"): decompress_result = subprocess.run( ["unzip", "-o", "-f", str(rdf_file)], capture_output=True, text=True ) if decompress_result is not None and decompress_result.returncode != 0: console.print(f"[red]Error decompressing {rdf_file}[/red]") console.print("[yellow]Output:[/yellow]") console.print(decompress_result.stdout) console.print("[yellow]Error output:[/yellow]") console.print(decompress_result.stderr) return False return True def get_valid_hf_license(license_string: str | None) -> str: """Map dataset license to valid HuggingFace license identifier. Args: license_string: License string from dataset registry Returns: Valid HuggingFace license identifier """ if not license_string: return "other" # Get license mappings from config config = get_dataset_config() license_map = config.get("license_mappings", {}) # Convert keys to lowercase for comparison license_map_lower = {k.lower(): v for k, v in license_map.items()} normalized = license_string.lower().strip() return license_map_lower.get(normalized, "other") def get_size_category(dataset_info: DatasetInfo) -> str: """Get HuggingFace size category. Args: dataset_info: Dataset information Returns: Size category string """ if dataset_info.size_gb < 0.01: return "n<1K" elif dataset_info.size_gb < 1: return "1K str: """Get comprehensive RDF format documentation for README. Args: dataset_info: Dataset information Returns: Markdown documentation for RDF format """ return f""" ## Dataset Format: Lossless RDF Representation This dataset uses a **standard lossless format** for representing RDF (Resource Description Framework) data in HuggingFace Datasets. All semantic information from the original RDF knowledge graph is preserved, enabling perfect round-trip conversion between RDF and HuggingFace formats. ### Schema Each RDF triple is represented as a row with **6 fields**: | Field | Type | Description | Example | |-------|------|-------------|---------| | `subject` | string | Subject of the triple (URI or blank node) | `"http://schema.org/Person"` | | `predicate` | string | Predicate URI | `"http://www.w3.org/1999/02/22-rdf-syntax-ns#type"` | | `object` | string | Object of the triple | `"John Doe"` or `"http://schema.org/Thing"` | | `object_type` | string | Type of object: `"uri"`, `"literal"`, or `"blank_node"` | `"literal"` | | `object_datatype` | string | XSD datatype URI (for typed literals) | `"http://www.w3.org/2001/XMLSchema#integer"` | | `object_language` | string | Language tag (for language-tagged literals) | `"en"` | ### Example: RDF Triple Representation **Original RDF (Turtle)**: ```turtle "John Doe"@en . ``` **HuggingFace Dataset Row**: ```python {{ "subject": "http://example.org/John", "predicate": "http://schema.org/name", "object": "John Doe", "object_type": "literal", "object_datatype": None, "object_language": "en" }} ``` ### Loading the Dataset ```python from datasets import load_dataset # Load the dataset dataset = load_dataset("CleverThis/{dataset_info.id}") # Access the data data = dataset["data"] # Iterate over triples for row in data: subject = row["subject"] predicate = row["predicate"] obj = row["object"] obj_type = row["object_type"] print(f"Triple: ({{subject}}, {{predicate}}, {{obj}})") print(f" Object type: {{obj_type}}") if row["object_language"]: print(f" Language: {{row['object_language']}}") if row["object_datatype"]: print(f" Datatype: {{row['object_datatype']}}") ``` ### Converting Back to RDF The dataset can be converted back to any RDF format (Turtle, N-Triples, RDF/XML, etc.) with **zero information loss**: ```python from datasets import load_dataset from rdflib import Graph, URIRef, Literal, BNode def convert_to_rdf(dataset_name, output_file="output.ttl", split="data"): \"\"\"Convert HuggingFace dataset back to RDF Turtle format.\"\"\" # Load dataset dataset = load_dataset(dataset_name) # Create RDF graph graph = Graph() # Convert each row to RDF triple for row in dataset[split]: # Subject if row["subject"].startswith("_:"): subject = BNode(row["subject"][2:]) else: subject = URIRef(row["subject"]) # Predicate (always URI) predicate = URIRef(row["predicate"]) # Object (depends on object_type) if row["object_type"] == "uri": obj = URIRef(row["object"]) elif row["object_type"] == "blank_node": obj = BNode(row["object"][2:]) elif row["object_type"] == "literal": if row["object_datatype"]: obj = Literal(row["object"], datatype=URIRef(row["object_datatype"])) elif row["object_language"]: obj = Literal(row["object"], lang=row["object_language"]) else: obj = Literal(row["object"]) graph.add((subject, predicate, obj)) # Serialize to Turtle (or any RDF format) graph.serialize(output_file, format="turtle") print(f"Exported {{len(graph)}} triples to {{output_file}}") return graph # Usage graph = convert_to_rdf("CleverThis/{dataset_info.id}", "reconstructed.ttl") ``` ### Information Preservation Guarantee This format preserves **100% of RDF information**: - ✅ **URIs**: Exact string representation preserved - ✅ **Literals**: Full text content preserved - ✅ **Datatypes**: XSD and custom datatypes preserved (e.g., `xsd:integer`, `xsd:dateTime`) - ✅ **Language Tags**: BCP 47 language tags preserved (e.g., `@en`, `@fr`, `@ja`) - ✅ **Blank Nodes**: Node structure preserved (identifiers may change but graph isomorphism maintained) **Round-trip guarantee**: Original RDF → HuggingFace → Reconstructed RDF produces **semantically identical** graphs. ### Querying the Dataset You can filter and query the dataset like any HuggingFace dataset: ```python from datasets import load_dataset dataset = load_dataset("CleverThis/{dataset_info.id}") # Find all triples with English literals english_literals = dataset["data"].filter( lambda x: x["object_type"] == "literal" and x["object_language"] == "en" ) print(f"Found {{len(english_literals)}} English literals") # Find all rdf:type statements type_statements = dataset["data"].filter( lambda x: "rdf-syntax-ns#type" in x["predicate"] ) print(f"Found {{len(type_statements)}} type statements") # Convert to Pandas for analysis import pandas as pd df = dataset["data"].to_pandas() # Analyze predicate distribution print(df["predicate"].value_counts()) ``` ### Dataset Format The dataset contains all triples in a single **data** split, suitable for machine learning tasks such as: - Knowledge graph completion - Link prediction - Entity embedding - Relation extraction - Graph neural networks ### Format Specification For complete technical documentation of the RDF-to-HuggingFace format, see: 📖 [RDF to HuggingFace Format Specification](https://github.com/CleverThis/cleverernie/blob/master/docs/rdf_huggingface_format_specification.md) The specification includes: - Detailed schema definition - All RDF node type mappings - Performance benchmarks - Edge cases and limitations - Complete code examples ### Conversion Metadata - **Source Format**: {dataset_info.format} - **Original Size**: {dataset_info.size_gb} GB - **Conversion Tool**: [CleverErnie RDF Pipeline](https://github.com/CleverThis/cleverernie) - **Format Version**: 1.0 - **Conversion Date**: {datetime.now().strftime("%Y-%m-%d")} """ # noqa: E501 def get_rdf_format(dataset_info: DatasetInfo) -> str: """Get RDF format string for rdflib. Args: dataset_info: Dataset information Returns: Format string for rdflib """ # Get format mappings from config config = get_dataset_config() format_map = config.get("format_mappings", {}) # Default mappings for common formats default_mappings = { "rdf": "xml", "rdf/xml": "xml", "xml": "xml", "turtle": "turtle", "ttl": "turtle", "nt": "nt", "ntriples": "nt", "n3": "n3", "jsonld": "json-ld", "json-ld": "json-ld" } format_lower = dataset_info.format.lower() # First try config mappings, then default mappings, then fallback mapped_format = format_map.get(format_lower) if mapped_format: return mapped_format mapped_format = default_mappings.get(format_lower) if mapped_format: return mapped_format # Fallback to turtle for unknown formats return "turtle" def create_dataset_card(dataset_info: DatasetInfo) -> str: """Create README.md content for the dataset card. Args: dataset_info: Dataset information Returns: Markdown content for dataset card """ return f"""--- license: {get_valid_hf_license(dataset_info.license)} task_categories: - text-generation - feature-extraction language: - en tags: - rdf - knowledge-graph - semantic-web - triples size_categories: - {get_size_category(dataset_info)} --- # {dataset_info.name} ## Dataset Description {dataset_info.description} **Original Source:** {dataset_info.url} ### Dataset Summary This dataset contains RDF triples from {dataset_info.name} converted to HuggingFace dataset format for easy use in machine learning pipelines. - **Format:** Originally {dataset_info.format}, converted to HuggingFace Dataset - **Size:** {dataset_info.size_gb} GB (extracted) - **Entities:** {dataset_info.entities if dataset_info.entities else "N/A"} - **Triples:** {dataset_info.triples if dataset_info.triples else "N/A"} - **Original License:** {dataset_info.license if dataset_info.license else "See original source"} ### Recommended Use {dataset_info.recommended_for} {f"### Notes: {dataset_info.notes}" if dataset_info.notes else ""} {get_rdf_format_documentation(dataset_info)} ## Citation If you use this dataset, please cite the original source: **Original Dataset:** {dataset_info.name} **URL:** {dataset_info.url} **License:** {dataset_info.license if dataset_info.license else "See original source"} ## Dataset Preparation This dataset was prepared using the CleverErnie GISM framework: ```bash # Download original dataset python scripts/rdf_dataset_downloader.py {dataset_info.id} -o datasets/ # Convert to HuggingFace format python scripts/convert_rdf_to_hf_dataset_unified.py \\ datasets/{dataset_info.id}/[file] \\ hf_datasets/{dataset_info.id} \\ --format {get_rdf_format(dataset_info)} \\ --strategy auto # Upload to HuggingFace Hub python scripts/upload_all_datasets.py --dataset {dataset_info.id} ``` ## Additional Information ### Original Source {dataset_info.url} ### Conversion Details - Converted using: [CleverErnie GISM](https://github.com/cleverthis/cleverernie) - Conversion script: `scripts/convert_rdf_to_hf_dataset_unified.py` - Dataset format: Single 'data' split with all triples - Strategy: Auto-selected based on dataset size and format ### Maintenance This dataset is maintained by the CleverThis organization. """ def get_dataset_dir(args: argparse.Namespace, dataset_id: str) -> Path | None: hf_dataset_dir = args.base_dir / "hf_datasets" / dataset_id try: hf_dataset_dir.mkdir(parents=True, exist_ok=True) except OSError as e: console.print(f"[red]Error creating directory: {e}[/red]") console.print(f"[yellow]HF dataset dir: {hf_dataset_dir}[/yellow]") console.print("[yellow]Check permissions and disk space[/yellow]") return None return hf_dataset_dir def convert(args: argparse.Namespace, rdf_file: Path, dataset_id: str) -> None: """ This function converts the dataset to the HuggingFace dataset format. At the end of the function, the dataset should be available in the hf_datasets directory. If the conversion was unsuccessful, the function will throw an exception. """ download_dir = rdf_file.parent dataset_info = get_dataset_info(dataset_id) if not dataset_info: raise ValueError(f"Dataset {dataset_id} not found in registry") hf_dataset_dir = get_dataset_dir(args, dataset_id) if hf_dataset_dir is None: raise ValueError("Could not retrieve dataset directory") if not args.skip_convert: # Special handling for FB15k-237 if dataset_id == "fb15k-237": # Use specialized FB15k-237 converter that handles all splits fb15k_script = Path(__file__).parent / "convert_fb15k237_to_hf.py" fb15k_dir = download_dir / "fb15k-237" / "FB15K-237.2" if args.dry_run: console.print( "[dim]Would run: python scripts/convert_fb15k237_to_hf.py " "{fb15k_dir} {hf_dataset_dir}[/dim]" ) else: result = subprocess.run( [ sys.executable, str(fb15k_script), str(fb15k_dir), str(hf_dataset_dir), ], capture_output=False, ) if result.returncode != 0: console.print( f"[red]Conversion failed for {dataset_id} " f"(exit code: {result.returncode})[/red]" ) console.print(f"[yellow]Input file: {rdf_file}[/yellow]") console.print(f"[yellow]Output dir: {hf_dataset_dir}[/yellow]") console.print( "[yellow]Try running conversion manually to see " "detailed error messages[/yellow]" ) raise ValueError("Conversion failed") # Create dataset card readme_path = hf_dataset_dir / "README.md" readme_path.write_text(create_dataset_card(dataset_info)) console.print(f"[green]Created dataset card: {readme_path}[/green]") # Special handling for NELL-995 elif dataset_id == "nell-995": # Use specialized NELL-995 converter that handles raw.kb format nell_script = Path(__file__).parent / "convert_nell995_to_hf.py" if args.dry_run: console.print( "[dim]Would run: python scripts/convert_nell995_to_hf.py " "{download_dir} {hf_dataset_dir}[/dim]" ) else: result = subprocess.run( [ sys.executable, str(nell_script), str(download_dir), str(hf_dataset_dir), ], capture_output=False, ) if result.returncode != 0: console.print( f"[red]Conversion failed for {dataset_id}. " f"Return code: {result.returncode}[/red]" ) raise ValueError("Conversion failed") # Create dataset card readme_path = hf_dataset_dir / "README.md" readme_path.write_text(create_dataset_card(dataset_info)) console.print(f"[green]Created dataset card: {readme_path}[/green]") # Special handling for ConceptNet elif dataset_id == "conceptnet": # Use specialized ConceptNet converter that handles CSV format conceptnet_script = Path(__file__).parent / "convert_conceptnet_to_hf.py" # Find the downloaded CSV file csv_files = list(download_dir.rglob("*.csv")) + list( download_dir.rglob("*.csv.gz") ) if not csv_files: console.print( f"[red]No CSV files found for ConceptNet in {download_dir}[/red]" ) raise ValueError("No CSV files found") csv_file = csv_files[0] console.print(f"[green]Found ConceptNet CSV file: {csv_file}[/green]") if args.dry_run: console.print( "[dim]Would run: python scripts/convert_conceptnet_to_hf.py " "{csv_file} {hf_dataset_dir}[/dim]" ) else: result = subprocess.run( [ sys.executable, str(conceptnet_script), str(csv_file), str(hf_dataset_dir), ], capture_output=False, ) if result.returncode != 0: console.print( f"[red]Conversion failed for {dataset_id}. " f"Return code: {result.returncode}[/red]" ) raise ValueError("Conversion failed") # Create dataset card readme_path = hf_dataset_dir / "README.md" readme_path.write_text(create_dataset_card(dataset_info)) console.print(f"[green]Created dataset card: {readme_path}[/green]") else: # Original conversion logic for other datasets # Use unified converter which automatically selects best strategy unified_script = ( Path(__file__).parent / "convert_rdf_to_hf_dataset_unified.py" ) # Unified script will auto-select best strategy based on file size # and format console.print( "[dim]Using unified converter (auto-selects best strategy)[/dim]" ) if args.dry_run: console.print( "[dim]Would run: python " "scripts/convert_rdf_to_hf_dataset_unified.py " "{rdf_file} {hf_dataset_dir} --strategy auto[/dim]" ) else: # Use unified converter with auto-strategy selection # Adjust chunk size based on dataset size chunk_size = 50000 # Default if dataset_info.size_gb > 10: chunk_size = 100000 # Larger chunks for very large datasets elif dataset_info.size_gb < 1: chunk_size = 10000 # Smaller chunks for small datasets result = subprocess.run( [ sys.executable, str(unified_script), str(rdf_file), str(hf_dataset_dir), "--format", get_rdf_format(dataset_info), "--strategy", "auto", # Let the unified script auto-select best strategy "--chunk-size", str(chunk_size), "--description", dataset_info.description, "--homepage", dataset_info.url, "--license", dataset_info.license if dataset_info.license else "Unknown", ], capture_output=False, ) if result.returncode != 0: console.print( f"[red]Conversion failed for {dataset_id}. " f"Return code: {result.returncode}[/red]" ) def upload(args: argparse.Namespace, dataset_id: str) -> None: """ This function uploads the dataset to the HuggingFace Hub. This function throws a ValueError if the upload fails. """ dataset_info = get_dataset_info(dataset_id) if not dataset_info: raise ValueError(f"Dataset {dataset_id} not found in registry") hf_dataset_dir = get_dataset_dir(args, dataset_id) if hf_dataset_dir is None: raise ValueError("Could not retrieve dataset directory") if not args.skip_upload: # Get repository ID from dataset info or config if hasattr(dataset_info, "huggingface_repo") and dataset_info.huggingface_repo: repo_id = dataset_info.huggingface_repo else: # Fallback to default pattern from config config = get_dataset_config() org = config.get("organization", "CleverThis") repo_id = f"{org}/{dataset_id}" if args.dry_run: console.print(f"[dim]Would upload to: {repo_id}[/dim]") else: try: from datasets import load_from_disk from huggingface_hub import HfApi # Load dataset dataset = load_from_disk(str(hf_dataset_dir)) # Upload console.print(f"[cyan]Uploading to {repo_id}...[/cyan]") dataset.push_to_hub(repo_id, private=False) # Upload README separately if it exists readme_path = hf_dataset_dir / "README.md" if readme_path.exists(): api = HfApi() api.upload_file( path_or_fileobj=str(readme_path), path_in_repo="README.md", repo_id=repo_id, repo_type="dataset", ) else: # Create README if it doesn't exist (e.g., when skipping conversion) console.print( "[yellow]README.md not found, creating it now...[/yellow]" ) readme_path.write_text(create_dataset_card(dataset_info)) api = HfApi() api.upload_file( path_or_fileobj=str(readme_path), path_in_repo="README.md", repo_id=repo_id, repo_type="dataset", ) console.print( f"[bold green]✓ Successfully uploaded to {repo_id}[/bold green]" ) except ImportError as e: console.print(f"[red]Missing required packages for upload: {e}[/red]") console.print( "[yellow]Install with: " "pip install datasets huggingface-hub[/yellow]" ) raise ValueError("Missing required packages for upload") from e except FileNotFoundError as e: console.print( f"[red]Dataset directory not found: {hf_dataset_dir}[/red]" ) console.print( "[yellow]Make sure conversion step completed successfully[/yellow]" ) raise ValueError("Dataset directory not found") from e except PermissionError as e: console.print( "[red]Permission denied uploading to HuggingFace Hub[/red]" ) console.print( "[yellow]Check your HuggingFace token: " "huggingface-cli login[/yellow]" ) console.print("[yellow]Or set HF_TOKEN environment variable[/yellow]") raise ValueError("Permission denied uploading to HuggingFace") from e except Exception as e: error_msg = str(e).lower() if "authentication" in error_msg or "token" in error_msg: console.print(f"[red]Authentication failed: {e}[/red]") console.print("[yellow]Run: huggingface-cli login[/yellow]") console.print( "[yellow]Or set HF_TOKEN environment variable[/yellow]" ) raise ValueError("Authentication failed") from e if "network" in error_msg or "connection" in error_msg: console.print(f"[red]Network error during upload: {e}[/red]") console.print( "[yellow]Check internet connection and try again[/yellow]" ) raise ValueError("Network error during upload") from e if "disk" in error_msg or "space" in error_msg: console.print(f"[red]Disk space error: {e}[/red]") console.print("[yellow]Check available disk space[/yellow]") raise ValueError("Disk space error") from e console.print(f"[red]Upload failed: {e}[/red]") console.print(f"[yellow]Repository: {repo_id}[/yellow]") console.print(f"[yellow]Dataset path: {hf_dataset_dir}[/yellow]") raise ValueError("Upload failed") from e else: console.print("[dim]Skipping upload[/dim]") def remove_dir(download_dir: Path) -> None: if download_dir.exists(): try: shutil.rmtree(download_dir) console.print(f"[green]✓ Removed {download_dir}[/green]") except OSError as e: console.print( f"[yellow]Warning: Could not remove {download_dir}: {e}[/yellow]" ) console.print("[yellow]You may need to remove it manually[/yellow]") def main() -> int: """ The main function calls everything else. Returns: A Unix-style exit code. 0 == success. """ args, parser = parse_args() # Fix for "AF_UNIX path too long" error in multiprocessing # This forces the temporary directory to be /tmp (short path) instead of # a potentially deep workspace path os.environ["TMPDIR"] = "/tmp" start_time = time.monotonic() # List datasets to process datasets_to_process = get_datasets_to_process(args) datasets_time = time.monotonic() datasets_duration = datasets_time - start_time print(f"`get_datasets_to_process` took {datasets_duration:.2f} seconds:") # List mode: if args.list: list_datasets(datasets_to_process) return 0 if not datasets_to_process: parser.print_help() console.print( "[red]Error: No datasets specified. Use --dataset " "or --category.[/red]" ) return 1 downloaded = False rdf_file = None for dataset_id in datasets_to_process: one_dataset_start_time = time.monotonic() # Download attempt_count = 0 while attempt_count < args.repeat: attempt_count += 1 console.print( f"[bold cyan]Downloading {dataset_id}... " f"Attempt {attempt_count}[/bold cyan]" ) (downloaded, rerun, rdf_file) = download( dataset_id=dataset_id, base_dir=args.base_dir, skip_download=args.skip_download, dry_run=args.dry_run, ) if not rerun: break if (not downloaded or not rdf_file) and not args.dry_run: console.print("[red]Download failed. Aborting.[/red]") return 1 one_dataset_download_time = time.monotonic() one_dataset_download_duration = ( one_dataset_download_time - one_dataset_start_time ) print(f"`download` took {one_dataset_download_duration:.2f} seconds:") # Decompress if args.dry_run: console.print(f"[dim][DRY RUN] Would decompress {dataset_id}...[/dim]") else: console.print(f"[bold cyan]Decompressing {dataset_id}...[/bold cyan]") assert rdf_file is not None result = decompress(rdf_file) if not result: console.print("[red]Decompression failed. Aborting.[/red]") return 1 one_dataset_decompress_time = time.monotonic() one_dataset_decompress_duration = ( one_dataset_decompress_time - one_dataset_download_time ) print(f"`decompress` took {one_dataset_decompress_duration:.2f} seconds:") # Convert assert rdf_file is not None if not args.dry_run: rdf_file = get_most_recent_file(rdf_file.parent) if rdf_file is None: console.print("[red]Could not find RDF file. Aborting.[/red]") return 1 console.print( "[bold cyan]Converting " f"{rdf_file if rdf_file else 'DRY RUN'}...[/bold cyan]" ) # Pass a dummy path for dry run if rdf_file is None if rdf_file: convert_file = rdf_file else: convert_file = Path(f"dataset_processing/downloads/{dataset_id}/dummy.ttl") try: convert(args, convert_file, dataset_id) except ValueError as e: console.print(f"[red]Conversion failed: {e} Aborting.[/red]") return 1 one_dataset_convert_time = time.monotonic() one_dataset_convert_duration = ( one_dataset_convert_time - one_dataset_decompress_time ) print(f"`convert` took {one_dataset_convert_duration:.2f} seconds:") # Upload attempt_count = 0 while attempt_count < args.repeat: attempt_count += 1 console.print( "[bold cyan]Uploading {dataset_id}... " f"Attempt {attempt_count}[/bold cyan]" ) try: upload(args, dataset_id) break except ValueError: pass if attempt_count == args.repeat: console.print("[red]Upload failed. Aborting.[/red]") return 1 one_dataset_upload_time = time.monotonic() one_dataset_upload_duration = one_dataset_upload_time - one_dataset_convert_time print(f"`upload` took {one_dataset_upload_duration:.2f} seconds:") # Remove if needed. if args.remove_downloaded and not args.dry_run and rdf_file is not None: console.print(f"[bold cyan]Removing {rdf_file}[/bold cyan]") remove_dir(rdf_file.parent) # Success! console.print("[green]All done![/green]") total_duration = time.monotonic() - start_time print(f"`main` took {total_duration:.2f} seconds:") return 0 if __name__ == "__main__": exit(main())