Adding the features file.

This commit is contained in:
2025-12-23 15:07:41 -08:00
parent c487b49f11
commit eb8bfdbee5
9 changed files with 2037 additions and 0 deletions
+79
View File
@@ -0,0 +1,79 @@
"""Behave environment configuration with setup/teardown hooks.
This module handles:
- Temporary directory creation and cleanup
- Shared test fixtures
- Logging configuration for tests
"""
from __future__ import annotations
import logging
import shutil
import tempfile
from pathlib import Path
from typing import TYPE_CHECKING
if TYPE_CHECKING:
from behave.model import Feature, Scenario
from behave.runner import Context
# Suppress verbose logging from libraries during tests
logging.getLogger("rdflib").setLevel(logging.ERROR)
logging.getLogger("datasets").setLevel(logging.ERROR)
logging.getLogger("urllib3").setLevel(logging.ERROR)
logging.getLogger("filelock").setLevel(logging.ERROR)
def before_all(context: Context) -> None:
"""Run once before all tests - create base temp directory."""
context.base_temp_dir = Path(tempfile.mkdtemp(prefix="behave_rdf_tests_"))
context.cleanup_dirs = [] # list[Path]
def after_all(context: Context) -> None:
"""Run once after all tests - cleanup base temp directory."""
if hasattr(context, "base_temp_dir") and context.base_temp_dir.exists():
shutil.rmtree(context.base_temp_dir, ignore_errors=True)
# Cleanup any remaining registered directories
if hasattr(context, "cleanup_dirs"):
for cleanup_dir in context.cleanup_dirs:
if cleanup_dir.exists():
shutil.rmtree(cleanup_dir, ignore_errors=True)
def before_feature(context: Context, feature: Feature) -> None:
"""Run before each feature file."""
# Create feature-specific temp directory
feature_name = feature.name.replace(" ", "_").lower()[:30]
context.feature_temp_dir = context.base_temp_dir / f"feature_{feature_name}"
context.feature_temp_dir.mkdir(parents=True, exist_ok=True)
def after_feature(context: Context, feature: Feature) -> None:
"""Run after each feature file - cleanup feature temp directory."""
if hasattr(context, "feature_temp_dir") and context.feature_temp_dir.exists():
shutil.rmtree(context.feature_temp_dir, ignore_errors=True)
def before_scenario(context: Context, scenario: Scenario) -> None:
"""Run before each scenario - create scenario-specific temp directory."""
scenario_name = scenario.name.replace(" ", "_").lower()[:30]
context.scenario_temp_dir = context.feature_temp_dir / f"scenario_{scenario_name}"
context.scenario_temp_dir.mkdir(parents=True, exist_ok=True)
# Initialize scenario-specific attributes
context.input_path = None
context.output_path = None
context.result = None
context.rdf_format = None
context.expected_triple_count = None
def after_scenario(context: Context, scenario: Scenario) -> None:
"""Run after each scenario - cleanup scenario temp directory."""
if hasattr(context, "scenario_temp_dir") and context.scenario_temp_dir.exists():
shutil.rmtree(context.scenario_temp_dir, ignore_errors=True)
@@ -0,0 +1,82 @@
Feature: CLI Integration Tests
As a user running the converter from command line
I want the CLI to work correctly
So that I can convert RDF files without writing code
Background:
Given a temporary output directory
# ============================================
# BASIC CLI EXECUTION
# ============================================
Scenario: CLI converts N-Triples file with auto strategy
Given an N-Triples file with 50 triples
When I run the CLI with auto strategy
Then the CLI should exit successfully
And the output directory should contain a dataset
Scenario: CLI converts file with explicit standard strategy
Given an N-Triples file with 30 triples
When I run the CLI with strategy "standard"
Then the CLI should exit successfully
Scenario: CLI converts file with streaming strategy
Given an N-Triples file with 100 triples
When I run the CLI with strategy "streaming"
Then the CLI should exit successfully
Scenario: CLI with custom chunk size
Given an N-Triples file with 200 triples
When I run the CLI with chunk size 50
Then the CLI should exit successfully
# ============================================
# CLI WITH METADATA
# ============================================
Scenario: CLI with description and license
Given an N-Triples file with 25 triples
When I run the CLI with description "Test" and license "MIT"
Then the CLI should exit successfully
Scenario: CLI with all metadata flags
Given an N-Triples file with 20 triples
When I run the CLI with all metadata flags
Then the CLI should exit successfully
# ============================================
# CLI ERROR HANDLING
# ============================================
Scenario: CLI fails for non-existent input file
Given a non-existent input file path
When I run the CLI expecting failure
Then the CLI should exit with error
Scenario: CLI fails for empty file and reports error
Given an empty N-Triples file
When I run the CLI expecting failure
Then the CLI should exit with error
Scenario: CLI with verbose flag
Given an N-Triples file with 20 triples
When I run the CLI with verbose flag
Then the CLI should exit successfully
Scenario: CLI with clean-cache flag
Given an N-Triples file with 30 triples
When I run the CLI with clean-cache flag
Then the CLI should exit successfully
# ============================================
# CLI WITH TRAIN/TEST SPLIT
# ============================================
Scenario: CLI creates train/test split
Given an N-Triples file with 500 triples
When I run the CLI with train/test split
Then the CLI should exit successfully
And the output should have train and test splits
@@ -0,0 +1,129 @@
Feature: Error Handling
As a data engineer
I want clear error messages when conversions fail
So that I can quickly diagnose and fix issues
The converter should handle:
- Non-existent files
- Empty files
- Invalid RDF syntax
- Unsupported formats
Background:
Given a temporary output directory
# ============================================
# FILE NOT FOUND
# ============================================
Scenario: Fail gracefully for non-existent file
Given a non-existent input file path
When I attempt conversion using the "standard" strategy
Then the conversion should fail
And the error message should contain "No such file"
Scenario: Fail gracefully for non-existent file with streaming
Given a non-existent input file path
When I attempt conversion using the "streaming" strategy
Then the conversion should fail
# ============================================
# EMPTY FILES
# ============================================
Scenario: Handle empty N-Triples file
Given an empty N-Triples file
When I convert it using the "standard" strategy
Then the conversion should fail
And the error message should contain "No triples"
Scenario: Handle empty Turtle file
Given an empty Turtle file
When I convert it using the "standard" strategy
Then the conversion should fail
# ============================================
# INVALID SYNTAX
# ============================================
Scenario: Handle malformed N-Triples gracefully
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/p> "valid" .
this is not valid ntriples syntax
<http://example.org/s2> <http://example.org/p> "also valid" .
"""
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 2 triples
Scenario: Skip malformed lines in streaming mode
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/p> "v1" .
broken line without proper format
<http://example.org/s2> <http://example.org/p> "v2" .
another broken line
<http://example.org/s3> <http://example.org/p> "v3" .
"""
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 3 triples
# ============================================
# RECOVERY
# ============================================
Scenario: Recover from partially valid file
Given an N-Triples file containing
"""
<http://example.org/valid1> <http://example.org/p> "value1" .
<malformed uri> <http://example.org/p> "value2" .
<http://example.org/valid2> <http://example.org/p> "value3" .
"""
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain at least 2 triples
# ============================================
# CONVERSION RESULT CONSISTENCY
# ============================================
Scenario: Failed conversion returns error details
Given an empty N-Triples file
When I convert it using the "standard" strategy
Then the conversion should fail
And the result should have an error message
Scenario: Successful conversion has no error message
Given an N-Triples file with 10 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the result should not have an error message
# ============================================
# INVALID STRATEGY
# ============================================
Scenario: Invalid strategy name raises error
Given an N-Triples file with 10 triples
When I attempt conversion with invalid strategy "nonexistent"
Then a ValueError should be raised
# ============================================
# STREAMING ERROR RECOVERY
# ============================================
Scenario: Streaming recovers from parsing errors in ntriples
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/p> "valid1" .
<broken <http://example.org/p> "invalid" .
<http://example.org/s2> <http://example.org/p> "valid2" .
not a triple at all
<http://example.org/s3> <http://example.org/p> "valid3" .
"""
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain at least 3 triples
@@ -0,0 +1,288 @@
Feature: File Format and Compression Handling
As a data engineer working with various RDF sources
I want to convert files in different formats and compressions
So that I can handle real-world data without preprocessing
The converter supports:
- Formats: N-Triples, Turtle, RDF/XML, N3, TriG, NQuads, TSV
- Compression: gzip (.gz), bzip2 (.bz2)
- Automatic format and compression detection
Background:
Given a temporary output directory
# ============================================
# N-TRIPLES FORMAT
# ============================================
Scenario: Convert uncompressed N-Triples file
Given an N-Triples file with 100 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 100 triples
Scenario: N-Triples with URI objects
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/rel> <http://example.org/o1> .
<http://example.org/s2> <http://example.org/rel> <http://example.org/o2> .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 2 triples
And uri objects should have object_type as uri
Scenario: N-Triples with literal objects
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/name> "Alice" .
<http://example.org/s2> <http://example.org/name> "Bob" .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And literal objects should have object_type as literal
Scenario: N-Triples with language-tagged literals
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/label> "Katze"@de .
<http://example.org/s2> <http://example.org/label> ""@ja .
<http://example.org/s3> <http://example.org/label> "кошка"@ru .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 3 triples
Scenario: N-Triples with typed literals
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/age> "25"^^<http://www.w3.org/2001/XMLSchema#integer> .
<http://example.org/s2> <http://example.org/score> "98.6"^^<http://www.w3.org/2001/XMLSchema#double> .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 2 triples
# ============================================
# TURTLE FORMAT
# ============================================
Scenario: Convert Turtle file with prefixes
Given a Turtle file with 50 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 50 triples
Scenario: Turtle with complex prefix declarations
Given a Turtle file containing
"""
@prefix ex: <http://example.org/> .
@prefix foaf: <http://xmlns.com/foaf/0.1/> .
@prefix xsd: <http://www.w3.org/2001/XMLSchema#> .
ex:alice foaf:name "Alice" ;
foaf:age "30"^^xsd:integer .
ex:bob foaf:name "Bob" ;
foaf:knows ex:alice .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 4 triples
Scenario: Turtle with streaming-turtle strategy
Given a Turtle file with 200 triples
When I convert it using the "streaming-turtle" strategy
Then the conversion should succeed
And the output should contain 200 triples
# ============================================
# GZIP COMPRESSION
# ============================================
Scenario: Convert gzip-compressed N-Triples
Given a gzip-compressed N-Triples file with 150 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 150 triples
Scenario: Stream gzip-compressed N-Triples
Given a gzip-compressed N-Triples file with 200 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 200 triples
Scenario: Parallel process gzip-compressed N-Triples
Given a gzip-compressed N-Triples file with 250 triples
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain 250 triples
# ============================================
# BZ2 COMPRESSION
# ============================================
Scenario: Convert bz2-compressed N-Triples
Given a bz2-compressed N-Triples file with 100 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 100 triples
Scenario: Stream bz2-compressed N-Triples
Given a bz2-compressed N-Triples file with 150 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 150 triples
# ============================================
# TSV FORMAT
# ============================================
Scenario: Convert TSV file with standard strategy
Given a TSV file with 100 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 100 triples
And the output should have the correct schema
Scenario: TSV file creates literal object types
Given a TSV file with 50 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And all triples should have valid object_type values
Scenario: TSV file with empty lines and malformed rows
Given a TSV file with empty lines and malformed rows
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 3 triples
# ============================================
# RDF/XML FORMAT
# ============================================
Scenario: Convert RDF/XML file with standard strategy
Given an RDF/XML file with 50 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 50 triples
And the output should have the correct schema
# ============================================
# GEONAMES FORMAT
# ============================================
Scenario: Convert GeoNames format file with standard strategy
Given a GeoNames format file with 10 features
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should have the correct schema
Scenario: Convert GeoNames format file with streaming strategy
Given a GeoNames format file with 20 features
When I convert it using the "streaming" strategy
Then the conversion should succeed
Scenario: Convert GeoNames format file with parallel strategy
Given a GeoNames format file with 15 features
When I convert it using the "streaming-parallel" strategy with 2 workers
Then the conversion should succeed
Scenario: Convert GeoNames format file with simple streaming
Given a GeoNames format file with 10 features
When I convert it using the "streaming-simple" strategy
Then the conversion should succeed
# ============================================
# BLANK NODES
# ============================================
Scenario: Handle blank nodes in N-Triples
Given an N-Triples file containing
"""
_:b1 <http://example.org/name> "Anonymous" .
<http://example.org/s1> <http://example.org/knows> _:b1 .
_:b2 <http://example.org/type> "Thing" .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 3 triples
And blank node objects should have object_type as blank_node
Scenario: Handle blank nodes in Turtle
Given a Turtle file containing
"""
@prefix ex: <http://example.org/> .
ex:person1 ex:address [
ex:street "123 Main St" ;
ex:city "Springfield"
] .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
# ============================================
# TURTLE EDGE CASES
# ============================================
Scenario: Turtle with @base declaration
Given a Turtle file containing
"""
@base <http://example.org/> .
@prefix foaf: <http://xmlns.com/foaf/0.1/> .
<alice> foaf:name "Alice" .
<bob> foaf:name "Bob" .
"""
When I convert it using the "streaming-turtle" strategy
Then the conversion should succeed
And the output should contain 2 triples
Scenario: Turtle streaming with very small chunks
Given a Turtle file with 50 triples
And chunk size is set to 5
When I convert it using the "streaming-turtle" strategy
Then the conversion should succeed
And the output should contain 50 triples
Scenario: Turtle with special characters in literals
Given a Turtle file containing
"""
@prefix ex: <http://example.org/> .
ex:doc ex:content "This has special chars: quotes \\\"test\\\" and newline \\n here" .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 1 triples
# ============================================
# EDGE CASES
# ============================================
Scenario: Handle N-Triples with blank lines
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/p> "v1" .
<http://example.org/s2> <http://example.org/p> "v2" .
<http://example.org/s3> <http://example.org/p> "v3" .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 3 triples
Scenario: Handle N-Triples with comments
Given an N-Triples file containing
"""
# This is a comment
<http://example.org/s1> <http://example.org/p> "v1" .
# Another comment
<http://example.org/s2> <http://example.org/p> "v2" .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 2 triples
@@ -0,0 +1,162 @@
Feature: Parallel Streaming RDF to HuggingFace Conversion
As a data engineer processing very large RDF files
I want to convert RDF files using parallel multiprocessing
So that I can maximize throughput on multi-core systems
The parallel streaming strategy uses multiprocessing.Pool to
distribute parsing work across multiple CPU cores. This is
optimal for files over 1GB on systems with multiple cores.
Background:
Given a temporary output directory
# ============================================
# BASIC PARALLEL CONVERSION
# ============================================
Scenario: Convert N-Triples file using parallel strategy
Given an N-Triples file with 500 triples
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain 500 triples
And the output should have the correct schema
Scenario: Convert Turtle file using parallel strategy
Given a Turtle file with 300 triples
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain 300 triples
# ============================================
# WORKER CONFIGURATION
# ============================================
Scenario: Parallel conversion with 2 workers
Given an N-Triples file with 400 triples
When I convert it using the "streaming-parallel" strategy with 2 workers
Then the conversion should succeed
And the output should contain 400 triples
And the output should have the correct schema
Scenario: Parallel conversion with 4 workers
Given an N-Triples file with 600 triples
When I convert it using the "streaming-parallel" strategy with 4 workers
Then the conversion should succeed
And the output should contain 600 triples
Scenario: Single worker behaves like serial processing
Given an N-Triples file with 200 triples
When I convert it using the "streaming-parallel" strategy with 1 workers
Then the conversion should succeed
And the output should contain 200 triples
# ============================================
# CHUNK SIZE CONFIGURATION
# ============================================
Scenario: Parallel with small chunk size creates more batches
Given an N-Triples file with 500 triples
And chunk size is set to 50
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain the expected number of triples
Scenario: Parallel with large chunk size
Given an N-Triples file with 300 triples
And chunk size is set to 500
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain 300 triples
# ============================================
# SCHEMA AND DATA INTEGRITY
# ============================================
Scenario: Parallel output has all required schema columns
Given an N-Triples file with 100 triples
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should have the correct schema
Scenario: Parallel preserves object type metadata
Given an N-Triples file with 150 triples and mixed object types
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And all triples should have valid object_type values
Scenario: Parallel conversion produces same results as serial
Given an N-Triples file with 300 triples
When I convert it using the "streaming-parallel" strategy with 2 workers
Then the conversion should succeed
And the output should contain 300 triples
And the output should have the correct schema
# ============================================
# DATASET INFO
# ============================================
Scenario: Parallel conversion creates dataset_info.json
Given an N-Triples file with 200 triples
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the dataset_info.json file should exist
And the dataset_info.json should contain total_triples
Scenario: Parallel dataset_info.json includes worker count
Given an N-Triples file with 100 triples
When I convert it using the "streaming-parallel" strategy with 2 workers
Then the conversion should succeed
And the dataset_info.json file should exist
And the dataset_info.json should contain num_workers
# ============================================
# PROCESSING METRICS
# ============================================
Scenario: Parallel conversion records processing time
Given an N-Triples file with 400 triples
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And processing time should be recorded
# ============================================
# LITERAL HANDLING
# ============================================
Scenario: Parallel correctly processes literals with language tags
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/label> "hello"@en .
<http://example.org/s2> <http://example.org/label> "hola"@es .
<http://example.org/s3> <http://example.org/label> "ciao"@it .
<http://example.org/s4> <http://example.org/label> "olá"@pt .
"""
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain 4 triples
And literal objects should have object_type as literal
Scenario: Parallel correctly processes URI objects
Given an N-Triples file containing
"""
<http://example.org/alice> <http://example.org/knows> <http://example.org/bob> .
<http://example.org/bob> <http://example.org/knows> <http://example.org/carol> .
<http://example.org/carol> <http://example.org/knows> <http://example.org/dave> .
"""
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain 3 triples
And uri objects should have object_type as uri
# ============================================
# PARALLEL TURTLE CONVERSION
# ============================================
Scenario: Parallel processes Turtle file with chunking
Given a Turtle file with 200 triples
And chunk size is set to 50
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain 200 triples
@@ -0,0 +1,126 @@
Feature: Standard In-Memory RDF to HuggingFace Conversion
As a data engineer processing small RDF files
I want to convert RDF files using in-memory processing
So that I get fast, simple conversions for files under 100MB
The standard strategy loads the entire RDF graph into memory,
converts all triples at once, and creates the dataset directly.
This is optimal for files under 100MB.
Background:
Given a temporary output directory
# ============================================
# BASIC STANDARD CONVERSION
# ============================================
Scenario: Convert N-Triples file using standard strategy
Given an N-Triples file with 100 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 100 triples
And the output should have the correct schema
Scenario: Convert Turtle file using standard strategy
Given a Turtle file with 150 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 150 triples
And the output should have the correct schema
Scenario: Convert small file quickly with standard strategy
Given an N-Triples file with 50 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And processing time should be recorded
# ============================================
# TRAIN/TEST SPLIT
# ============================================
Scenario: Standard conversion with train/test split
Given an N-Triples file with 1000 triples
When I convert it using the "standard" strategy with train/test split
Then the conversion should succeed
And the output should have train and test splits
Scenario: Standard conversion creates data split by default
Given an N-Triples file with 200 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should have a data split
# ============================================
# SCHEMA VALIDATION
# ============================================
Scenario: Standard output has all required schema columns
Given an N-Triples file with 25 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should have the correct schema
Scenario: Standard preserves object type metadata
Given an N-Triples file with 60 triples and mixed object types
When I convert it using the "standard" strategy
Then the conversion should succeed
And all triples should have valid object_type values
# ============================================
# DATASET INFO
# ============================================
Scenario: Standard conversion creates dataset_info.json
Given an N-Triples file with 75 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the dataset_info.json file should exist
And the dataset_info.json should contain total_triples
# ============================================
# METADATA HANDLING
# ============================================
Scenario: Standard conversion with metadata
Given an N-Triples file with 50 triples
And metadata with description "Test dataset" and license "MIT"
When I convert it using the "standard" strategy with metadata
Then the conversion should succeed
And the dataset should have description metadata
Scenario: Metadata with citation and homepage
Given an N-Triples file with 30 triples
And full metadata with all fields
When I convert it using the "standard" strategy with metadata
Then the conversion should succeed
And the dataset should have all metadata fields
# ============================================
# LITERAL HANDLING
# ============================================
Scenario: Standard correctly processes literals with datatypes
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/p> "42"^^<http://www.w3.org/2001/XMLSchema#integer> .
<http://example.org/s2> <http://example.org/p> "3.14"^^<http://www.w3.org/2001/XMLSchema#decimal> .
<http://example.org/s3> <http://example.org/p> "true"^^<http://www.w3.org/2001/XMLSchema#boolean> .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 3 triples
And literal objects should have object_type as literal
Scenario: Standard correctly processes URI objects
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/rel> <http://example.org/o1> .
<http://example.org/s2> <http://example.org/rel> <http://example.org/o2> .
<http://example.org/s3> <http://example.org/rel> <http://example.org/o3> .
"""
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 3 triples
And uri objects should have object_type as uri
@@ -0,0 +1,119 @@
Feature: Automatic Strategy Selection
As a data engineer
I want the converter to automatically select the best strategy
So that I get optimal performance without manual configuration
The auto-selection logic considers:
- File size (< 100MB: standard, 100MB-1GB: streaming, > 1GB: parallel)
- RDF format (Turtle files may use streaming-turtle)
- GeoNames format detection
- Available CPU cores
Background:
Given a temporary output directory
# ============================================
# SMALL FILE SELECTION (< 100MB)
# ============================================
Scenario: Auto selects standard strategy for small N-Triples file
Given an N-Triples file with 50 triples
When I convert it using the "auto" strategy
Then the conversion should succeed
And the output should contain 50 triples
And the output should have the correct schema
Scenario: Auto selects strategy for small Turtle file
Given a Turtle file with 100 triples
When I convert it using the "auto" strategy
Then the conversion should succeed
And the output should contain 100 triples
# ============================================
# FORMAT DETECTION
# ============================================
Scenario: Auto handles N-Triples format correctly
Given an N-Triples file with 75 triples
When I convert it using the "auto" strategy
Then the conversion should succeed
And the output should have the correct schema
Scenario: Auto handles Turtle format correctly
Given a Turtle file with 80 triples
When I convert it using the "auto" strategy
Then the conversion should succeed
And the output should have the correct schema
# ============================================
# MANUAL STRATEGY OVERRIDE
# ============================================
Scenario: Can override auto with manual standard selection
Given an N-Triples file with 200 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should contain 200 triples
Scenario: Can override auto with manual streaming selection
Given an N-Triples file with 150 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 150 triples
Scenario: Can override auto with manual parallel selection
Given an N-Triples file with 250 triples
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should contain 250 triples
Scenario: Can override auto with streaming-simple selection
Given an N-Triples file with 100 triples
When I convert it using the "streaming-simple" strategy
Then the conversion should succeed
And the output should contain 100 triples
Scenario: Can override auto with streaming-turtle selection
Given a Turtle file with 120 triples
When I convert it using the "streaming-turtle" strategy
Then the conversion should succeed
And the output should contain 120 triples
# ============================================
# STRATEGY PRODUCES CONSISTENT RESULTS
# ============================================
Scenario: All strategies produce correct schema
Given an N-Triples file with 100 triples
When I convert it using the "standard" strategy
Then the conversion should succeed
And the output should have the correct schema
Scenario: Streaming strategy produces correct schema
Given an N-Triples file with 100 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should have the correct schema
Scenario: Parallel strategy produces correct schema
Given an N-Triples file with 100 triples
When I convert it using the "streaming-parallel" strategy
Then the conversion should succeed
And the output should have the correct schema
Scenario: Simple streaming produces correct schema
Given an N-Triples file with 100 triples
When I convert it using the "streaming-simple" strategy
Then the conversion should succeed
And the output should have the correct schema
# ============================================
# STRATEGY LISTING
# ============================================
Scenario: List all available strategies
When I list all available strategies
Then I should get 5 strategies
And strategies should include "standard, streaming"
@@ -0,0 +1,172 @@
Feature: Streaming RDF to HuggingFace Conversion
As a data engineer processing medium to large RDF files
I want to convert RDF files using chunked streaming
So that I can process files that don't fit in memory
The streaming strategy processes files in configurable chunks,
writes intermediate Parquet files, and merges them into a final dataset.
This is optimal for files between 100MB and 1GB.
Background:
Given a temporary output directory
# ============================================
# BASIC STREAMING CONVERSION
# ============================================
Scenario: Convert N-Triples file using streaming strategy
Given an N-Triples file with 500 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 500 triples
And the output should have the correct schema
Scenario: Convert Turtle file using streaming strategy
Given a Turtle file with 200 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 200 triples
And the output should have the correct schema
Scenario: Streaming conversion with small chunk size
Given an N-Triples file with 500 triples
And chunk size is set to 50
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain the expected number of triples
# ============================================
# TRAIN/TEST SPLIT
# ============================================
Scenario: Streaming conversion creates train/test splits when requested
Given an N-Triples file with 1000 triples
When I convert it using the "streaming" strategy with train/test split
Then the conversion should succeed
And the output should have train and test splits
Scenario: Streaming conversion creates data split by default
Given an N-Triples file with 100 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should have a data split
# ============================================
# COMPRESSION SUPPORT
# ============================================
Scenario: Stream gzip-compressed N-Triples file
Given a gzip-compressed N-Triples file with 300 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 300 triples
# ============================================
# SCHEMA VALIDATION
# ============================================
Scenario: Streaming output has all required schema columns
Given an N-Triples file with 50 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should have the correct schema
Scenario: Streaming preserves object type metadata
Given an N-Triples file with 100 triples and mixed object types
When I convert it using the "streaming" strategy
Then the conversion should succeed
And all triples should have valid object_type values
# ============================================
# DATASET INFO JSON
# ============================================
Scenario: Streaming conversion creates dataset_info.json
Given an N-Triples file with 100 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the dataset_info.json file should exist
And the dataset_info.json should contain total_triples
# ============================================
# PROCESSING METRICS
# ============================================
Scenario: Streaming conversion records processing time
Given an N-Triples file with 200 triples
When I convert it using the "streaming" strategy
Then the conversion should succeed
And processing time should be recorded
# ============================================
# EDGE CASES
# ============================================
Scenario: Streaming handles file with exactly one chunk
Given an N-Triples file with 50 triples
And chunk size is set to 100
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 50 triples
Scenario: Streaming handles file smaller than chunk size
Given an N-Triples file with 10 triples
And chunk size is set to 1000
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 10 triples
# ============================================
# LITERAL HANDLING
# ============================================
Scenario: Streaming correctly processes literals with language tags
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/p> "hello"@en .
<http://example.org/s2> <http://example.org/p> "bonjour"@fr .
<http://example.org/s3> <http://example.org/p> "hallo"@de .
"""
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 3 triples
And literal objects should have object_type as literal
Scenario: Streaming correctly processes URI objects
Given an N-Triples file containing
"""
<http://example.org/s1> <http://example.org/p> <http://example.org/o1> .
<http://example.org/s2> <http://example.org/p> <http://example.org/o2> .
"""
When I convert it using the "streaming" strategy
Then the conversion should succeed
And the output should contain 2 triples
And uri objects should have object_type as uri
# ============================================
# STREAMING TURTLE WITH MULTILINE
# ============================================
Scenario: Streaming-turtle handles multiline statements
Given a Turtle file with multiline statements
When I convert it using the "streaming-turtle" strategy
Then the conversion should succeed
And the output should contain 2 triples
Scenario: Streaming-turtle with small chunk triggers chunking
Given a Turtle file with 100 triples
And chunk size is set to 10
When I convert it using the "streaming-turtle" strategy
Then the conversion should succeed
And the output should contain 100 triples
# ============================================
# STREAMING WITHOUT CLEAN CACHE
# ============================================
Scenario: Streaming conversion without clean cache
Given an N-Triples file with 100 triples
When I convert it using streaming without clean cache
Then the conversion should succeed
And the output should contain 100 triples
+880
View File
@@ -0,0 +1,880 @@
"""Step definitions for RDF to HuggingFace converter tests.
This module provides reusable step definitions for testing:
- All conversion strategies (standard, streaming, parallel, etc.)
- File format handling (N-Triples, Turtle, RDF/XML, etc.)
- Compression support (gzip, bz2)
- Schema validation
- Error handling
"""
from __future__ import annotations
import bz2
import gzip
import sys
from pathlib import Path
from typing import TYPE_CHECKING, Any
from behave import given, then, when
# Add scripts directory to path for imports
SCRIPTS_DIR = Path(__file__).parent.parent.parent / "scripts"
sys.path.insert(0, str(SCRIPTS_DIR))
from convert_rdf_to_hf_dataset_unified import (
ConversionConfig,
ConversionResult,
FileHandler,
ProgressTracker,
SchemaManager,
StrategySelector,
)
if TYPE_CHECKING:
from behave.runner import Context
# HELPER FUNCTIONS
def _create_ntriples_content(count: int, include_types: bool = False) -> str:
"""Generate N-Triples content with specified number of triples."""
lines = []
for i in range(count):
if include_types and i % 3 == 0:
# URI object
lines.append(
f"<http://example.org/s{i}> "
f"<http://example.org/p> "
f"<http://example.org/o{i}> ."
)
elif include_types and i % 3 == 1:
# Literal with language
lines.append(
f"<http://example.org/s{i}> "
f"<http://example.org/p> "
f'"value{i}"@en .'
)
else:
# Plain literal
lines.append(
f"<http://example.org/s{i}> "
f"<http://example.org/p> "
f'"value{i}" .'
)
return "\n".join(lines)
def _create_turtle_content(count: int) -> str:
"""Generate Turtle content with prefixes and specified number of triples."""
lines = [
"@prefix ex: <http://example.org/> .",
"@prefix xsd: <http://www.w3.org/2001/XMLSchema#> .",
"",
]
for i in range(count):
lines.append(f'ex:s{i} ex:p "value{i}" .')
return "\n".join(lines)
def _get_strategy(context: Context, strategy_name: str):
"""Get a conversion strategy by name."""
progress = ProgressTracker(verbose=False)
selector = StrategySelector(progress)
if strategy_name == "auto":
return selector.select_auto(context.input_path, context.rdf_format)
return selector.select_manual(strategy_name)
def _run_conversion(
context: Context,
strategy_name: str,
num_workers: int | None = None,
chunk_size: int = 1000,
create_splits: bool = False,
) -> ConversionResult:
"""Execute conversion and store result in context."""
strategy = _get_strategy(context, strategy_name)
config = ConversionConfig(
input_path=context.input_path,
output_path=context.output_path,
rdf_format=context.rdf_format,
chunk_size=chunk_size,
num_workers=num_workers,
create_train_test_split=create_splits,
clean_cache=True,
)
return strategy.convert(config)
def _run_conversion_with_metadata(
context: Context,
strategy_name: str,
num_workers: int | None = None,
chunk_size: int = 1000,
metadata: dict | None = None,
) -> ConversionResult:
"""Execute conversion with metadata."""
strategy = _get_strategy(context, strategy_name)
config = ConversionConfig(
input_path=context.input_path,
output_path=context.output_path,
rdf_format=context.rdf_format,
chunk_size=chunk_size,
num_workers=num_workers,
metadata=metadata,
clean_cache=True,
)
return strategy.convert(config)
def _load_output_dataset(context: Context):
"""Load the output dataset from disk."""
from datasets import load_from_disk
dataset_dict = load_from_disk(str(context.output_path))
# Get the actual dataset (could be 'data', 'train', etc.)
if "data" in dataset_dict:
return dataset_dict, dataset_dict["data"]
if "train" in dataset_dict:
return dataset_dict, dataset_dict["train"]
return dataset_dict, list(dataset_dict.values())[0]
# =============================================================================
# GIVEN STEPS - Test Setup
# =============================================================================
@given("a temporary output directory")
def step_create_temp_output_dir(context: Context) -> None:
"""Create a temporary output directory for the conversion."""
context.output_path = context.scenario_temp_dir / "output"
context.output_path.mkdir(parents=True, exist_ok=True)
@given("an N-Triples file with {count:d} triples")
def step_create_ntriples_file(context: Context, count: int) -> None:
"""Create a test N-Triples file with the specified number of triples."""
context.input_path = context.scenario_temp_dir / "test.nt"
content = _create_ntriples_content(count)
context.input_path.write_text(content)
context.expected_triple_count = count
context.rdf_format = "nt"
@given("an N-Triples file with {count:d} triples and mixed object types")
def step_create_mixed_ntriples_file(context: Context, count: int) -> None:
"""Create N-Triples with URIs, literals with language tags, and plain literals."""
context.input_path = context.scenario_temp_dir / "test_mixed.nt"
content = _create_ntriples_content(count, include_types=True)
context.input_path.write_text(content)
context.expected_triple_count = count
context.rdf_format = "nt"
@given("a Turtle file with {count:d} triples")
def step_create_turtle_file(context: Context, count: int) -> None:
"""Create a test Turtle file with prefixes and specified number of triples."""
context.input_path = context.scenario_temp_dir / "test.ttl"
content = _create_turtle_content(count)
context.input_path.write_text(content)
context.expected_triple_count = count
context.rdf_format = "turtle"
@given("a gzip-compressed N-Triples file with {count:d} triples")
def step_create_gzipped_ntriples(context: Context, count: int) -> None:
"""Create a gzip-compressed N-Triples file."""
context.input_path = context.scenario_temp_dir / "test.nt.gz"
content = _create_ntriples_content(count)
with gzip.open(context.input_path, "wt", encoding="utf-8") as f:
f.write(content)
context.expected_triple_count = count
context.rdf_format = "nt"
@given("an N-Triples file containing")
def step_create_ntriples_from_docstring(context: Context) -> None:
"""Create an N-Triples file from the scenario's docstring content."""
context.input_path = context.scenario_temp_dir / "test_custom.nt"
context.input_path.write_text(context.text.strip())
context.rdf_format = "nt"
# Count non-empty, non-comment lines
lines = [
line
for line in context.text.strip().split("\n")
if line.strip() and not line.strip().startswith("#")
]
context.expected_triple_count = len(lines)
@given("a Turtle file containing")
def step_create_turtle_from_docstring(context: Context) -> None:
"""Create a Turtle file from the scenario's docstring content."""
context.input_path = context.scenario_temp_dir / "test_custom.ttl"
context.input_path.write_text(context.text.strip())
context.rdf_format = "turtle"
@given("chunk size is set to {size:d}")
def step_set_chunk_size(context: Context, size: int) -> None:
"""Set the chunk size for streaming conversion."""
context.chunk_size = size
@given("a bz2-compressed N-Triples file with {count:d} triples")
def step_create_bz2_ntriples(context: Context, count: int) -> None:
"""Create a bz2-compressed N-Triples file."""
context.input_path = context.scenario_temp_dir / "test.nt.bz2"
content = _create_ntriples_content(count)
with bz2.open(context.input_path, "wt", encoding="utf-8") as f:
f.write(content)
context.expected_triple_count = count
context.rdf_format = "nt"
@given("a TSV file with {count:d} triples")
def step_create_tsv_file(context: Context, count: int) -> None:
"""Create a TSV file with subject/predicate/object columns."""
context.input_path = context.scenario_temp_dir / "test.tsv"
lines = []
for i in range(count):
lines.append(f"entity{i}\trelation{i % 10}\tentity{i + 1}")
context.input_path.write_text("\n".join(lines))
context.expected_triple_count = count
context.rdf_format = "tsv"
@given("a non-existent input file path")
def step_set_nonexistent_file(context: Context) -> None:
"""Set input path to a non-existent file."""
context.input_path = context.scenario_temp_dir / "does_not_exist.nt"
context.rdf_format = "nt"
@given("an empty N-Triples file")
def step_create_empty_ntriples(context: Context) -> None:
"""Create an empty N-Triples file."""
context.input_path = context.scenario_temp_dir / "empty.nt"
context.input_path.write_text("")
context.rdf_format = "nt"
@given("an empty Turtle file")
def step_create_empty_turtle(context: Context) -> None:
"""Create an empty Turtle file."""
context.input_path = context.scenario_temp_dir / "empty.ttl"
context.input_path.write_text("")
context.rdf_format = "turtle"
@given("an RDF/XML file with {count:d} triples")
def step_create_rdfxml_file(context: Context, count: int) -> None:
"""Create an RDF/XML file with specified number of triples."""
context.input_path = context.scenario_temp_dir / "test.rdf"
triples_xml = "\n".join(
f' <rdf:Description rdf:about="http://example.org/s{i}">\n'
f' <ex:property>value{i}</ex:property>\n'
f' </rdf:Description>'
for i in range(count)
)
content = f'''<?xml version="1.0" encoding="UTF-8"?>
<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
xmlns:ex="http://example.org/">
{triples_xml}
</rdf:RDF>'''
context.input_path.write_text(content)
context.expected_triple_count = count
context.rdf_format = "xml"
@given('metadata with description "{desc}" and license "{lic}"')
def step_set_metadata(context: Context, desc: str, lic: str) -> None:
"""Set metadata for the conversion."""
context.metadata = {"description": desc, "license": lic}
@given("full metadata with all fields")
def step_set_full_metadata(context: Context) -> None:
"""Set metadata with all fields: description, citation, homepage, license."""
context.metadata = {
"description": "Test dataset description",
"citation": "@misc{test2024, title={Test}}",
"homepage": "https://example.org/dataset",
"license": "MIT",
}
@given("a TSV file with empty lines and malformed rows")
def step_create_tsv_with_issues(context: Context) -> None:
"""Create a TSV file with empty lines and rows that don't have 3 columns."""
context.input_path = context.scenario_temp_dir / "test_issues.tsv"
content = """entity1\trelation1\tentity2
entity3\trelation3\tentity4
malformed_row_only_two_columns\tvalue
entity5\trelation5\tentity6
single_column
"""
context.input_path.write_text(content)
context.expected_triple_count = 3
context.rdf_format = "tsv"
@given("a Turtle file with multiline statements")
def step_create_turtle_multiline(context: Context) -> None:
"""Create a Turtle file with statements spanning multiple lines."""
context.input_path = context.scenario_temp_dir / "multiline.ttl"
content = """@prefix ex: <http://example.org/> .
@prefix foaf: <http://xmlns.com/foaf/0.1/> .
ex:person1 foaf:name "Alice" ;
foaf:age "30" .
"""
context.input_path.write_text(content)
context.expected_triple_count = 2
context.rdf_format = "turtle"
@given("a GeoNames format file with {count:d} features")
def step_create_geonames_file(context: Context, count: int) -> None:
"""Create a GeoNames-style RDF/XML file with URL prefixes."""
context.input_path = context.scenario_temp_dir / "geonames_test.xml"
docs = []
for i in range(count):
docs.append(f"""http://sws.geonames.org/{1000000 + i}/
<?xml version="1.0" encoding="UTF-8"?>
<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
xmlns:gn="http://www.geonames.org/ontology#">
<gn:Feature rdf:about="http://sws.geonames.org/{1000000 + i}/">
<gn:name>Location{i}</gn:name>
</gn:Feature>
</rdf:RDF>""")
context.input_path.write_text("\n".join(docs))
context.expected_triple_count = count
context.rdf_format = "xml"
# =============================================================================
# WHEN STEPS - Execute Actions
# =============================================================================
@when('I convert it using the "{strategy_name}" strategy')
def step_convert_with_strategy(context: Context, strategy_name: str) -> None:
"""Execute conversion with the specified strategy."""
chunk_size = getattr(context, "chunk_size", 1000)
context.result = _run_conversion(context, strategy_name, chunk_size=chunk_size)
context.strategy_used = strategy_name
@when('I convert it using the "{strategy_name}" strategy with {workers:d} workers')
def step_convert_with_workers(context: Context, strategy_name: str, workers: int) -> None:
"""Execute conversion with specified strategy and worker count."""
chunk_size = getattr(context, "chunk_size", 1000)
context.result = _run_conversion(
context, strategy_name, num_workers=workers, chunk_size=chunk_size
)
context.strategy_used = strategy_name
context.num_workers_used = workers
@when('I convert it using the "{strategy_name}" strategy with train/test split')
def step_convert_with_split(context: Context, strategy_name: str) -> None:
"""Execute conversion with train/test split enabled."""
chunk_size = getattr(context, "chunk_size", 1000)
context.result = _run_conversion(
context, strategy_name, chunk_size=chunk_size, create_splits=True
)
context.strategy_used = strategy_name
@when("I convert it using streaming without clean cache")
def step_convert_streaming_no_clean(context: Context) -> None:
"""Execute streaming conversion without clean_cache flag."""
strategy = _get_strategy(context, "streaming")
config = ConversionConfig(
input_path=context.input_path,
output_path=context.output_path,
rdf_format=context.rdf_format,
chunk_size=getattr(context, "chunk_size", 1000),
clean_cache=False, # This is the key difference
)
context.result = strategy.convert(config)
context.strategy_used = "streaming"
@when('I attempt conversion using the "{strategy_name}" strategy')
def step_attempt_conversion(context: Context, strategy_name: str) -> None:
"""Attempt conversion that may fail (e.g., file not found)."""
chunk_size = getattr(context, "chunk_size", 1000)
try:
context.result = _run_conversion(context, strategy_name, chunk_size=chunk_size)
except FileNotFoundError as e:
context.result = ConversionResult(
success=False,
error_message=f"File not found: {e}"
)
except Exception as e:
context.result = ConversionResult(
success=False,
error_message=str(e)
)
context.strategy_used = strategy_name
@when('I convert it using the "{strategy_name}" strategy with metadata')
def step_convert_with_metadata(context: Context, strategy_name: str) -> None:
"""Execute conversion with metadata."""
chunk_size = getattr(context, "chunk_size", 1000)
metadata = getattr(context, "metadata", None)
context.result = _run_conversion_with_metadata(
context, strategy_name, chunk_size=chunk_size, metadata=metadata
)
context.strategy_used = strategy_name
@when('I attempt conversion with invalid strategy "{strategy_name}"')
def step_attempt_invalid_strategy(context: Context, strategy_name: str) -> None:
"""Attempt conversion with an invalid strategy name."""
context.raised_exception = None
try:
_get_strategy(context, strategy_name)
except ValueError as e:
context.raised_exception = e
@when("I list all available strategies")
def step_list_strategies(context: Context) -> None:
"""List all available strategies."""
progress = ProgressTracker(verbose=False)
selector = StrategySelector(progress)
context.strategies_list = selector.list_strategies()
# =============================================================================
# THEN STEPS - Verify Outcomes
# =============================================================================
@then("the conversion should succeed")
def step_verify_success(context: Context) -> None:
"""Verify the conversion completed successfully."""
assert context.result is not None, "No conversion result found"
assert context.result.success, (
f"Conversion failed: {context.result.error_message}"
)
@then("the conversion should fail")
def step_verify_failure(context: Context) -> None:
"""Verify the conversion failed as expected."""
assert context.result is not None, "No conversion result found"
assert not context.result.success, "Expected conversion to fail but it succeeded"
@then("the conversion should fail with error containing {error_text}")
def step_verify_failure_message(context: Context, error_text: str) -> None:
"""Verify the conversion failed with a specific error message."""
assert not context.result.success, "Expected conversion to fail"
assert error_text in (context.result.error_message or ""), (
f"Expected error containing '{error_text}', "
f"got: {context.result.error_message}"
)
@then("the output should contain {count:d} triples")
def step_verify_triple_count(context: Context, count: int) -> None:
"""Verify the number of triples in the output dataset."""
assert context.result.total_triples == count, (
f"Expected {count} triples, got {context.result.total_triples}"
)
@then("the output should contain the expected number of triples")
def step_verify_expected_triple_count(context: Context) -> None:
"""Verify the output contains the expected number of triples from setup."""
expected = context.expected_triple_count
assert context.result.total_triples == expected, (
f"Expected {expected} triples, got {context.result.total_triples}"
)
@then("the output should have the correct schema")
def step_verify_schema(context: Context) -> None:
"""Verify the output dataset has the expected RDF triple schema."""
_, dataset = _load_output_dataset(context)
expected_columns = {
"subject",
"predicate",
"object",
"object_type",
"object_datatype",
"object_language",
}
actual_columns = set(dataset.column_names)
assert expected_columns == actual_columns, (
f"Schema mismatch. Expected {expected_columns}, got {actual_columns}"
)
@then("the output should have a {split_name} split")
def step_verify_split_exists(context: Context, split_name: str) -> None:
"""Verify the output has a specific split (train, test, data)."""
from datasets import load_from_disk
dataset_dict = load_from_disk(str(context.output_path))
assert split_name in dataset_dict, (
f"Expected '{split_name}' split, found: {list(dataset_dict.keys())}"
)
@then("the output should have train and test splits")
def step_verify_train_test_splits(context: Context) -> None:
"""Verify the output has both train and test splits."""
from datasets import load_from_disk
dataset_dict = load_from_disk(str(context.output_path))
assert "train" in dataset_dict, f"Missing 'train' split: {list(dataset_dict.keys())}"
assert "test" in dataset_dict, f"Missing 'test' split: {list(dataset_dict.keys())}"
@then("the dataset_info.json file should exist")
def step_verify_info_json_exists(context: Context) -> None:
"""Verify the dataset_info.json file was created."""
info_path = context.output_path / "dataset_info.json"
assert info_path.exists(), f"dataset_info.json not found at {info_path}"
@then("the dataset_info.json should contain total_triples")
def step_verify_info_has_total_triples(context: Context) -> None:
"""Verify dataset_info.json contains total_triples field."""
import json
info_path = context.output_path / "dataset_info.json"
with open(info_path) as f:
info = json.load(f)
assert "total_triples" in info, f"Missing 'total_triples' in info: {info.keys()}"
assert info["total_triples"] == context.result.total_triples
@then("processing time should be recorded")
def step_verify_processing_time(context: Context) -> None:
"""Verify processing time was recorded."""
assert context.result.processing_time_seconds > 0, "Processing time not recorded"
@then("all triples should have valid object_type values")
def step_verify_object_types(context: Context) -> None:
"""Verify all triples have valid object_type (uri, literal, blank_node)."""
_, dataset = _load_output_dataset(context)
valid_types = {"uri", "literal", "blank_node"}
for row in dataset:
obj_type = row["object_type"]
assert obj_type in valid_types, (
f"Invalid object_type: {obj_type}. Expected one of {valid_types}"
)
@then("literal objects should have object_type as literal")
def step_verify_literal_type(context: Context) -> None:
"""Verify literal objects have correct object_type."""
_, dataset = _load_output_dataset(context)
for row in dataset:
if row["object"].startswith('"') or row["object_language"]:
assert row["object_type"] == "literal", (
f"Expected object_type 'literal' for {row['object']}"
)
@then("uri objects should have object_type as uri")
def step_verify_uri_type(context: Context) -> None:
"""Verify URI objects have correct object_type."""
_, dataset = _load_output_dataset(context)
for row in dataset:
if row["object"].startswith("http://") or row["object"].startswith("https://"):
if row["object_type"] != "blank_node":
assert row["object_type"] == "uri", (
f"Expected object_type 'uri' for {row['object']}"
)
@then('the error message should contain "{text}"')
def step_verify_error_contains(context: Context, text: str) -> None:
"""Verify error message contains expected text."""
assert context.result is not None, "No conversion result found"
assert not context.result.success, "Expected conversion to fail"
error_msg = context.result.error_message or ""
assert text.lower() in error_msg.lower(), (
f"Expected error containing '{text}', got: {error_msg}"
)
@then("the result should have an error message")
def step_verify_has_error_message(context: Context) -> None:
"""Verify the result has an error message."""
assert context.result is not None, "No conversion result found"
assert context.result.error_message, "Expected an error message but found none"
@then("the result should not have an error message")
def step_verify_no_error_message(context: Context) -> None:
"""Verify the result has no error message."""
assert context.result is not None, "No conversion result found"
assert not context.result.error_message, (
f"Expected no error message but got: {context.result.error_message}"
)
@then("the output should contain at least {count:d} triples")
def step_verify_min_triple_count(context: Context, count: int) -> None:
"""Verify the output contains at least the specified number of triples."""
assert context.result.total_triples >= count, (
f"Expected at least {count} triples, got {context.result.total_triples}"
)
@then("the dataset_info.json should contain num_workers")
def step_verify_info_has_workers(context: Context) -> None:
"""Verify dataset_info.json contains num_workers field."""
import json
info_path = context.output_path / "dataset_info.json"
with open(info_path) as f:
info = json.load(f)
assert "num_workers" in info, f"Missing 'num_workers' in info: {info.keys()}"
@then("blank node objects should have object_type as blank_node")
def step_verify_blank_node_type(context: Context) -> None:
"""Verify at least one triple has a blank_node object_type."""
_, dataset = _load_output_dataset(context)
blank_node_count = sum(
1 for row in dataset if row["object_type"] == "blank_node"
)
assert blank_node_count > 0, (
f"Expected at least one blank_node object, found none. "
f"Object types: {[row['object_type'] for row in dataset]}"
)
@then("the dataset should have description metadata")
def step_verify_description_metadata(context: Context) -> None:
"""Verify the dataset has description metadata."""
dataset_dict, dataset = _load_output_dataset(context)
info = dataset.info
assert info.description, "Dataset has no description metadata"
@then("the dataset should have all metadata fields")
def step_verify_all_metadata(context: Context) -> None:
"""Verify the dataset has all metadata fields."""
dataset_dict, dataset = _load_output_dataset(context)
info = dataset.info
assert isinstance(info.description, str) and info.description, "Missing or invalid description"
assert isinstance(info.citation, str) and info.citation, "Missing or invalid citation"
assert isinstance(info.homepage, str) and info.homepage, "Missing or invalid homepage"
assert isinstance(info.license, str) and info.license, "Missing or invalid license"
@then("a ValueError should be raised")
def step_verify_valueerror(context: Context) -> None:
"""Verify a ValueError was raised."""
assert context.raised_exception is not None, "No exception was raised"
assert isinstance(context.raised_exception, ValueError), (
f"Expected ValueError, got {type(context.raised_exception)}"
)
@then("I should get {count:d} strategies")
def step_verify_strategy_count(context: Context, count: int) -> None:
"""Verify the number of strategies returned."""
assert len(context.strategies_list) == count, (
f"Expected {count} strategies, got {len(context.strategies_list)}"
)
@then('strategies should include "{strategy_names}"')
def step_verify_strategies_include(context: Context, strategy_names: str) -> None:
"""Verify specific strategies are in the list."""
names = [s[0] for s in context.strategies_list]
expected_strategies = [s.strip() for s in strategy_names.split(",")]
for strategy in expected_strategies:
assert strategy in names, f"Strategy '{strategy}' not found in {names}"
# =============================================================================
# CLI INTEGRATION STEPS
# =============================================================================
def _run_cli(args: list[str]) -> tuple[int, str, str]:
"""Run the CLI by calling main() directly for coverage tracking."""
import io
from contextlib import redirect_stdout, redirect_stderr
from convert_rdf_to_hf_dataset_unified import main
old_argv = sys.argv
stdout_capture = io.StringIO()
stderr_capture = io.StringIO()
try:
sys.argv = ["convert_rdf_to_hf_dataset_unified.py"] + args
with redirect_stdout(stdout_capture), redirect_stderr(stderr_capture):
try:
exit_code = main()
except SystemExit as e:
exit_code = e.code if e.code is not None else 0
finally:
sys.argv = old_argv
return exit_code, stdout_capture.getvalue(), stderr_capture.getvalue()
@when("I run the CLI with auto strategy")
def step_run_cli_auto(context: Context) -> None:
"""Run CLI with auto strategy."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@when('I run the CLI with strategy "{strategy}"')
def step_run_cli_strategy(context: Context, strategy: str) -> None:
"""Run CLI with specified strategy."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
"--strategy", strategy,
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@when("I run the CLI with chunk size {size:d}")
def step_run_cli_chunk_size(context: Context, size: int) -> None:
"""Run CLI with custom chunk size."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
"--chunk-size", str(size),
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@when('I run the CLI with description "{desc}" and license "{lic}"')
def step_run_cli_metadata(context: Context, desc: str, lic: str) -> None:
"""Run CLI with metadata options."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
"--description", desc,
"--license", lic,
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@when("I run the CLI with all metadata flags")
def step_run_cli_all_metadata(context: Context) -> None:
"""Run CLI with all metadata flags: description, citation, homepage, license."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
"--description", "Test Dataset",
"--citation", "@misc{test}",
"--homepage", "https://example.org",
"--license", "MIT",
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@when("I run the CLI expecting failure")
def step_run_cli_expect_failure(context: Context) -> None:
"""Run CLI expecting it to fail."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@when("I run the CLI with verbose flag")
def step_run_cli_verbose(context: Context) -> None:
"""Run CLI with verbose flag."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
"--verbose",
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@when("I run the CLI with clean-cache flag")
def step_run_cli_clean_cache(context: Context) -> None:
"""Run CLI with clean-cache flag."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
"--clean-cache",
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@when("I run the CLI with train/test split")
def step_run_cli_split(context: Context) -> None:
"""Run CLI with train/test split option."""
args = [
str(context.input_path),
str(context.output_path),
"--format", context.rdf_format,
"--create-splits",
]
context.cli_exit_code, context.cli_stdout, context.cli_stderr = _run_cli(args)
@then("the CLI should exit successfully")
def step_verify_cli_success(context: Context) -> None:
"""Verify CLI exited with code 0."""
assert context.cli_exit_code == 0, (
f"CLI failed with exit code {context.cli_exit_code}.\n"
f"stdout: {context.cli_stdout}\nstderr: {context.cli_stderr}"
)
@then("the CLI should exit with error")
def step_verify_cli_error(context: Context) -> None:
"""Verify CLI exited with non-zero code."""
assert context.cli_exit_code != 0, "Expected CLI to fail but it succeeded"
@then("the output directory should contain a dataset")
def step_verify_output_has_dataset(context: Context) -> None:
"""Verify output directory has dataset files."""
assert context.output_path.exists(), f"Output path doesn't exist: {context.output_path}"
files = list(context.output_path.iterdir())
assert len(files) > 0, "Output directory is empty"