This guide demonstrates cross-language interoperability and best practices for working with BCSV files across different APIs.
All BCSV APIs produce identical binary format. Files written by any API can be read by any other API without conversion.
┌─────────┐
│ C++ │────┐
├─────────┤ │
│ C API │────┼────► .bcsv file ◄────┬────┤ Python │
├─────────┤ │ (binary) │ ├─────────┤
│ Python │────┘ └────┤ C# │
├─────────┤ └─────────┘
│ C# │
└─────────┘
Write Read
Data acquisition in C++:
// sensor_recorder.cpp
#include <bcsv/bcsv.h>
int main() {
bcsv::Layout layout;
layout.addColumn({"timestamp", bcsv::ColumnType::DOUBLE});
layout.addColumn({"temperature", bcsv::ColumnType::FLOAT});
layout.addColumn({"humidity", bcsv::ColumnType::FLOAT});
layout.addColumn({"sensor_id", bcsv::ColumnType::STRING});
bcsv::Writer<bcsv::Layout> writer(layout);
writer.open("sensor_data.bcsv", true, 6);
// Record 1000 samples
for (int i = 0; i < 1000; i++) {
writer.row().set(0, getCurrentTime());
writer.row().set(1, readTemperature());
writer.row().set(2, readHumidity());
writer.row().set(3, std::string("SENSOR_01"));
writer.writeRow();
}
writer.close();
return 0;
}Analysis in Python:
# analyze_sensors.py
import pybcsv
import pandas as pd
import matplotlib.pyplot as plt
# Read C++-generated file
df = pybcsv.read_dataframe("sensor_data.bcsv")
# Pandas analysis
print(f"Temperature range: {df['temperature'].min():.1f} to {df['temperature'].max():.1f}°C")
print(f"Average humidity: {df['humidity'].mean():.1f}%")
# Plot
df.plot(x='timestamp', y=['temperature', 'humidity'], subplots=True)
plt.show()ML model output in Python:
# train_model.py
import pybcsv
import pandas as pd
# Train model and generate predictions
predictions = model.predict(test_data)
# Save predictions to BCSV
df = pd.DataFrame({
'player_id': test_data['player_id'],
'predicted_score': predictions,
'confidence': confidence_scores,
'recommendation': recommendations
})
pybcsv.write_dataframe(df, "ml_predictions.bcsv", compression_level=6)
print(f"Saved {len(df)} predictions to ml_predictions.bcsv")Load in Unity game:
// PredictionLoader.cs
using UnityEngine;
using BCSV;
public class PredictionLoader : MonoBehaviour
{
void LoadPredictions()
{
using (var reader = new BCSVReader())
{
if (!reader.Open("StreamingAssets/ml_predictions.bcsv"))
{
Debug.LogError("Failed to load predictions");
return;
}
while (reader.ReadNext())
{
int playerId = reader.GetInt32(0);
float score = reader.GetFloat(1);
float confidence = reader.GetFloat(2);
string recommendation = reader.GetString(3);
ApplyPrediction(playerId, score, confidence, recommendation);
}
Debug.Log("Predictions loaded successfully");
}
}
}Data logger in C:
// embedded_logger.c
#include <bcsv/bcsv_c_api.h>
void log_system_metrics(void) {
bcsv_layout_t layout = bcsv_layout_create();
bcsv_layout_add_column(layout, 0, "timestamp", BCSV_TYPE_UINT64);
bcsv_layout_add_column(layout, 1, "cpu_usage", BCSV_TYPE_FLOAT);
bcsv_layout_add_column(layout, 2, "memory_mb", BCSV_TYPE_UINT32);
bcsv_layout_add_column(layout, 3, "error_code", BCSV_TYPE_INT32);
bcsv_writer_t writer = bcsv_writer_create(layout);
bcsv_writer_open(writer, "system_metrics.bcsv", true, 0, 0, BCSV_FLAG_NONE);
for (int i = 0; i < 10000; i++) {
bcsv_row_t row = bcsv_writer_row(writer);
bcsv_row_set_uint64(row, 0, get_timestamp_us());
bcsv_row_set_float(row, 1, get_cpu_usage());
bcsv_row_set_uint32(row, 2, get_memory_usage_mb());
bcsv_row_set_int32(row, 3, get_last_error());
bcsv_writer_next(writer);
}
bcsv_writer_close(writer); // complete the file (footer) before teardown
bcsv_writer_destroy(writer);
bcsv_layout_destroy(layout);
}Analysis tool in C++:
// metric_analyzer.cpp
#include <bcsv/bcsv.h>
int main() {
bcsv::Reader<bcsv::Layout> reader;
if (!reader.open("system_metrics.bcsv")) {
std::cerr << "Error: " << reader.getErrorMsg() << "\n";
return 1;
}
double maxCpu = 0.0;
uint32_t maxMemory = 0;
int errorCount = 0;
while (reader.readNext()) {
auto cpu = reader.row().get<float>(1);
auto memory = reader.row().get<uint32_t>(2);
auto error = reader.row().get<int32_t>(3);
maxCpu = std::max(maxCpu, static_cast<double>(cpu));
maxMemory = std::max(maxMemory, memory);
if (error != 0) errorCount++;
}
std::cout << "Max CPU: " << maxCpu << "%\n";
std::cout << "Max Memory: " << maxMemory << " MB\n";
std::cout << "Error count: " << errorCount << "\n";
return 0;
}All numeric types are bit-compatible across languages:
| BCSV Type | C++ | C | Python | C# | Size | Endianness |
|---|---|---|---|---|---|---|
| BOOL | bool | uint8_t | bool | bool | 1 byte | N/A |
| UINT8 | uint8_t | uint8_t | int | byte | 1 byte | N/A |
| UINT16 | uint16_t | uint16_t | int | ushort | 2 bytes | Little |
| UINT32 | uint32_t | uint32_t | int | uint | 4 bytes | Little |
| UINT64 | uint64_t | uint64_t | int | ulong | 8 bytes | Little |
| INT8 | int8_t | int8_t | int | sbyte | 1 byte | N/A |
| INT16 | int16_t | int16_t | int | short | 2 bytes | Little |
| INT32 | int32_t | int32_t | int | int | 4 bytes | Little |
| INT64 | int64_t | int64_t | int | long | 8 bytes | Little |
| FLOAT | float | float | float | float | 4 bytes | IEEE 754 |
| DOUBLE | double | double | float | double | 8 bytes | IEEE 754 |
All platforms use little-endian and IEEE 754 floating point.
The binary format is bit-exact for every IEEE-754 pattern: NaN
(including payload bits), ±Inf, signed zero (-0.0), and subnormals
round-trip unchanged through all row codecs (flat, ZoH, delta) on both the
flexible and static APIs. ZoH/Delta change detection compares bit patterns,
so repeated NaNs compress via ZoH hold, and -0.0 vs +0.0 counts as a
change (the sign bit is never lost).
Caveats at the boundaries:
- CSV bridge (
csv2bcsv/bcsv2csv,CsvWriter/CsvReader): special values are preserved (nan,inf,-inf,-0), NaN payload bits are not (text has no representation for them). - pandas (
pybcsv.write_dataframe): floatNaNis preserved by default (nan_policy="preserve"). pandas cannot distinguishNonefromNaNin float columns — both are written asNaN. Non-float columns cannot represent missing values and are coerced with a warning (or usenan_policy="raise"). - Parquet (
pybcsv.parquet_utils): NaN values pass through. Parquet nulls are a distinct concept;parquet_to_bcsv(null_policy=...)decides what happens to them —"reject"(the default) aborts naming the column and row;"nan"fills float columns (float16/32/64, i.e. C++floatanddouble) with NaN and still aborts on every other type;"zero"fills every column with the BCSV default (0/False/""), which is what an unset BCSV cell already holds. Both filling policies lose information:"nan"only where a column also holds genuine NaNs (the two collapse andbcsv_to_parquetcannot separate them),"zero"unconditionally — a filled zero is indistinguishable from a measured zero. - Parquet file-level key/value metadata has no home in the BCSV header, so
the transcoders carry it in a
<output>.meta.jsoncompanion file — written byparquet_to_bcsv(metadata2json=True), read back bybcsv_to_parquet(json2metadata=True), both on by default and both optional, and readable from C#/Unity viaBcsvMetadata.ReadCompanion. The document records the BCSV file's SHA-256 and is refused if it does not match, so provenance cannot attach to data it does not describe. Both readers hand back only thekey_value_metadataobject — the Parquet footer pairs — and not the document level that surrounds it (source_path,source_sha256,bcsv_sha256,bcsv_bytes,bcsv_rows,metadata_json_version). The two levels share a namespace:source_sha256at the document level is the digest of the Parquet input, so a chain that already recorded asource_sha256in the Parquet footer ends up with two different digests under one name at two levels, describing two different links. Parse the JSON yourself for the document level. An in-format channel is planned for 1.6.0, after which the companion andBcsvMetadataare retired. - BCSV has no null type:
NaNis a value, not a missing-data marker.
Strings are stored as UTF-8 encoded byte sequences with length prefix:
- C++:
std::string(UTF-8) - C:
char*(null-terminated UTF-8) - Python:
str(automatically UTF-8) - C#:
string(converted to/from UTF-8)
Cross-language string example:
// C++: Write emoji and Unicode
writer.row().set(0, std::string("Hello 世界 🌍"));# Python: Read Unicode perfectly
text = reader.row()[0]
print(text) # "Hello 世界 🌍"// C#: Read Unicode perfectly
string text = reader.GetString(0);
Debug.Log(text); // "Hello 世界 🌍"All APIs require identical layouts (same column count, types, and order):
✅ Compatible:
# Python writer
layout.add_column("id", ColumnType.INT32)
layout.add_column("name", ColumnType.STRING)// C++ reader
bcsv::Layout layout;
layout.addColumn({"id", bcsv::ColumnType::INT32});
layout.addColumn({"name", bcsv::ColumnType::STRING});❌ Incompatible:
# Python writer (INT32, STRING)
layout.add_column("id", ColumnType.INT32)
layout.add_column("name", ColumnType.STRING)// C++ reader (INT32, INT32) - Type mismatch!
bcsv::Layout layout;
layout.addColumn({"id", bcsv::ColumnType::INT32});
layout.addColumn({"age", bcsv::ColumnType::INT32}); // ❌ Wrong typeAll APIs validate layout compatibility at open():
// C++
if (!reader.open("data.bcsv")) {
std::cerr << reader.getErrorMsg() << "\n";
// "Column type mismatch at index 1. Expected STRING, got INT32"
}# Python
try:
reader.open("data.bcsv")
except RuntimeError as e:
print(e) # "Column type mismatch..."All APIs support identical LZ4 compression levels (0-9):
// C++: Write with compression level 6
writer.open("data.bcsv", true, 6);# Python: Read compressed file (automatic)
reader.open("data.bcsv") # Decompression automatic// C#: Read compressed file (automatic)
reader.Open("data.bcsv"); // Decompression automaticCompression levels (default: 6):
level is not a smooth dial — it selects between two LZ4 compressors, and which
one depends on the file codec:
| level | packet_lz4_batch (default codec) |
packet_lz4, stream_lz4 |
|---|---|---|
| 0 | no compression | no compression |
| 1-5 | LZ4_compress_fast, acceleration 10 - level |
LZ4_compress_fast, acceleration 10 - level |
| 6-9 | LZ4HC, hc level level + 3 |
still LZ4_compress_fast (per-row blocks are too small for HC) |
So on the default codec the step from 5 to 6 is a change of compressor, not an increment. Measured on wide sensor recordings (650-1052 columns): levels 1-5 land within 4% of each other, level 6 is ~27% smaller for ~50-70% more write CPU, and levels 7-9 add well under 1% for substantially more CPU again.
The level is written into the file header but a reader only tests level > 0 to
decide whether the payload is compressed. LZ4HC emits ordinary LZ4 blocks, so
files written at any level are readable by every 1.5.x reader — the level is a
writer-side choice with no wire-format consequence.
Document your schema in a central location:
# sensor_data.bcsv Schema
| Column | Type | Description |
|--------|------|-------------|
| timestamp | DOUBLE | Unix timestamp (seconds) |
| temperature | FLOAT | Temperature in Celsius |
| humidity | FLOAT | Relative humidity (0-100%) |
| sensor_id | STRING | Sensor identifier |Before reading, inspect the file header to understand the schema:
C++:
bcsv::Reader<bcsv::Layout> reader;
reader.open("unknown.bcsv");
const auto& layout = reader.layout();
for (size_t i = 0; i < layout.columnCount(); i++) {
std::cout << layout.columnName(i) << ": "
<< toString(layout.columnType(i)) << "\n";
}Python:
reader = pybcsv.Reader()
reader.open("unknown.bcsv")
layout = reader.get_layout()
for i in range(layout.column_count()):
print(f"{layout.column_name(i)}: {layout.column_type(i)}")BCSV format version is stored in header. Check compatibility:
if (reader.fileVersion() != bcsv::VERSION) {
std::cerr << "Warning: File version mismatch\n";
// Decide whether to proceed
}Establish naming conventions across teams:
- snake_case: Python-style (recommended for cross-language)
- camelCase: C#/JavaScript-style
- PascalCase: C#-style
# Recommended: snake_case (works everywhere)
layout.add_column("player_id", ColumnType.INT32)
layout.add_column("score_value", ColumnType.FLOAT)
layout.add_column("player_name", ColumnType.STRING)Add metadata in a companion file or comments:
# Write metadata file
metadata = {
"timestamp": "Unix seconds since epoch",
"temperature": "Celsius, typical range -40 to 85",
"humidity": "Percentage, 0-100",
"sensor_id": "Format: SENSOR_XX where XX is 01-99"
}
with open("sensor_data.meta.json", "w") as f:
json.dump(metadata, f)┌──────────────┐
│ C++ Sensor │ Collect data (high-speed)
│ Recorder │ Write: sensor_raw.bcsv
└──────┬───────┘
│
▼
┌──────────────┐
│ Python ML │ Feature engineering
│ Pipeline │ Train model
└──────┬───────┘ Write: features.bcsv, predictions.bcsv
│
▼
┌──────────────┐
│ Unity Game │ Load predictions
│ Client │ Apply AI behaviors
└──────────────┘
# etl_pipeline.py
import pybcsv
import pandas as pd
import sqlalchemy
# Step 1: Read BCSV (from any source)
df = pybcsv.read_dataframe("data.bcsv")
# Step 2: Transform
df['timestamp'] = pd.to_datetime(df['timestamp'], unit='s')
df['category'] = df['value'].apply(categorize)
# Step 3: Load to database
engine = sqlalchemy.create_engine('postgresql://...')
df.to_sql('sensor_data', engine, if_exists='append')Cause: Writer and reader have different schemas
Solution: Inspect file header and match layout exactly
# Inspect existing file
reader = pybcsv.Reader()
reader.open("data.bcsv")
layout = reader.get_layout()
# Print schema
for i in range(layout.column_count()):
print(f"{i}: {layout.column_name(i)} ({layout.column_type(i)})")Cause: C API requires manual UTF-8 handling
Solution: Ensure UTF-8 encoding in C:
// C: Ensure UTF-8
const char* utf8_string = "Hello 世界";
bcsv_row_set_string(row, 0, utf8_string);Cause: Wrong compression level or API choice
Solution:
- Use compression level 0-3 for high-speed writing
- Use C++ Static API for maximum performance
- Profile your specific use case
For implementers creating new language bindings, see:
- ARCHITECTURE.md - Binary format details
- include/bcsv/file_header.h - Header structure
- tests/ - Reference test cases
✅ All APIs produce identical binary format
✅ Files are 100% cross-compatible
✅ Layout must match exactly
✅ UTF-8 strings work everywhere
✅ Compression transparent to reader
✅ Best practice: Document your schema
For specific API documentation:
- API_OVERVIEW.md - Compare all APIs
- C++ examples/
- Python python/README.md
- C# unity/README.md