Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 23 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
# Dedomena

Consistent research data access for agents and scientists across biology and finance.
OpenAlex, Europe PMC, EPO, SEC EDGAR, FRED/ALFRED, ECB and World Bank share a
OpenAlex, Europe PMC, EPO, SEC EDGAR, FRED/ALFRED, ECB, World Bank and Hyperliquid share a
streaming page and provenance contract. Queries retain their native source semantics.

## Install
Expand Down Expand Up @@ -38,7 +38,7 @@ source identifiers, exact request provenance, SHA-256 hashes, a saved-response
snapshot ID, retrieval times, completeness, cost and transfer size. Partial
enumeration fails explicitly. Native records retain source-specific fields.

Default caches persist for 24 hours at `~/.cache/dedomena/sources.sqlite3`.
Research and traditional finance caches persist for 24 hours at `~/.cache/dedomena/sources.sqlite3`.
`refresh=True` fetches new data while keeping previous snapshots for offline replay.
`DEDOMENA_SOURCE_STORE` selects a shared store, including across Canary workers.

Expand Down Expand Up @@ -73,13 +73,34 @@ FRED supports ALFRED knowledge dates; SEC retains amendments; ECB exposes
revisions; World Bank serves current revised indicators.
See [finance access, throughput and semantics](docs/FINANCE.md).

~~~python
from dedomena.sources import Hyperliquid

with Hyperliquid() as source: # Public market data, no key or wallet
markets = source.markets() # Native metadata and asset contexts
book = source.order_book("BTC")
consume(book.records, book.provenance.to_dict())
~~~

Hyperliquid adds mids, spot/perpetual metadata, books, candles and funding history.
Market snapshots default to a zero cache TTL. Candle retention is explicitly
incomplete; funding supports timestamp checkpoints.
See [Hyperliquid retrieval and weight](docs/HYPERLIQUID.md).

`IPPool` routes the same source clients through owned local IPs or proxies, with
shared weighted per-IP admission, key budgets and cooldowns. Reuse one pool for
Hyperliquid and OpenAlex; extra IPs do not multiply OpenAlex's key allowance.
See [shared IP routing examples](docs/IP_ROUTING.md).

## Agent CLI

~~~sh
python -m dedomena.sources benchmark openalex 'CRISPR'
python -m dedomena.sources benchmark europepmc 'TITLE:CRISPR'
python -m dedomena.sources search epo 'ta="CRISPR"' --max-pages 1
python -m dedomena.sources quota openalex
python -m dedomena.sources markets hyperliquid
python -m dedomena.sources book hyperliquid BTC
python -m dedomena.sources fetch sec 320193
python -m dedomena.sources benchmark fred 53 --operation release
python -m dedomena.sources observations fred GDP --as-of 2020-01-01
Expand Down
6 changes: 4 additions & 2 deletions dedomena/sources/__init__.py
Original file line number Diff line number Diff line change
@@ -1,15 +1,17 @@
"""Agent-ready research sources; streaming pages share one provenance contract."""
from .core import (BudgetExceeded, InvalidResponse, Page, Provenance, SearchLimitExceeded,
SourceError, Store, Throttled)
from .egress import IPPool, IPRoute
from .openalex import OpenAlex
from .europepmc import EuropePMC
from .epo import EPO
from .sec import SEC
from .fred import FRED
from .ecb import ECB
from .worldbank import WorldBank
from .hyperliquid import Hyperliquid

__all__ = [
"OpenAlex", "EuropePMC", "EPO", "SEC", "FRED", "ECB", "WorldBank", "Page", "Provenance", "Store",
"SourceError", "BudgetExceeded", "Throttled", "InvalidResponse", "SearchLimitExceeded",
"OpenAlex", "EuropePMC", "EPO", "SEC", "FRED", "ECB", "WorldBank", "Hyperliquid", "Page", "Provenance", "Store",
"IPPool", "IPRoute", "SourceError", "BudgetExceeded", "Throttled", "InvalidResponse", "SearchLimitExceeded",
]
74 changes: 67 additions & 7 deletions dedomena/sources/__main__.py
Original file line number Diff line number Diff line change
@@ -1,22 +1,48 @@
"""Agent-friendly JSON CLI. Credentials are read exclusively from the environment."""
import argparse
from contextlib import ExitStack
from datetime import date
import json
import os
from pathlib import Path
import sys
import time

from . import ECB, EPO, FRED, SEC, EuropePMC, OpenAlex, SourceError, Store, WorldBank
from . import (ECB, EPO, FRED, SEC, EuropePMC, Hyperliquid, IPPool, IPRoute,
OpenAlex, SourceError, Store, WorldBank)


def _ip_pool(path):
try:
with Path(path).expanduser().open("rb") as handle:
data = handle.read(65537)
if len(data) > 65536:
raise ValueError
routes = json.loads(data)
except (OSError, ValueError, UnicodeDecodeError):
raise ValueError("IP routes require a readable JSON file of at most 64 KiB") from None
if (not isinstance(routes, list) or not 1 <= len(routes) <= 1000
or any(not isinstance(route, dict) or "public_ip" not in route
or not set(route).issubset({"public_ip", "local_address", "proxy"})
for route in routes)):
raise ValueError("IP routes require public_ip and one local_address or proxy per entry")
return IPPool(IPRoute(**route) for route in routes)


def main(argv=None):
parser = argparse.ArgumentParser(description=__doc__)
operations = ("search", "observations", "release", "catalogue")
market_operations = ("markets", "mids", "book", "candles", "funding")
operations = ("search", "observations", "release", "catalogue", *market_operations)
parser.add_argument("action", choices=(*operations, "fetch", "quota", "usage", "replay", "benchmark"))
parser.add_argument("source", choices=("openalex", "europepmc", "epo", "sec", "fred", "ecb", "worldbank"))
parser.add_argument("source", choices=("openalex", "europepmc", "epo", "sec", "fred", "ecb", "worldbank", "hyperliquid"))
parser.add_argument("query", nargs="?")
parser.add_argument("--store", help="Shared SQLite snapshot/allowance store")
parser.add_argument("--ip-routes", help="JSON file of owned public IP routes and local bindings/proxies")
parser.add_argument("--start-time", type=int, help="Hyperliquid start time in epoch milliseconds")
parser.add_argument("--end-time", type=int, help="Hyperliquid end time in epoch milliseconds")
parser.add_argument("--interval", help="Hyperliquid candle interval, e.g. 1h")
parser.add_argument("--spot", action="store_true", help="Hyperliquid spot market metadata")
parser.add_argument("--dex", help="Hyperliquid perpetual DEX name")
parser.add_argument("--profile", choices=("discovery", "evidence"), default="discovery")
parser.add_argument("--filter")
parser.add_argument("--refresh", action="store_true")
Expand Down Expand Up @@ -45,24 +71,54 @@ def main(argv=None):
parser.error("SDMX period/delta/last-n options require ECB search or observations")
if args.action != "benchmark" and args.operation != "search":
parser.error("--operation selects the benchmark operation only")
if operation in market_operations and args.source != "hyperliquid":
parser.error("market operations require hyperliquid")
if args.source == "hyperliquid" and operation in ("search", "observations", "release"):
parser.error("hyperliquid supports markets, mids, book, candles, funding, and fetch")
if (args.start_time is not None or args.end_time is not None or args.interval) and (
args.source != "hyperliquid" or operation not in ("candles", "funding")):
parser.error("epoch time and interval options require Hyperliquid candles or funding")
if args.interval and operation != "candles":
parser.error("--interval requires candles")
if operation in ("candles", "funding") and args.start_time is None:
parser.error("historical market operations require --start-time")
if operation == "candles" and (args.end_time is None or not args.interval):
parser.error("candles require --end-time and --interval")
if args.spot and (args.source != "hyperliquid" or operation not in ("markets", "catalogue")):
parser.error("--spot requires Hyperliquid markets or catalogue")
if args.dex is not None and (args.source != "hyperliquid" or operation not in ("markets", "catalogue", "mids")):
parser.error("--dex requires Hyperliquid markets or mids")
if operation == "release" and args.source != "fred":
parser.error("release requires fred")
if operation == "catalogue" and args.source not in ("ecb", "worldbank"):
parser.error("catalogue requires ecb or worldbank")
if operation == "catalogue" and args.source not in ("ecb", "worldbank", "hyperliquid"):
parser.error("catalogue requires ecb, worldbank or hyperliquid")
if operation == "observations" and args.source not in ("fred", "ecb", "worldbank"):
parser.error("observations requires fred, ecb, or worldbank")
if operation in ("fetch", "replay", "release", "observations") and not args.query:
if operation in ("fetch", "replay", "release", "observations", "book", "candles", "funding") and not args.query:
parser.error("this action needs an identifier")
if operation == "search" and not args.query and not (args.source == "openalex" and args.filter):
parser.error("search needs a query, or an OpenAlex filter")
if bool(args.start_date) != bool(args.end_date) or (args.start_date and args.source != "epo"):
parser.error("date partitions require both dates and the epo source")
store = None
resources = ExitStack()

def emit(value):
print(json.dumps(value, ensure_ascii=False, allow_nan=False))

def pages(source):
if args.source == "hyperliquid":
if operation in ("markets", "catalogue"):
return iter((source.markets(spot=args.spot, dex=args.dex or "", refresh=args.refresh),))
if operation == "mids":
return iter((source.all_mids(dex=args.dex or "", refresh=args.refresh),))
if operation == "book":
return iter((source.order_book(args.query, refresh=args.refresh),))
if operation == "candles":
return iter((source.candles(args.query, args.interval, start_time=args.start_time,
end_time=args.end_time, refresh=args.refresh),))
return source.funding_history(args.query, start_time=args.start_time,
end_time=args.end_time, refresh=args.refresh)
if operation == "release":
if not args.query.isascii() or not args.query.isdecimal():
raise ValueError("release identifier must be a positive integer")
Expand Down Expand Up @@ -105,8 +161,11 @@ def pages(source):
return 0
store = Store(args.store) if args.store else None
constructors = {"openalex": OpenAlex, "europepmc": EuropePMC, "epo": EPO,
"sec": SEC, "fred": FRED, "ecb": ECB, "worldbank": WorldBank}
"sec": SEC, "fred": FRED, "ecb": ECB, "worldbank": WorldBank,
"hyperliquid": Hyperliquid}
kwargs = {"store": store} if store else {}
if args.ip_routes:
kwargs["ip_pool"] = resources.enter_context(_ip_pool(args.ip_routes))
if args.source == "openalex":
kwargs["profile"] = args.profile
elif args.source == "europepmc":
Expand Down Expand Up @@ -154,6 +213,7 @@ def pages(source):
print(json.dumps(emit_error), file=sys.stderr)
return 2
finally:
resources.close()
if store:
store.close()
return 0
Expand Down
Loading
Loading