Source code for hed.schema.hed_cache

"""Infrastructure for caching HED schema from remote repositories."""

from __future__ import annotations

import functools
import json
import os
import re
import shutil
import time
import urllib
from hashlib import sha1
from pathlib import Path
from shutil import copyfile
from urllib.error import URLError

from semantic_version import Version

from hed.schema.hed_cache_lock import CacheError, CacheLock
from hed.schema.schema_io.schema_util import make_url_request, url_to_file

# From https://semver.org/#is-there-a-suggested-regular-expression-regex-to-check-a-semver-string
HED_VERSION_P1 = r"(?P<major>0|[1-9]\d*)\.(?P<minor>0|[1-9]\d*)\.(?P<patch>0|[1-9]\d*)"
HED_VERSION_P2 = (
    r"(?:-(?P<prerelease>(?:0|[1-9]\d*|\d*[a-zA-Z-][0-9a-zA-Z-]*)" r"(?:\.(?:0|[1-9]\d*|\d*[a-zA-Z-][0-9a-zA-Z-]*))*))?"
)
HED_VERSION_P3 = r"(?:\+(?P<buildmetadata>[0-9a-zA-Z-]+(?:\.[0-9a-zA-Z-]+)*))?"
HED_VERSION = HED_VERSION_P1 + HED_VERSION_P2 + HED_VERSION_P3
# Actual local HED filename re.
HED_VERSION_FINAL = r"^[hH][eE][dD](_([a-z0-9]+)_)?(" + HED_VERSION + r")\.[xX][mM][lL]$"

HED_XML_PREFIX = "HED"
HED_XML_EXTENSION = ".xml"
hedxml_suffix = "/hedxml"  # The suffix for schema and library schema at the given urls
prerelease_suffix = "/prerelease"  # The prerelease schemas at the given URLs

DEFAULT_HED_LIST_VERSIONS_URL = "https://api.github.com/repos/hed-standard/hed-schemas/contents/standard_schema"
LIBRARY_HED_URL = "https://api.github.com/repos/hed-standard/hed-schemas/contents/library_schemas"
LIBRARY_DATA_URL = "https://raw.githubusercontent.com/hed-standard/hed-schemas/main/library_data.json"
DEFAULT_URL_LIST = (DEFAULT_HED_LIST_VERSIONS_URL,)
DEFAULT_LIBRARY_URL_LIST = (LIBRARY_HED_URL,)


DEFAULT_SKIP_FOLDERS = ("deprecated",)

# Short-lived listing cache used by get_available_hed_versions() - deliberately much shorter
# than hed_cache_lock.CACHE_TIME_THRESHOLD (which throttles the much more expensive
# cache_xml_versions() download path).
AVAILABLE_VERSIONS_CACHE_FILENAME = "available_versions_cache.json"
AVAILABLE_VERSIONS_TIME_THRESHOLD = 60

HED_CACHE_DIRECTORY = os.path.join(Path.home(), ".hedtools/hed_cache/")

# This is the schemas included in the hedtools package.
INSTALLED_CACHE_LOCATION = os.path.realpath(os.path.join(os.path.dirname(__file__), "schema_data/"))
version_pattern = re.compile(HED_VERSION_FINAL)


[docs] def set_cache_directory(new_cache_dir): """Set default global HED cache directory. Parameters: new_cache_dir (str): Directory to check for versions. """ if new_cache_dir: global HED_CACHE_DIRECTORY HED_CACHE_DIRECTORY = new_cache_dir os.makedirs(new_cache_dir, exist_ok=True)
[docs] def get_cache_directory(cache_folder=None) -> str: """Return the current value of HED_CACHE_DIRECTORY. Parameters: cache_folder (str): Optional cache folder override. Returns: str: The cache directory path. """ if cache_folder: return cache_folder return HED_CACHE_DIRECTORY
[docs] def get_hed_versions(local_hed_directory=None, library_name=None, check_prerelease=False) -> list | dict: """Get the HED versions in the HED directory. Parameters: local_hed_directory (str): Directory to check for versions which defaults to hed_cache. library_name (str or None): An optional schema library name. None retrieves the standard schema only. Pass "all" to retrieve all standard and library schemas as a dict. check_prerelease (bool): If True, results can include prerelease schemas. Default is False, returning only released versions. Returns: Union[list, dict]: List of version numbers or dictionary {library_name: [versions]}. """ if not local_hed_directory: local_hed_directory = HED_CACHE_DIRECTORY if not library_name: library_name = None all_hed_versions = {} local_directories = [local_hed_directory] if check_prerelease and Path(local_hed_directory).name != "prerelease": local_directories.append(os.path.join(local_hed_directory, "prerelease")) hed_files = [] for hed_dir in local_directories: try: hed_files += os.listdir(hed_dir) except FileNotFoundError: pass if not any(version_pattern.match(f) for f in hed_files): cache_local_versions(local_hed_directory) hed_files = [] for hed_dir in local_directories: try: hed_files += os.listdir(hed_dir) except FileNotFoundError: pass for hed_file in hed_files: expression_match = version_pattern.match(hed_file) if expression_match is not None: version = expression_match.group(3) found_library_name = expression_match.group(2) if library_name != "all" and found_library_name != library_name: continue if found_library_name not in all_hed_versions: all_hed_versions[found_library_name] = [] all_hed_versions[found_library_name].append(version) for name, hed_versions in all_hed_versions.items(): all_hed_versions[name] = _sort_version_list(hed_versions) if library_name == "all": return all_hed_versions if library_name in all_hed_versions: return all_hed_versions[library_name] return []
[docs] def get_hed_version_path(xml_version, library_name=None, local_hed_directory=None) -> str | None: """Get the HED XML file path for a given version. Searches the local cache first (including the bundled schemas that are always present). If the version is not found and local_hed_directory is the default HED cache, downloads only the single requested file from GitHub — never the entire catalog. No network call is made for custom directories. Parameters: xml_version (str): The version string to look up. library_name (str or None): Optional schema library name. local_hed_directory (str or None): Path to local HED directory. Defaults to HED_CACHE_DIRECTORY. Passing a custom path disables the automatic GitHub download. Returns: Union[str, None]: The path to the requested HED XML file, or None. """ if not local_hed_directory: local_hed_directory = HED_CACHE_DIRECTORY result = _find_hed_version_path(xml_version, library_name, local_hed_directory) if result: return result # Version not found locally — download only this specific version from GitHub. # Never bulk-download the entire catalog; that is cache_xml_versions()'s job. if not xml_version or local_hed_directory != HED_CACHE_DIRECTORY: return None _download_schema_version(xml_version, library_name, local_hed_directory) return _find_hed_version_path(xml_version, library_name, local_hed_directory)
def _find_hed_version_path(xml_version, library_name, local_hed_directory): """Look up a HED version path in the given directory without downloading. Parameters: xml_version (str): The version to find. library_name (str or None): Optional schema library name. local_hed_directory (str): Directory to search. Returns: Union[str, None]: The path if found, None otherwise. """ hed_versions = get_hed_versions(local_hed_directory, library_name, check_prerelease=True) if not hed_versions or not xml_version: return None if xml_version in hed_versions: # Check regular directory first regular_path = _create_xml_filename(xml_version, library_name, local_hed_directory, False) if os.path.exists(regular_path): return regular_path # Also check prerelease directory prerelease_path = _create_xml_filename(xml_version, library_name, local_hed_directory, True) if os.path.exists(prerelease_path): return prerelease_path return None
[docs] def cache_local_versions(cache_folder) -> int | None: """Cache all schemas included with the HED installation. Parameters: cache_folder (str): The folder holding the cache. Returns: Union[int, None]: Returns -1 on cache access failure. None otherwise """ if not cache_folder: cache_folder = HED_CACHE_DIRECTORY try: with CacheLock(cache_folder, write_time=False): _copy_installed_folder_to_cache(cache_folder) except CacheError: return -1
[docs] def cache_xml_versions( hed_base_urls=DEFAULT_URL_LIST, hed_library_urls=DEFAULT_LIBRARY_URL_LIST, skip_folders=DEFAULT_SKIP_FOLDERS, cache_folder=None, ) -> float: """Cache all schemas at the given URLs. Parameters: hed_base_urls (str or list): Path or list of paths. These should point to a single folder. hed_library_urls (str or list): Path or list of paths. These should point to folder containing library folders. skip_folders (list): A list of subfolders to skip over when downloading. cache_folder (str): The folder holding the cache. Returns: float: Returns -1 if cache failed for any reason, including having been cached too recently. Returns 0 if it successfully cached this time. Notes: - The Default skip_folders is 'deprecated'. - The HED cache folder defaults to HED_CACHE_DIRECTORY. - The directories on GitHub are of the form: https://api.github.com/repos/hed-standard/hed-schemas/contents/standard_schema """ if not cache_folder: cache_folder = HED_CACHE_DIRECTORY # Always seed the cache with bundled schemas first so the cache is usable even if the # subsequent GitHub download fails (network error, rate limit, etc.). cache_local_versions(cache_folder) try: with CacheLock(cache_folder): if isinstance(hed_base_urls, str): hed_base_urls = [hed_base_urls] if isinstance(hed_library_urls, str): hed_library_urls = [hed_library_urls] all_hed_versions = {} for hed_base_url in hed_base_urls: new_hed_versions = _get_hed_xml_versions_one_library(hed_base_url) _merge_in_versions(all_hed_versions, new_hed_versions) for hed_library_url in hed_library_urls: new_hed_versions = _get_hed_xml_versions_from_url_all_libraries( hed_library_url, skip_folders=skip_folders ) _merge_in_versions(all_hed_versions, new_hed_versions) for library_name, hed_versions in all_hed_versions.items(): for version, version_info in hed_versions.items(): _cache_hed_version(version, library_name, version_info, cache_folder=cache_folder) except (CacheError, ValueError, URLError): return -1 return 0
[docs] def get_available_hed_versions( hed_base_urls=DEFAULT_URL_LIST, hed_library_urls=DEFAULT_LIBRARY_URL_LIST, skip_folders=DEFAULT_SKIP_FOLDERS, library_name=None, check_prerelease=False, cache_folder=None, force_refresh=False, cache_time_threshold=AVAILABLE_VERSIONS_TIME_THRESHOLD, ) -> list | dict: """List HED schema versions available on GitHub, without downloading or caching their content. For the canonical hed-schemas URLs this reads a single repository-level manifest (schema_versions.json) from the raw/CDN host in one request - see schema_version_manifest - which is not subject to GitHub's REST API rate limit. If that manifest can't be read (a custom/forked URL set, or any fetch/parse failure) the function falls back to crawling GitHub's REST API directory listings. That fallback never fetches a schema file's actual XML content, but listing everything can still add up to a couple dozen small JSON directory-listing requests in one call: 1-2 for the standard schema (plus its prerelease folder), 1 to enumerate the library folders, and 1-2 more per library folder found. That worst case only applies to library_name="all"; passing library_name=None (the default) skips every library-related request entirely, and passing a specific library name skips the standard-schema request and restricts the library side to just that one library's folder. It's the live-from-GitHub counterpart to get_hed_versions() (which only reports what's already bundled with hedtools or previously cached on disk, with zero network calls), and it's still far cheaper than cache_xml_versions() (which makes those same listing calls AND then downloads every version's full content - fine to do once for a version you're about to use, wasteful to do just to show a list of names). The REST fallback caches its own results on disk (see Notes) so that a caller polling it frequently - e.g. a web service handling many requests - doesn't trip GitHub's API rate limits. Callers don't need to implement their own throttling on top of this. Typical usage is: call this to populate something like a version-picker dropdown, then only fetch the one version the user actually selects, via load_schema_version() (which downloads and caches just that version, lazily, the first time it's needed). Parameters: hed_base_urls (str or list): Path or list of paths for the standard schema folder(s). hed_library_urls (str or list): Path or list of paths for folder(s) containing library schema subfolders. skip_folders (list): A list of library subfolders to skip. Default is 'deprecated'. library_name (str or None): None retrieves the standard schema only. Pass "all" to retrieve all standard and library schemas as a dict. Pass a specific library name to retrieve just that library. check_prerelease (bool): If True, results can include prerelease schemas. Default is False, returning only released versions. cache_folder (str or None): Where to read/write the listing cache (see Notes). None uses the default HED cache folder. force_refresh (bool): If True, skip the "don't even check yet" shortcut described in Notes and always at least ask GitHub whether anything changed. Use this when you specifically need a confirmed up-to-date answer - e.g. right after publishing a new release. cache_time_threshold (int): How long, in seconds, a URL (the manifest, or any REST listing URL used by the fallback) that was just checked is reused without even asking GitHub whether it changed. Default is 60 seconds - short enough that new releases show up quickly, long enough that a caller polling this in a tight loop doesn't generate a request per call. Returns: Union[list, dict]: List of version numbers, or {library_name: [versions]} if library_name is "all". Returns an empty list/dict for any URL that couldn't be reached rather than raising - this is meant to degrade gracefully for use in unattended user-facing listings. Examples: Standard schema only (the default) - just a list of version strings, newest first:: >>> get_available_hed_versions() ['8.4.0', '8.3.0', '8.2.0', '8.1.0', '8.0.0'] Everything - standard schema plus every library, as a dict keyed by library name (the standard schema is under the None key):: >>> get_available_hed_versions(library_name="all") {None: ['8.4.0', '8.3.0', '8.2.0'], 'score': ['2.1.0', '1.0.0'], 'lang': ['1.1.0']} Just one library, by name:: >>> get_available_hed_versions(library_name="score") ['2.1.0', '1.0.0'] Including prereleases (adds anything only found in GitHub's "prerelease" folders):: >>> get_available_hed_versions(check_prerelease=True) ['8.5.0', '8.4.0', '8.3.0', '8.2.0', '8.1.0', '8.0.0'] Once the user picks a version from a list like the above, fetch that one version's actual XML content (this is the step that downloads a schema file - the calls above only ever downloaded small directory listings, never a schema itself):: >>> from hed.schema import load_schema_version >>> schema = load_schema_version("8.4.0") Force a fresh listing right after publishing a release, instead of possibly getting a cached result from just before it went live:: >>> get_available_hed_versions(force_refresh=True) ['8.5.0', '8.4.0', '8.3.0', ...] Notes: - The manifest fast path is used only for the canonical hed-schemas URLs (the defaults for hed_base_urls, hed_library_urls, and skip_folders). Any other URL set, or any failure reading or parsing the manifest, transparently falls through to the REST crawl described below, so behavior is never worse than before. - The REST fallback caches per GitHub URL (there are several under the hood: the standard schema folder and its prerelease folder, the library-folder listing, and each library's own folder and prerelease folder), in a small metadata file (available_versions_cache.json) inside the cache folder, in two layers: 1. If a given URL was checked within cache_time_threshold seconds (default 60), it's reused with no network call at all. 2. Otherwise, a conditional GET is made using the ETag from the last time that URL was fetched. A 304 response means GitHub confirms nothing changed there, so the prior result is reused - and, per GitHub's own documented behavior, this doesn't count against the primary rate limit for authenticated requests. A 200 means something changed, and the new content and ETag are stored for next time. - A URL that failed (GitHub unreachable, rate-limited, etc.) is also remembered for cache_time_threshold seconds, so a string of calls during an outage doesn't retry it on every single one - it raises immediately instead, same as if it were just fetched and failed. - This uses a much shorter threshold than the one in hed_cache_lock.py, which throttles cache_xml_versions()'s far more expensive per-version download step. - Unlike cache_xml_versions(), this never writes schema content - the on-disk cache used here holds only the same small directory-listing JSON GitHub itself returns (version names, SHAs, and download URLs), never a schema file itself. It has no interaction with get_hed_versions(), cache_local_versions(), or the schema files cache_xml_versions() downloads. - force_refresh=True skips layer 1 above but still uses layer 2 (the conditional GET), so it stays cheap when nothing has actually changed. """ if isinstance(hed_base_urls, str): hed_base_urls = [hed_base_urls] if isinstance(hed_library_urls, str): hed_library_urls = [hed_library_urls] if not cache_folder: cache_folder = HED_CACHE_DIRECTORY url_cache = _read_available_versions_cache(cache_folder) cache_before = json.dumps(url_cache, sort_keys=True) # Fast path: read the repo-level manifest in a single fetch from the raw/CDN host (not subject # to the GitHub REST API rate limit) instead of crawling the API directory listings. Only used # for the canonical hed-schemas URLs; any custom/forked URL set falls through to the crawl. Any # failure (unreachable, malformed, or an unrecognized manifest format) also falls through. if ( list(hed_base_urls) == list(DEFAULT_URL_LIST) and list(hed_library_urls) == list(DEFAULT_LIBRARY_URL_LIST) and tuple(skip_folders) == tuple(DEFAULT_SKIP_FOLDERS) ): from hed.schema import schema_version_manifest as _manifest try: manifest_json = _get_json_with_etag(_manifest.MANIFEST_URL, url_cache, force_refresh, cache_time_threshold) if _manifest.is_supported(manifest_json): if json.dumps(url_cache, sort_keys=True) != cache_before: _write_available_versions_cache(cache_folder, url_cache) return _manifest.available_versions(manifest_json, library_name, check_prerelease) except Exception: pass # fall through to the REST API crawl below # Only fetch the standard-schema URLs when the result could actually include them # (library_name is None or "all") - a request for one specific library has no use for # this data, and fetching it anyway would be pure wasted requests. needs_standard = library_name is None or library_name == "all" # Likewise, only touch any library URL when the result could include library data at # all - a plain standard-schema request (library_name=None) never looks at it. needs_libraries = library_name is not None all_hed_versions = {} if needs_standard: for hed_base_url in hed_base_urls: try: new_hed_versions = _get_hed_xml_versions_one_library( hed_base_url, url_cache, force_refresh, cache_time_threshold ) _merge_in_versions(all_hed_versions, new_hed_versions) except Exception: # GitHub unreachable, or an unexpected/malformed response, for this # particular URL - skip it so the caller still gets whatever else could be # listed. Deliberately broad: this function's whole contract is to degrade # gracefully rather than raise, so this isn't limited to network-level # errors (URLError/HTTPError) - a rate-limited or otherwise unexpected # response body (e.g. missing an expected JSON key) should be just as # harmless to the caller as a plain connection failure. continue if needs_libraries: # "all" means no filter (list every library); a specific name restricts the # helper to just that one library's folder, instead of listing and fetching # every library found under hed_library_urls. library_filter = None if library_name == "all" else library_name for hed_library_url in hed_library_urls: try: new_hed_versions = _get_hed_xml_versions_from_url_all_libraries( hed_library_url, library_name=library_filter, skip_folders=skip_folders, etag_cache=url_cache, force_refresh=force_refresh, cache_time_threshold=cache_time_threshold, ) if library_filter is not None and new_hed_versions: # When filtered to one library, _get_hed_xml_versions_from_url_all_libraries() # returns that library's {version: (...)} dict directly, unwrapped, rather # than nested under its name - re-nest it here so _merge_in_versions() sees # the same {lib_name: {version: (...)}} shape it gets from the "all" case. new_hed_versions = {library_filter: new_hed_versions} _merge_in_versions(all_hed_versions, new_hed_versions) except Exception: continue # Only rewrite the cache file if something was actually checked over the network (a fresh # fetch or a conditional 304) - if every URL was served from tier 1 above, nothing changed # and there's no reason to touch the file. if json.dumps(url_cache, sort_keys=True) != cache_before: _write_available_versions_cache(cache_folder, url_cache) result = {} for lib_name, versions_info in all_hed_versions.items(): filtered = [ version for version, (_sha, _download_url, prerelease) in versions_info.items() if check_prerelease or not prerelease ] if filtered: result[lib_name] = _sort_version_list(filtered) if library_name == "all": return result if library_name in result: return result[library_name] return []
def _read_available_versions_cache(cache_folder): """Load the on-disk per-URL listing cache used by get_available_hed_versions(). Parameters: cache_folder (str): Folder the listing cache file lives in. Returns: dict: {url: {"etag": str or None, "body": <parsed GitHub JSON>, "timestamp": float}}. Empty dict if the file is absent, corrupt, or otherwise unreadable - this is treated the same as "nothing cached yet" rather than raising. """ cache_filename = os.path.join(cache_folder, AVAILABLE_VERSIONS_CACHE_FILENAME) try: with open(cache_filename) as f: data = json.load(f) except (FileNotFoundError, ValueError, OSError): return {} if not isinstance(data, dict): return {} return data def _write_available_versions_cache(cache_folder, url_cache): """Best-effort write of the per-URL listing cache back to disk. Writes to a temp file and renames it into place so a reader never sees a partial file. Any failure here is silently ignored - this cache is a performance optimization, not a correctness requirement, so it should never turn a successful listing into an error. Parameters: cache_folder (str): Folder to write the listing cache file into. url_cache (dict): {url: {"etag", "body", "timestamp"}} to persist, as produced by _read_available_versions_cache() and updated by _get_json_with_etag(). """ cache_filename = os.path.join(cache_folder, AVAILABLE_VERSIONS_CACHE_FILENAME) tmp_filename = cache_filename + ".tmp" try: os.makedirs(cache_folder, exist_ok=True) with open(tmp_filename, "w") as f: json.dump(url_cache, f) os.replace(tmp_filename, cache_filename) except OSError: pass def _get_json_with_etag(url, etag_cache, force_refresh=False, cache_time_threshold=0): """Fetch a GitHub JSON listing, reusing a recent or confirmed-unchanged result when possible. Three tiers, cheapest first: 1. If this exact URL was checked within cache_time_threshold seconds (and force_refresh is not set), skip the network entirely and reuse the stored body. 2. Otherwise, make a conditional GET using the stored ETag, if any (If-None-Match). A 304 response means GitHub confirms nothing changed - reuse the stored body. Per GitHub's documented behavior, this doesn't count against the primary rate limit for authenticated requests (see https://docs.github.com/en/rest/using-the-rest-api/ best-practices-for-using-the-rest-api#use-conditional-requests-if-appropriate). 3. A 200 response means something changed (or there was no prior ETag) - parse it and store the new body and ETag for next time. Parameters: url (str): The URL to fetch. etag_cache (dict or None): Maps url -> {"etag", "body", "timestamp"}, updated in place. A failed attempt is recorded too, as {"etag": None, "body": None, "timestamp": <now>} - this is what keeps a URL that's currently unreachable (GitHub down, rate-limited) from being retried on every single call; tier 1 re-raises immediately for a recent failure instead of hitting the network again. Pass None to always do a plain, unconditional fetch with none of this caching behavior (used by callers, like cache_xml_versions(), that don't want it). A per-URL entry that isn't a dict (corrupted cache file, older/newer format) is treated as a cache miss rather than raising. force_refresh (bool): Skip tier 1 (the time-based shortcut) even if the cached entry is still within cache_time_threshold. Tier 2's conditional GET is still used, so this remains cheap when nothing has changed. cache_time_threshold (int): How long, in seconds, tier 1 applies for. For a *recorded failure* specifically, this is capped at AVAILABLE_VERSIONS_TIME_THRESHOLD regardless of the value passed in - see the note below. Returns: The parsed JSON body for this URL - freshly fetched, or reused from etag_cache. Raises: urllib.error.URLError: If the URL could not be reached (including a recent, still-fresh failure recorded by an earlier call - see etag_cache above). Note: A successful result and a recorded failure are not equally safe to hold onto for a long cache_time_threshold. A large threshold on a *successful* entry is fine - it's known-good data that GitHub's ETag will still revalidate. But a *failure* entry (body is None) isn't known-good data whose staying unchanged makes it safe to keep reusing - it's the absence of data, typically from a transient rate-limit or network error. Honoring a large threshold for a failure would mean a transient error keeps being silently skipped (see the broad except clauses in _get_hed_xml_versions_one_library() and _get_hed_xml_versions_from_url_all_libraries()) for far longer than the outage that caused it. To prevent that, a recorded failure is always retried within AVAILABLE_VERSIONS_TIME_THRESHOLD seconds, no matter how large a cache_time_threshold was passed in for this call. """ cached_entry = etag_cache.get(url) if etag_cache is not None else None if not isinstance(cached_entry, dict): # A non-dict entry (e.g. a corrupted cache file, or one written by a future/older # format) can't be trusted - treat it exactly like "nothing cached yet" rather than # letting cached_entry.get(...) raise AttributeError below. cached_entry = None if cached_entry and not force_refresh: effective_threshold = cache_time_threshold if cached_entry.get("body") is None: # See the "Note" above: never let a recorded failure be shielded from retry by a # large threshold that was only ever meant to apply to confirmed-good data. effective_threshold = min(cache_time_threshold, AVAILABLE_VERSIONS_TIME_THRESHOLD) age = time.time() - cached_entry.get("timestamp", 0) if age <= effective_threshold: if cached_entry.get("body") is None: raise URLError(f"A recent attempt to reach {url} failed; not retrying yet") return cached_entry["body"] extra_headers = {} if cached_entry and cached_entry.get("etag"): extra_headers["If-None-Match"] = cached_entry["etag"] try: url_request = make_url_request(url, extra_headers=extra_headers or None) except urllib.error.HTTPError as e: if e.code == 304 and cached_entry is not None and cached_entry.get("body") is not None: # GitHub confirmed nothing changed since our last fetch of this URL. cached_entry["timestamp"] = time.time() return cached_entry["body"] if etag_cache is not None: etag_cache[url] = {"etag": None, "body": None, "timestamp": time.time()} raise except URLError: if etag_cache is not None: etag_cache[url] = {"etag": None, "body": None, "timestamp": time.time()} raise url_data = str(url_request.read(), "utf-8") loaded_json = json.loads(url_data) if etag_cache is not None: etag_cache[url] = { "etag": url_request.headers.get("ETag"), "body": loaded_json, "timestamp": time.time(), } return loaded_json
[docs] @functools.lru_cache(maxsize=50) def get_library_data(library_name, cache_folder=None) -> dict: """Retrieve the library data for the given library. The registry is library_data.json on hed-schemas main. For each library (keyed by name, "" for the standard schema) it records the valid hedId range as "id_range" and, where elements have been removed, the ids that must never be assigned again as "retired_ids" - an object keyed by hedId (e.g. "HED_0011644") whose values carry label, section, last_version, removed_in, reason, an optional replacement, and a reference. Sources, in order: 1. The GitHub copy (LIBRARY_DATA_URL), which is authoritative. It is fetched once per process (this function is cached) with the same conditional-request machinery as get_available_hed_versions(): no network call if this URL was checked within AVAILABLE_VERSIONS_TIME_THRESHOLD seconds, otherwise a conditional GET that GitHub answers with 304 when nothing changed. A successful fetch is written to <cache_folder>/library_data/library_data.json for use offline. 2. That cached copy, when the URL is unreachable. 3. The copy packaged with hedtools, when there is no cached copy either. Parameters: library_name (str): The schema name. "" for standard schema. cache_folder (str): The cache folder to use if not using the default. Returns: dict: The data for a specific library. Empty if no source could be read or none of them lists this library. """ if cache_folder is None: cache_folder = HED_CACHE_DIRECTORY cache_lib_data_folder = os.path.join(cache_folder, "library_data") local_library_data_filename = os.path.join(cache_lib_data_folder, "library_data.json") library_data = _fetch_library_data(cache_folder, local_library_data_filename) if library_data is None: library_data = _read_library_data_file(local_library_data_filename) if library_data is None: try: with CacheLock(cache_lib_data_folder, write_time=False): _copy_installed_folder_to_cache(cache_lib_data_folder, "library_data") except (OSError, CacheError): pass library_data = _read_library_data_file(local_library_data_filename) if library_data is None: # The cache folder could not be written to; read the packaged copy where it is. installed_filename = os.path.join(INSTALLED_CACHE_LOCATION, "library_data", "library_data.json") library_data = _read_library_data_file(installed_filename) if library_data is None: return {} specific_library = library_data.get(library_name) if not isinstance(specific_library, dict): return {} return specific_library
def _fetch_library_data(cache_folder, local_library_data_filename): """Fetch library_data.json from hed-schemas main and store it in the cache. Parameters: cache_folder (str): The cache folder holding the per-URL ETag cache (AVAILABLE_VERSIONS_CACHE_FILENAME). local_library_data_filename (str): Where to write the fetched registry. Returns: dict or None: The parsed registry, or None if the URL could not be reached or did not hold a JSON object. Failures are never raised: the caller falls back to the cached and packaged copies. """ url_cache = _read_available_versions_cache(cache_folder) cache_before = json.dumps(url_cache, sort_keys=True) try: library_data = _get_json_with_etag( LIBRARY_DATA_URL, url_cache, cache_time_threshold=AVAILABLE_VERSIONS_TIME_THRESHOLD ) except (OSError, ValueError, URLError): library_data = None if json.dumps(url_cache, sort_keys=True) != cache_before: _write_available_versions_cache(cache_folder, url_cache) if not isinstance(library_data, dict): return None _write_library_data_file(local_library_data_filename, library_data) return library_data def _read_library_data_file(filename): """Read a library_data.json copy. Parameters: filename (str): The file to read. Returns: dict or None: The parsed registry, or None if the file is absent, unreadable, or not a JSON object. """ try: with open(filename, encoding="utf-8") as file: library_data = json.load(file) except (OSError, ValueError): return None if not isinstance(library_data, dict): return None return library_data def _write_library_data_file(filename, library_data): """Best-effort write of the registry to the cache, via a temp file and rename. Any failure is ignored: the cached copy is only the offline fallback, so an unwritable cache folder must not turn a successful fetch into an error. Parameters: filename (str): The cache file to write. library_data (dict): The registry to store. """ tmp_filename = filename + ".tmp" try: os.makedirs(os.path.dirname(filename), exist_ok=True) with open(tmp_filename, "w", encoding="utf-8") as file: json.dump(library_data, file, indent=2) os.replace(tmp_filename, filename) except OSError: pass def _copy_installed_folder_to_cache(cache_folder, sub_folder=""): """Copies the schemas from the install folder to the cache""" source_folder = INSTALLED_CACHE_LOCATION if sub_folder: source_folder = os.path.join(INSTALLED_CACHE_LOCATION, sub_folder) installed_files = os.listdir(source_folder) for install_name in installed_files: _, basename = os.path.split(install_name) cache_name = os.path.join(cache_folder, basename) install_name = os.path.join(source_folder, basename) if not os.path.isdir(install_name) and not os.path.exists(cache_name): shutil.copy(install_name, cache_name) def _check_if_url(hed_xml_or_url): """Returns true if this is a url""" if hed_xml_or_url.startswith("http://") or hed_xml_or_url.startswith("https://"): return True return False def _create_xml_filename(hed_xml_version, library_name=None, hed_directory=None, prerelease=False): """Returns the default file name format for the given version""" prerelease_prefix = "prerelease/" if prerelease else "" if library_name: hed_xml_basename = f"{prerelease_prefix}{HED_XML_PREFIX}_{library_name}_{hed_xml_version}{HED_XML_EXTENSION}" else: hed_xml_basename = prerelease_prefix + HED_XML_PREFIX + hed_xml_version + HED_XML_EXTENSION if hed_directory: hed_xml_filename = os.path.join(hed_directory, hed_xml_basename) return hed_xml_filename return hed_xml_basename def _sort_version_list(hed_versions): return sorted(hed_versions, key=Version, reverse=True) def _get_hed_xml_versions_one_folder(hed_folder_url, etag_cache=None, force_refresh=False, cache_time_threshold=0): loaded_json = _get_json_with_etag(hed_folder_url, etag_cache, force_refresh, cache_time_threshold) all_hed_versions = {} for file_entry in loaded_json: if file_entry["type"] == "dir": continue expression_match = version_pattern.match(file_entry["name"]) if expression_match is not None: version = expression_match.group(3) found_library_name = expression_match.group(2) if found_library_name not in all_hed_versions: all_hed_versions[found_library_name] = {} all_hed_versions[found_library_name][version] = ( file_entry["sha"], file_entry["download_url"], hed_folder_url.endswith(prerelease_suffix), ) return all_hed_versions def _get_hed_xml_versions_one_library( hed_one_library_url, etag_cache=None, force_refresh=False, cache_time_threshold=0 ): all_hed_versions = {} try: finalized_versions = _get_hed_xml_versions_one_folder( hed_one_library_url + hedxml_suffix, etag_cache, force_refresh, cache_time_threshold ) _merge_in_versions(all_hed_versions, finalized_versions) except Exception: # Silently ignore ones without a hedxml section for now. Deliberately broad (not just # URLError) - a rate-limited or otherwise unexpected GitHub response can fail with a # KeyError/ValueError while parsing rather than a network-level error, and this # function's contract is to degrade gracefully either way. pass try: pre_release_folder_versions = _get_hed_xml_versions_one_folder( hed_one_library_url + prerelease_suffix, etag_cache, force_refresh, cache_time_threshold ) _merge_in_versions(all_hed_versions, pre_release_folder_versions) except Exception: # Silently ignore ones without a prerelease section for now. See note above. pass ordered_versions = {} for hed_library_name, hed_versions in all_hed_versions.items(): ordered_versions1 = _sort_version_list(hed_versions) ordered_versions2 = [(version, hed_versions[version]) for version in ordered_versions1] ordered_versions[hed_library_name] = dict(ordered_versions2) return ordered_versions def _get_hed_xml_versions_from_url_all_libraries( hed_base_library_url, library_name=None, skip_folders=DEFAULT_SKIP_FOLDERS, etag_cache=None, force_refresh=False, cache_time_threshold=0, ) -> list | dict: """Get all available schemas and their hash values Parameters: hed_base_library_url(str): A single GitHub API url to cache, which contains library schema folders The subfolders should be a schema folder containing hedxml and/or prerelease folders. library_name(str or None): If str, cache only the named library schemas. skip_folders (list): A list of sub folders to skip over when downloading. etag_cache (dict or None): Passed through to _get_json_with_etag() for every request this makes, including one per discovered library subfolder. None (the default) disables conditional/cached requests. force_refresh (bool): Passed through to _get_json_with_etag(). cache_time_threshold (int): Passed through to _get_json_with_etag(). Returns: Union[list, dict]: List of version numbers or dictionary {library_name: [versions]}. Notes: - The Default skip_folders is 'deprecated'. - The HED cache folder defaults to HED_CACHE_DIRECTORY. - The directories on GitHub are of the form: https://api.github.com/repos/hed-standard/hed-schemas/contents/standard_schema/hedxml """ loaded_json = _get_json_with_etag(hed_base_library_url, etag_cache, force_refresh, cache_time_threshold) all_hed_versions = {} for file_entry in loaded_json: if file_entry["type"] == "dir": if file_entry["name"] in skip_folders: continue found_library_name = file_entry["name"] if library_name is not None and found_library_name != library_name: continue single_library_versions = _get_hed_xml_versions_one_library( hed_base_library_url + "/" + found_library_name, etag_cache, force_refresh, cache_time_threshold ) _merge_in_versions(all_hed_versions, single_library_versions) continue if library_name in all_hed_versions: return all_hed_versions[library_name] return all_hed_versions def _merge_in_versions(all_hed_versions, sub_folder_versions): """Build up the version dictionary, divided by library""" for lib_name, _hed_versions in sub_folder_versions.items(): if lib_name not in all_hed_versions: all_hed_versions[lib_name] = {} all_hed_versions[lib_name].update(sub_folder_versions[lib_name]) def _calculate_sha1(filename): """Calculate sha1 hash for filename Can be compared to GitHub hash values """ try: with open(filename, "rb") as f: data = f.read() githash = sha1() githash.update(f"blob {len(data)}\0".encode()) githash.update(data) return githash.hexdigest() except FileNotFoundError: return None def _safe_move_tmp_to_folder(temp_hed_xml_file, dest_filename): """Copy to destination folder and rename. Parameters: temp_hed_xml_file (str): An XML file, generally from a temp folder. dest_filename (str): A destination folder and filename. Returns: str: The new filename on success or None on failure. """ _, temp_xml_file = os.path.split(temp_hed_xml_file) dest_folder, _ = os.path.split(dest_filename) temp_filename_in_cache = os.path.join(dest_folder, temp_xml_file) copyfile(temp_hed_xml_file, temp_filename_in_cache) try: os.replace(temp_filename_in_cache, dest_filename) except OSError: os.remove(temp_filename_in_cache) return None return dest_filename def _download_schema_version(xml_version, library_name, cache_folder): """Download a single specific schema version from GitHub. Fetches the directory listing for only the relevant library (2-3 small API calls), then downloads only the one requested XML file. Uses SHA comparison so a file whose content has not changed on GitHub is never re-downloaded. This is the on-demand complement to cache_xml_versions() (which bulk-downloads every version). Call this when you need one specific version that is not in the local cache; call cache_xml_versions() only when you want to pre-populate the entire catalog (e.g. for an offline/air-gapped environment). Parameters: xml_version (str): The version string to download. library_name (str or None): The library name, None for the standard schema. cache_folder (str): The folder to save the downloaded file in. Returns: str or None: The local file path on success, None if the version was not found on GitHub or the download failed. """ # Fast path: resolve this version from the manifest (single raw/CDN fetch, no REST API), then # download just the one file. Falls back to the REST API resolution below on any failure, # including a malformed manifest that passes is_supported() but raises inside find_version_info # or _cache_hed_version (unexpected types, missing keys, etc.). from hed.schema import schema_version_manifest as _manifest try: manifest_json = _get_json_with_etag(_manifest.MANIFEST_URL, None) if _manifest.is_supported(manifest_json): version_info = _manifest.find_version_info(manifest_json, xml_version, library_name) if version_info is not None: return _cache_hed_version(xml_version, library_name, version_info, cache_folder) except Exception: pass # fall through to REST API resolution below if library_name is None: url = DEFAULT_HED_LIST_VERSIONS_URL else: url = f"{LIBRARY_HED_URL}/{library_name}" try: available = _get_hed_xml_versions_one_library(url) except Exception: return None versions_for_library = available.get(library_name) if not versions_for_library: return None version_info = versions_for_library.get(xml_version) if version_info is None: return None return _cache_hed_version(xml_version, library_name, version_info, cache_folder) def _cache_hed_version(version, library_name, version_info, cache_folder): """Cache the given HED version""" sha_hash, download_url, prerelease = version_info possible_cache_filename = _create_xml_filename(version, library_name, cache_folder, prerelease) local_sha_hash = _calculate_sha1(possible_cache_filename) if sha_hash == local_sha_hash: return possible_cache_filename return _cache_specific_url(download_url, possible_cache_filename) def _cache_specific_url(source_url, cache_filename): """Copies a specific url to the cache at the given filename""" cache_folder = cache_filename.rpartition("/")[0] os.makedirs(cache_folder, exist_ok=True) temp_filename = url_to_file(source_url) if temp_filename: cache_filename = _safe_move_tmp_to_folder(temp_filename, cache_filename) os.remove(temp_filename) return cache_filename return None