From f0cf3af5117e70dbe7fdf739996dbed0eba50a43 Mon Sep 17 00:00:00 2001 From: Anant Jain Date: Sun, 20 Sep 2026 15:42:11 +0530 Subject: [PATCH 1/8] feat(api): link-only dataset import from Hugging Face, GitHub and Kaggle Let a publisher add a dataset that already lives on a third-party platform by giving its identifier (or page URL). Only metadata is fetched; files are never copied or listed, and downloads redirect to the platform. - previewPlatformDataset query: fetch title, description, license, tags, author and last-updated from the platform, with no side effects. - importPlatformDataset mutation: create a DRAFT dataset prefilled from the platform, attach sectors/geographies whose names match the platform's tags, fill matching dataset metadata fields, add one EXTERNAL resource linking to the dataset page, record provenance, and grant the owner role, all in one transaction. Optional title override. Duplicate imports within the same organisation/user are rejected. - DatasetSource model (one-to-one with Dataset) for provenance; exposed as TypeDataset.source. TypeResource.url is now exposed. - Importers for Hugging Face, GitHub and Kaggle behind a registry; all work without API keys for public datasets. HF_TOKEN, GITHUB_TOKEN and KAGGLE_USERNAME/KAGGLE_KEY are optional. - Download view redirects EXTERNAL resources to their URL and returns 404 instead of raising when a resource has no file. - Dataset search document gains source_platform; /api/search/dataset/ returns it, aggregates on it and filters by it (NATIVE = not imported). formats indexing skips resources without file details. After deploy, run `manage.py search_index --rebuild` once so Elasticsearch maps source_platform as a keyword before the first import is indexed. Refs CivicDataLab/DataSpace#174, #190, #191, #192 --- .env.example | 6 + DataSpace/settings.py | 10 + api/migrations/0048_platform_import.py | 75 ++++++ api/models/Dataset.py | 26 +- api/models/DatasetSource.py | 46 ++++ api/models/__init__.py | 1 + api/schema/platform_import_schema.py | 99 +++++++ api/schema/schema.py | 3 + api/services/platform_import_service.py | 241 ++++++++++++++++++ api/services/platform_importers/__init__.py | 64 +++++ api/services/platform_importers/base.py | 196 ++++++++++++++ api/services/platform_importers/github.py | 144 +++++++++++ .../platform_importers/huggingface.py | 165 ++++++++++++ api/services/platform_importers/kaggle.py | 121 +++++++++ api/types/type_dataset.py | 7 + api/types/type_dataset_source.py | 62 +++++ api/types/type_resource.py | 1 + api/utils/enums.py | 8 + api/views/download_view.py | 43 ++-- api/views/search_dataset.py | 10 + api/views/search_unified.py | 1 + search/documents/dataset_document.py | 8 + 22 files changed, 1310 insertions(+), 27 deletions(-) create mode 100644 api/migrations/0048_platform_import.py create mode 100644 api/models/DatasetSource.py create mode 100644 api/schema/platform_import_schema.py create mode 100644 api/services/platform_import_service.py create mode 100644 api/services/platform_importers/__init__.py create mode 100644 api/services/platform_importers/base.py create mode 100644 api/services/platform_importers/github.py create mode 100644 api/services/platform_importers/huggingface.py create mode 100644 api/services/platform_importers/kaggle.py create mode 100644 api/types/type_dataset_source.py diff --git a/.env.example b/.env.example index 58fefe7e..f1a77c70 100644 --- a/.env.example +++ b/.env.example @@ -12,3 +12,9 @@ URL_WHITELIST=http://localhost:8000,http://localhost,http://localhost:3000 DEBUG=True SECRET_KEY=your-secret-key REDIS_URL=redis://redis:6379/1 + +# Third-party platform imports (optional) +KAGGLE_USERNAME= +KAGGLE_KEY= +HF_TOKEN= +GITHUB_TOKEN= diff --git a/DataSpace/settings.py b/DataSpace/settings.py index 4d25acca..f3fcd32c 100644 --- a/DataSpace/settings.py +++ b/DataSpace/settings.py @@ -289,6 +289,16 @@ } +# Third-party platform imports (link-only). All three platforms work with no +# key for public datasets. KAGGLE_* adds a file count to Kaggle imports, +# HF_TOKEN unlocks gated Hugging Face repos, GITHUB_TOKEN lifts GitHub's +# anonymous rate limit. +KAGGLE_USERNAME = os.getenv("KAGGLE_USERNAME", None) +KAGGLE_KEY = os.getenv("KAGGLE_KEY", None) +HF_TOKEN = os.getenv("HF_TOKEN", None) +GITHUB_TOKEN = os.getenv("GITHUB_TOKEN", None) # optional, lifts the 60 req/hour anonymous limit +PLATFORM_IMPORT_TIMEOUT = float(os.getenv("PLATFORM_IMPORT_TIMEOUT", "15")) + # DVC settings DVC_REPO_PATH = os.path.join(BASE_DIR, "dvc") DVC_REMOTE_NAME = os.getenv("DVC_REMOTE_NAME", None) diff --git a/api/migrations/0048_platform_import.py b/api/migrations/0048_platform_import.py new file mode 100644 index 00000000..ac2e5107 --- /dev/null +++ b/api/migrations/0048_platform_import.py @@ -0,0 +1,75 @@ +# Generated by Django 5.0.4 on 2026-09-20 10:13 + +import uuid + +import django.db.models.deletion +from django.conf import settings +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ("api", "0047_resourcetype_publication_collaborative_publications_and_more"), + migrations.swappable_dependency(settings.AUTH_USER_MODEL), + ] + + operations = [ + migrations.CreateModel( + name="DatasetSource", + fields=[ + ( + "id", + models.UUIDField( + default=uuid.uuid4, editable=False, primary_key=True, serialize=False + ), + ), + ( + "platform", + models.CharField( + choices=[ + ("KAGGLE", "Kaggle"), + ("HUGGINGFACE", "Huggingface"), + ("GITHUB", "Github"), + ], + max_length=50, + ), + ), + ("source_identifier", models.CharField(max_length=300)), + ("source_url", models.URLField(max_length=500)), + ("source_author", models.CharField(blank=True, max_length=300)), + ("source_license", models.CharField(blank=True, max_length=300)), + ("source_last_updated", models.DateTimeField(blank=True, null=True)), + ("raw_metadata", models.JSONField(blank=True, default=dict)), + ("imported_at", models.DateTimeField(auto_now_add=True)), + ("last_synced_at", models.DateTimeField(auto_now=True)), + ( + "dataset", + models.OneToOneField( + on_delete=django.db.models.deletion.CASCADE, + related_name="source", + to="api.dataset", + ), + ), + ( + "imported_by", + models.ForeignKey( + blank=True, + null=True, + on_delete=django.db.models.deletion.SET_NULL, + related_name="imported_dataset_sources", + to=settings.AUTH_USER_MODEL, + ), + ), + ], + options={ + "db_table": "dataset_source", + "indexes": [ + models.Index( + fields=["platform", "source_identifier"], + name="dataset_sou_platfor_38ca23_idx", + ) + ], + }, + ), + ] diff --git a/api/models/Dataset.py b/api/models/Dataset.py index a2b99a1a..296f509c 100644 --- a/api/models/Dataset.py +++ b/api/models/Dataset.py @@ -1,5 +1,5 @@ import uuid -from typing import TYPE_CHECKING, Any +from typing import TYPE_CHECKING, Any, Optional from django.db import models from django.db.models import Sum @@ -124,14 +124,22 @@ def formats_indexing(self) -> list[str]: Used in Elasticsearch indexing. """ - return list( - set( - [ - resource.resourcefiledetails.format # type: ignore - for resource in self.resources.all() - ] - ).difference({""}) - ) + formats: set[str] = set() + for resource in self.resources.all(): + # Link-only (EXTERNAL) resources have no file details; skip them. + file_details = getattr(resource, "resourcefiledetails", None) + if file_details is not None and file_details.format: + formats.add(file_details.format) + return list(formats) + + @property + def source_platform_indexing(self) -> Optional[str]: + """Platform this dataset was imported from, or None for native datasets. + + Used in Elasticsearch indexing. + """ + source = getattr(self, "source", None) + return source.platform if source is not None else None @property def catalogs_indexing(self) -> list[str]: diff --git a/api/models/DatasetSource.py b/api/models/DatasetSource.py new file mode 100644 index 00000000..e9aa7817 --- /dev/null +++ b/api/models/DatasetSource.py @@ -0,0 +1,46 @@ +import uuid + +from django.db import models + +from api.utils.enums import ImportPlatform + + +class DatasetSource(models.Model): + """Provenance record for a dataset imported from a third-party platform. + + Imports are link-only: DataSpace never copies the platform's files. This + row remembers where the dataset came from so the UI can attribute it, link + back to it, and (later) re-sync its metadata. The raw platform response is + kept in ``raw_metadata`` for debugging and future field mapping. + """ + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + dataset = models.OneToOneField("api.Dataset", on_delete=models.CASCADE, related_name="source") + platform = models.CharField(max_length=50, choices=ImportPlatform.choices) + # Platform-native identifier, e.g. "owner/dataset-slug" (Kaggle) or + # "namespace/name" (Hugging Face). Normalised by the importer. + source_identifier = models.CharField(max_length=300) + # Human-facing page on the platform. + source_url = models.URLField(max_length=500) + source_author = models.CharField(max_length=300, blank=True) + # License string exactly as the platform reported it (may not map onto + # DatasetLicense; the mapped value lives on Dataset.license). + source_license = models.CharField(max_length=300, blank=True) + source_last_updated = models.DateTimeField(null=True, blank=True) + raw_metadata = models.JSONField(default=dict, blank=True) + imported_by = models.ForeignKey( + "authorization.User", + on_delete=models.SET_NULL, + null=True, + blank=True, + related_name="imported_dataset_sources", + ) + imported_at = models.DateTimeField(auto_now_add=True) + last_synced_at = models.DateTimeField(auto_now=True) + + class Meta: + db_table = "dataset_source" + indexes = [models.Index(fields=["platform", "source_identifier"])] + + def __str__(self) -> str: + return f"{self.platform}:{self.source_identifier}" diff --git a/api/models/__init__.py b/api/models/__init__.py index 6c54b34c..19f7c515 100644 --- a/api/models/__init__.py +++ b/api/models/__init__.py @@ -9,6 +9,7 @@ ) from api.models.Dataset import Dataset, Tag from api.models.DatasetMetadata import DatasetMetadata +from api.models.DatasetSource import DatasetSource from api.models.DataSpace import DataSpace from api.models.Geography import Geography from api.models.Metadata import Metadata diff --git a/api/schema/platform_import_schema.py b/api/schema/platform_import_schema.py new file mode 100644 index 00000000..12de66ad --- /dev/null +++ b/api/schema/platform_import_schema.py @@ -0,0 +1,99 @@ +"""GraphQL surface for link-only imports from third-party platforms. + +- ``preview_platform_dataset`` fetches normalised metadata, no side effects. +- ``import_platform_dataset`` creates a DRAFT dataset with one EXTERNAL resource + linking to the dataset page on the platform (files are not imported). + +Both take the same organization/dataspace request headers as ``add_dataset``; +the imported dataset is owned the same way a manually created one would be. +""" + +from typing import Optional + +import strawberry +from strawberry.types import Info + +from api.schema.base_mutation import ( + BaseMutation, + GraphQLValidationError, + MutationResponse, +) +from api.services.platform_import_service import ( + import_platform_dataset, + preview_platform_dataset, +) +from api.services.platform_importers import PlatformImportError +from api.types.type_dataset import TypeDataset +from api.types.type_dataset_source import ( + TypePlatformDatasetPreview, + import_platform_enum, +) +from api.utils.graphql_telemetry import trace_resolver +from authorization.graphql_permissions import IsAuthenticated +from authorization.permissions import CreateDatasetPermission + + +@strawberry.input +class ImportPlatformDatasetInput: + platform: import_platform_enum # type: ignore + #: Short id ("owner/name") or a pasted platform URL. + identifier: str + #: Optional display title on DataSpace; defaults to the platform's title. + title: Optional[str] = None + + +@strawberry.type +class Query: + @strawberry.field(permission_classes=[IsAuthenticated]) + @trace_resolver(name="preview_platform_dataset", attributes={"component": "platform_import"}) + def preview_platform_dataset( + self, info: Info, platform: import_platform_enum, identifier: str # type: ignore + ) -> TypePlatformDatasetPreview: + """Look up a Hugging Face / GitHub / Kaggle dataset and show what an import would create.""" + try: + data = preview_platform_dataset(platform.value, identifier) + except PlatformImportError as exc: + # Surface the importer's user-safe message as a GraphQL error. + raise ValueError(exc.message) from exc + return TypePlatformDatasetPreview.from_info(data) + + +@strawberry.type +class Mutation: + @strawberry.mutation + @BaseMutation.mutation( + permission_classes=[IsAuthenticated, CreateDatasetPermission], + trace_name="import_platform_dataset", + trace_attributes={"component": "platform_import"}, + track_activity={ + "verb": "imported", + "get_data": lambda result, import_input=None, **kwargs: { + "dataset_id": str(result.id), + "dataset_title": result.title, + "platform": import_input.platform.value if import_input else None, + "identifier": import_input.identifier if import_input else None, + "organization": (str(result.organization.id) if result.organization else None), + }, + }, + ) + def import_platform_dataset( + self, info: Info, import_input: ImportPlatformDatasetInput + ) -> MutationResponse[TypeDataset]: + """Create a DRAFT dataset that links to the dataset on the platform.""" + organization = info.context.context.get("organization") + dataspace = info.context.context.get("dataspace") + user = info.context.user + + try: + dataset = import_platform_dataset( + platform=import_input.platform.value, + identifier=import_input.identifier, + user=user, + organization=organization, + dataspace=dataspace, + title=import_input.title, + ) + except PlatformImportError as exc: + return MutationResponse.error_response(GraphQLValidationError.from_message(exc.message)) + + return MutationResponse.success_response(TypeDataset.from_django(dataset)) diff --git a/api/schema/schema.py b/api/schema/schema.py index 4678b63d..5f919cfc 100644 --- a/api/schema/schema.py +++ b/api/schema/schema.py @@ -16,6 +16,7 @@ import api.schema.metadata_schema import api.schema.organization_data_schema import api.schema.organization_schema +import api.schema.platform_import_schema import api.schema.publication_schema import api.schema.resource_chart_schema import api.schema.resource_schema @@ -77,6 +78,7 @@ def tags(self, info: Info) -> List[TypeTag]: api.schema.user_schema.Query, api.schema.collaborative_schema.Query, api.schema.publication_schema.Query, + api.schema.platform_import_schema.Query, AuthQuery, ), ) @@ -100,6 +102,7 @@ def tags(self, info: Info) -> List[TypeTag]: api.schema.tags_schema.Mutation, api.schema.collaborative_schema.Mutation, api.schema.publication_schema.Mutation, + api.schema.platform_import_schema.Mutation, AuthMutation, ), ) diff --git a/api/services/platform_import_service.py b/api/services/platform_import_service.py new file mode 100644 index 00000000..b30fe17b --- /dev/null +++ b/api/services/platform_import_service.py @@ -0,0 +1,241 @@ +"""Turn a third-party platform dataset into a DataSpace Dataset (link-only). + +Creates the Dataset (DRAFT), a single EXTERNAL Resource that links to the +dataset's page on the platform, a DatasetSource provenance row, tags, taxonomy +and metadata prefill, and the creator's owner permission — all inside one +transaction. Files are never listed or copied: people reach them through the +original dataset link. +""" + +from __future__ import annotations + +from typing import Iterable, Optional + +import structlog +from django.db import transaction +from django.utils.text import slugify + +from api.models import ( + Dataset, + DatasetMetadata, + DatasetSource, + DataSpace, + Geography, + Metadata, + Organization, + Resource, + Sector, + Tag, +) +from api.services.platform_importers import ( + PlatformDatasetInfo, + PlatformImportError, + get_importer, +) +from api.utils.enums import ( + DatasetAccessType, + DatasetStatus, + DatasetType, + DataType, + MetadataModels, +) +from authorization.models import DatasetPermission, Role, User + +logger = structlog.get_logger("dataspace.platform_import") + + +class DuplicateImportError(PlatformImportError): + """The same platform dataset was already imported into this publisher scope.""" + + def __init__(self, existing: Dataset) -> None: + super().__init__( + f"This dataset was already imported as '{existing.title}' ({existing.slug})" + ) + self.existing = existing + + +def preview_platform_dataset(platform: str, identifier: str) -> PlatformDatasetInfo: + """Fetch normalised metadata without creating anything.""" + importer = get_importer(platform) + return importer.fetch_dataset_info(identifier) + + +def find_existing_import( + platform: str, + identifier: str, + organization: Optional[Organization], + user: Optional[User], +) -> Optional[Dataset]: + """Return a dataset already imported from this source in the same scope. + + Scope is the organization when importing on its behalf, otherwise the + individual user — mirroring how datasets are owned elsewhere. + """ + qs = DatasetSource.objects.select_related("dataset").filter( + platform=platform, source_identifier=identifier + ) + if organization is not None: + qs = qs.filter(dataset__organization=organization) + else: + qs = qs.filter(dataset__organization__isnull=True, dataset__user=user) + source = qs.first() + return source.dataset if source else None + + +def _unique_dataset_slug(title: str) -> str: + """Pick a free slug up front. Dataset.save() retries on a slug collision, + but only a handful of times; platform titles ("Iris", "Wine Reviews") + repeat across organisations far more often than timestamped native ones, + so choose a free slug before saving rather than rely on the retry.""" + base = slugify(title)[:240] or "imported-dataset" + slug, counter = base, 2 + while Dataset.objects.filter(slug=slug).exists(): + slug = f"{base}-{counter}" + counter += 1 + return slug + + +PLATFORM_LABELS = {"HUGGINGFACE": "Hugging Face", "KAGGLE": "Kaggle", "GITHUB": "GitHub"} + + +# Platform tags/topics are matched (case-insensitively) against our own +# taxonomies so the publisher lands on the metadata step with sectors and +# geographies already ticked where the names line up. Publishing requires +# sectors, so this is the most valuable prefill we can do without a human. +def _prefill_taxonomies(dataset: Dataset, tags: Iterable[str]) -> None: + names = {t.strip().lower() for t in tags if t and t.strip()} + if not names: + return + sectors = [s for s in Sector.objects.all() if s.name.strip().lower() in names] + if sectors: + dataset.sectors.add(*sectors) + geographies = [g for g in Geography.objects.all() if g.name.strip().lower() in names] + if geographies: + dataset.geographies.add(*geographies) + + +# Optional EAV prefill: if the deployment defines dataset metadata fields whose +# label matches one of these (case-insensitive), fill it from the platform. +# Deployments without such fields are simply skipped. +METADATA_LABEL_SOURCES = { + "source": "source_url", + "source url": "source_url", + "source platform": "platform_label", + "original source": "source_url", + "author": "author", + "creator": "author", + "publisher": "author", + "license": "license", + "original license": "license", + "last updated": "last_updated", + "source last updated": "last_updated", +} + + +def _prefill_metadata(dataset: Dataset, info: PlatformDatasetInfo) -> None: + values = { + "source_url": info.source_url, + "platform_label": PLATFORM_LABELS.get(str(info.platform), str(info.platform).title()), + "author": info.author, + "license": info.license, + "last_updated": info.last_updated.date().isoformat() if info.last_updated else "", + } + fields = Metadata.objects.filter(enabled=True, model=MetadataModels.DATASET) + for field in fields: + source_key = METADATA_LABEL_SOURCES.get((field.label or "").strip().lower()) + value = values.get(source_key or "", "") + if not value: + continue + try: + DatasetMetadata(dataset=dataset, metadata_item=field, value=str(value)[:1000]).save() + except Exception as exc: # validators on the field may reject the value; that's fine + logger.info("platform_import_metadata_skipped", label=field.label, error=str(exc)) + + +@transaction.atomic +def import_platform_dataset( + *, + platform: str, + identifier: str, + user: User, + organization: Optional[Organization] = None, + dataspace: Optional[DataSpace] = None, + info: Optional[PlatformDatasetInfo] = None, + title: Optional[str] = None, +) -> Dataset: + """Create a DRAFT dataset from a platform source. Raises PlatformImportError.""" + importer = get_importer(platform) + canonical_id = importer.parse_identifier(identifier) + + existing = find_existing_import(platform, canonical_id, organization, user) + if existing is not None: + raise DuplicateImportError(existing) + + if info is None: + info = importer.fetch_dataset_info(canonical_id) + + # Publisher may choose the name shown on DataSpace; platform title otherwise. + display_title = (title or "").strip()[:300] or info.title + dataset = Dataset.objects.create( + title=display_title, + slug=_unique_dataset_slug(display_title), + description=info.description, + user=user, + organization=organization, + dataspace=dataspace, + status=DatasetStatus.DRAFT, + access_type=DatasetAccessType.PUBLIC, + license=info.mapped_license, + dataset_type=DatasetType.DATA, + ) + + if info.tags: + tags = [ + Tag.objects.get_or_create(defaults={"value": value}, value__iexact=value)[0] + for value in info.tags + ] + dataset.tags.set(tags) + _prefill_taxonomies(dataset, info.tags) + _prefill_metadata(dataset, info) + + # One resource standing for the whole dataset on the platform. Its URL is + # the dataset page, so "download" redirects there and people browse/fetch + # files with the platform's own tooling. + platform_label = PLATFORM_LABELS.get(str(info.platform), str(info.platform).title()) + Resource.objects.create( + dataset=dataset, + type=DataType.EXTERNAL, + name=f"Dataset on {platform_label}"[:200], + url=info.source_url[:500], + description=f"Files are hosted on {platform_label}. Open the link to browse and download them.", + ) + + DatasetSource.objects.create( + dataset=dataset, + platform=info.platform, + source_identifier=info.identifier, + source_url=info.source_url[:500], + source_author=info.author[:300], + source_license=info.license[:300], + source_last_updated=info.last_updated, + raw_metadata=info.raw, + imported_by=user, + ) + + try: + owner_role = Role.objects.get(name="owner") + except Role.DoesNotExist as exc: + # Same seed the rest of the app relies on (add_dataset does a bare .get()). + raise PlatformImportError( + "Roles are not initialised on this server (run `manage.py init_roles`)" + ) from exc + DatasetPermission.objects.create(user=user, dataset=dataset, role=owner_role) + + logger.info( + "platform_dataset_imported", + platform=info.platform, + identifier=info.identifier, + dataset_id=str(dataset.id), + user_id=str(user.id), + ) + return dataset diff --git a/api/services/platform_importers/__init__.py b/api/services/platform_importers/__init__.py new file mode 100644 index 00000000..246c6b08 --- /dev/null +++ b/api/services/platform_importers/__init__.py @@ -0,0 +1,64 @@ +"""Registry of third-party platform importers. + +Add a platform by subclassing ``PlatformImporter`` and registering it here; +the GraphQL layer and import service never reference a concrete importer. +""" + +from typing import Dict, Optional, Type +from urllib.parse import urlparse + +from api.services.platform_importers.base import ( + InvalidIdentifierError, + PlatformAuthError, + PlatformDatasetInfo, + PlatformDatasetNotFoundError, + PlatformImporter, + PlatformImportError, + PlatformUnavailableError, +) +from api.services.platform_importers.github import GitHubImporter +from api.services.platform_importers.huggingface import HuggingFaceImporter +from api.services.platform_importers.kaggle import KaggleImporter +from api.utils.enums import ImportPlatform + +IMPORTERS: Dict[str, Type[PlatformImporter]] = { + ImportPlatform.KAGGLE: KaggleImporter, + ImportPlatform.HUGGINGFACE: HuggingFaceImporter, + ImportPlatform.GITHUB: GitHubImporter, +} + + +def get_importer(platform: str) -> PlatformImporter: + try: + return IMPORTERS[str(platform)]() + except KeyError as exc: + raise InvalidIdentifierError(f"Unsupported platform: {platform}") from exc + + +def detect_platform(value: str) -> Optional[str]: + """Guess the platform from a pasted URL (None for bare identifiers).""" + raw = (value or "").strip() + if not raw: + return None + host = urlparse(raw if "://" in raw else f"https://{raw}").netloc.lower() + for platform, importer in IMPORTERS.items(): + if host in importer.hosts: + return platform + return None + + +__all__ = [ + "IMPORTERS", + "get_importer", + "detect_platform", + "PlatformImporter", + "PlatformDatasetInfo", + "PlatformImportError", + "InvalidIdentifierError", + "PlatformDatasetNotFoundError", + "PlatformAuthError", + "PlatformUnavailableError", + "HuggingFaceImporter", + "GitHubImporter", + "KaggleImporter", +] diff --git a/api/services/platform_importers/base.py b/api/services/platform_importers/base.py new file mode 100644 index 00000000..bc6a671a --- /dev/null +++ b/api/services/platform_importers/base.py @@ -0,0 +1,196 @@ +"""Shared contract for third-party platform importers. + +An importer turns a user-supplied identifier (short id or pasted URL) into a +normalised ``PlatformDatasetInfo`` by calling the platform's public API. +Importers fetch *metadata only* — title, description, license, tags, author, +last updated and the dataset's page URL. Files are never listed or copied. +""" + +from __future__ import annotations + +from abc import ABC, abstractmethod +from dataclasses import dataclass, field +from datetime import datetime +from typing import Any, Dict, List, Optional + +import requests +import structlog +from django.conf import settings +from django.utils.dateparse import parse_datetime + +from api.utils.enums import DatasetLicense, ImportPlatform + +logger = structlog.get_logger("dataspace.platform_import") + + +# --------------------------------------------------------------------------- # +# Errors +# --------------------------------------------------------------------------- # +class PlatformImportError(Exception): + """Base class for importer failures. ``message`` is safe to show to users.""" + + def __init__(self, message: str) -> None: + super().__init__(message) + self.message = message + + +class InvalidIdentifierError(PlatformImportError): + """The identifier/URL does not look like anything the platform accepts.""" + + +class PlatformDatasetNotFoundError(PlatformImportError): + """The platform reported no dataset for this identifier.""" + + +class PlatformAuthError(PlatformImportError): + """Credentials are missing/invalid, or the dataset is private/gated.""" + + +class PlatformUnavailableError(PlatformImportError): + """Network failure, timeout, rate limit, or a 5xx from the platform.""" + + +# --------------------------------------------------------------------------- # +# Normalised result +# --------------------------------------------------------------------------- # +@dataclass +class PlatformDatasetInfo: + platform: str + identifier: str + title: str + description: str + source_url: str + author: str = "" + license: str = "" + tags: List[str] = field(default_factory=list) + last_updated: Optional[datetime] = None + raw: Dict[str, Any] = field(default_factory=dict) + + @property + def mapped_license(self) -> str: + return map_license(self.license) + + +# --------------------------------------------------------------------------- # +# Helpers shared by importers +# --------------------------------------------------------------------------- # +# Platform license strings (lower-cased) -> DatasetLicense. Anything not listed +# falls back to CC-BY 4.0 and the raw string is kept on DatasetSource. +LICENSE_ALIASES: Dict[str, str] = { + "cc-by-4.0": DatasetLicense.CC_BY_4_0_ATTRIBUTION, + "cc-by": DatasetLicense.CC_BY_4_0_ATTRIBUTION, + "cc by 4.0": DatasetLicense.CC_BY_4_0_ATTRIBUTION, + "attribution 4.0 international (cc by 4.0)": DatasetLicense.CC_BY_4_0_ATTRIBUTION, + "cc-by-sa-4.0": DatasetLicense.CC_BY_SA_4_0_ATTRIBUTION_SHARE_ALIKE, + "cc-by-sa": DatasetLicense.CC_BY_SA_4_0_ATTRIBUTION_SHARE_ALIKE, + "attribution-sharealike 4.0 international (cc by-sa 4.0)": ( + DatasetLicense.CC_BY_SA_4_0_ATTRIBUTION_SHARE_ALIKE + ), + "odc-by": DatasetLicense.OPEN_DATA_COMMONS_BY_ATTRIBUTION, + "odc-by-1.0": DatasetLicense.OPEN_DATA_COMMONS_BY_ATTRIBUTION, + "odc attribution license (odc-by)": DatasetLicense.OPEN_DATA_COMMONS_BY_ATTRIBUTION, + "odbl": DatasetLicense.OPEN_DATABASE_LICENSE, + "odbl-1.0": DatasetLicense.OPEN_DATABASE_LICENSE, + "odc-odbl": DatasetLicense.OPEN_DATABASE_LICENSE, + "database: open database, contents: database contents": DatasetLicense.OPEN_DATABASE_LICENSE, + "database: open database, contents: © original authors": DatasetLicense.OPEN_DATABASE_LICENSE, +} + + +def map_license(raw: str) -> str: + """Map a platform license string onto DatasetLicense (default CC-BY-4.0).""" + key = (raw or "").strip().lower() + return LICENSE_ALIASES.get(key, DatasetLicense.CC_BY_4_0_ATTRIBUTION) + + +def parse_iso_datetime(value: Optional[str]) -> Optional[datetime]: + """Parse a platform timestamp. Django's parser copes with a trailing ``Z`` + and any number of fractional digits (Kaggle sends e.g. ``...:04.7Z``), + which ``datetime.fromisoformat`` on Python 3.10 does not.""" + if not value: + return None + try: + return parse_datetime(value) + except ValueError: + return None + + +# --------------------------------------------------------------------------- # +# Base importer +# --------------------------------------------------------------------------- # +class PlatformImporter(ABC): + """Contract every platform importer implements.""" + + platform: str + #: Human name used in user-facing messages, e.g. "Hugging Face". + label: str = "" + #: Hostnames whose URLs this importer can parse (used by parse_identifier). + hosts: tuple = () + + def __init__(self, session: Optional[requests.Session] = None) -> None: + self.session = session or requests.Session() + self.timeout: float = float(getattr(settings, "PLATFORM_IMPORT_TIMEOUT", 15)) + + @abstractmethod + def parse_identifier(self, value: str) -> str: + """Normalise a short id or pasted URL to the platform's canonical id. + + Raises InvalidIdentifierError when it cannot. + """ + + @abstractmethod + def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: + """Call the platform API and return normalised metadata.""" + + # -- HTTP plumbing ------------------------------------------------------- # + def _get(self, url: str, **kwargs: Any) -> requests.Response: + """GET with uniform error mapping. Subclasses add auth via kwargs.""" + kwargs.setdefault("timeout", self.timeout) + name = self.label or str(self.platform) + try: + response = self.session.get(url, **kwargs) + except requests.Timeout as exc: + raise PlatformUnavailableError(f"{name} timed out") from exc + except requests.RequestException as exc: + logger.warning( + "platform_import_request_failed", platform=self.platform, url=url, error=str(exc) + ) + raise PlatformUnavailableError(f"Could not reach {name}") from exc + + if response.status_code == 404: + raise PlatformDatasetNotFoundError(f"Dataset not found on {name}") + if response.status_code in (401, 403): + raise PlatformAuthError( + f"{name} refused access. The dataset may be private/gated, " + "or the server credentials are missing or invalid." + ) + if response.status_code == 429: + raise PlatformUnavailableError(f"{name} rate limit reached, try again later") + if response.status_code >= 500: + raise PlatformUnavailableError(f"{name} returned an error ({response.status_code})") + if response.status_code >= 400: + raise PlatformImportError(f"{name} rejected the request ({response.status_code})") + return response + + def _get_json(self, url: str, **kwargs: Any) -> Any: + response = self._get(url, **kwargs) + try: + return response.json() + except ValueError as exc: + raise PlatformUnavailableError( + f"{self.label or self.platform} returned an unreadable response" + ) from exc + + +__all__ = [ + "ImportPlatform", + "PlatformImporter", + "PlatformDatasetInfo", + "PlatformImportError", + "InvalidIdentifierError", + "PlatformDatasetNotFoundError", + "PlatformAuthError", + "PlatformUnavailableError", + "map_license", + "parse_iso_datetime", +] diff --git a/api/services/platform_importers/github.py b/api/services/platform_importers/github.py new file mode 100644 index 00000000..c5202e31 --- /dev/null +++ b/api/services/platform_importers/github.py @@ -0,0 +1,144 @@ +"""GitHub repository importer (metadata only). + +A repo (optionally a branch and sub-folder) is treated as a dataset. Two +requests per import: the repo metadata and the README. Public repos need no +token; ``GITHUB_TOKEN`` (settings) is sent when present, mainly to lift the +anonymous 60 requests/hour rate limit. +""" + +from __future__ import annotations + +import re +from typing import Any, Dict, Optional, Tuple +from urllib.parse import urlparse + +from django.conf import settings + +from api.services.platform_importers.base import ( + InvalidIdentifierError, + PlatformDatasetInfo, + PlatformImporter, + PlatformImportError, + parse_iso_datetime, +) +from api.utils.enums import ImportPlatform + +GITHUB_HOSTS = ("github.com", "www.github.com") +GITHUB_API = "https://api.github.com" +GITHUB_WEB = "https://github.com" + +_NAME_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.-]*$") +MAX_DESCRIPTION = 1000 + + +class GitHubImporter(PlatformImporter): + platform = ImportPlatform.GITHUB + label = "GitHub" + hosts = GITHUB_HOSTS + + def _headers(self) -> Dict[str, str]: + headers = {"Accept": "application/vnd.github+json"} + token = getattr(settings, "GITHUB_TOKEN", None) + if token: + headers["Authorization"] = f"Bearer {token}" + return headers + + # -- identifier ---------------------------------------------------------- # + def parse_identifier(self, value: str) -> str: + """Normalise to ``owner/repo`` or ``owner/repo@branch:sub/path``.""" + owner, repo, branch, path = self._parse(value) + ident = f"{owner}/{repo}" + if branch: + ident += f"@{branch}" + if path: + ident += f":{path}" + return ident + + def _parse(self, value: str) -> Tuple[str, str, Optional[str], str]: + raw = (value or "").strip() + if not raw: + raise InvalidIdentifierError("Enter a GitHub repository (owner/repo) or URL") + + branch: Optional[str] = None + path = "" + if "://" in raw or raw.startswith(GITHUB_HOSTS): + parsed = urlparse(raw if "://" in raw else f"https://{raw}") + if parsed.netloc.lower() not in GITHUB_HOSTS: + raise InvalidIdentifierError("That is not a github.com URL") + parts = [p for p in parsed.path.split("/") if p] + if len(parts) < 2: + raise InvalidIdentifierError( + "Expected a repository URL like https://github.com//" + ) + owner, repo = parts[0], parts[1].removesuffix(".git") + # /tree// or /blob// + if len(parts) >= 4 and parts[2] in ("tree", "blob"): + branch = parts[3] + path = "/".join(parts[4:]) + else: + spec = raw.strip("/") + if ":" in spec: + spec, path = spec.split(":", 1) + if "@" in spec: + spec, branch = spec.split("@", 1) + parts = spec.split("/") + if len(parts) != 2: + raise InvalidIdentifierError( + "GitHub repos look like 'owner/repo' (optionally owner/repo@branch:path)" + ) + owner, repo = parts + + if not (_NAME_RE.match(owner) and _NAME_RE.match(repo)): + raise InvalidIdentifierError( + "GitHub owner and repo names use letters, digits, '-', '_' and '.'" + ) + return owner, repo, (branch or None), path.strip("/") + + # -- fetch --------------------------------------------------------------- # + def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: + owner, repo, branch, sub_path = self._parse(identifier) + headers = self._headers() + + meta: Dict[str, Any] = self._get_json(f"{GITHUB_API}/repos/{owner}/{repo}", headers=headers) + if meta.get("disabled"): + raise PlatformImportError("This repository has been disabled on GitHub") + branch = branch or meta.get("default_branch") or "main" + full_name = meta.get("full_name") or f"{owner}/{repo}" + + # Repo names are slugs ("covid-19-data"); make a readable title from them. + title = repo.replace("-", " ").replace("_", " ").strip() + if sub_path: + title = f"{title} / {sub_path}" + + source_url = f"{GITHUB_WEB}/{full_name}" + if sub_path: + source_url += f"/tree/{branch}/{sub_path}" + + description = self._description(full_name, headers) or str(meta.get("description") or "") + license_info = meta.get("license") or {} + raw = dict(meta) + raw["_branch"] = branch + + return PlatformDatasetInfo( + platform=self.platform, + identifier=self.parse_identifier(identifier), + title=title[:300], + description=description[:MAX_DESCRIPTION], + source_url=source_url, + author=str((meta.get("owner") or {}).get("login") or owner)[:300], + license=str(license_info.get("spdx_id") or license_info.get("name") or ""), + tags=[t[:50] for t in (meta.get("topics") or []) if isinstance(t, str)][:30], + last_updated=parse_iso_datetime(meta.get("pushed_at") or meta.get("updated_at")), + raw=raw, + ) + + # -- pieces -------------------------------------------------------------- # + def _description(self, full_name: str, headers: Dict[str, str]) -> str: + try: + response = self._get( + f"{GITHUB_API}/repos/{full_name}/readme", + headers={**headers, "Accept": "application/vnd.github.raw"}, + ) + return response.text.strip() + except PlatformImportError: + return "" diff --git a/api/services/platform_importers/huggingface.py b/api/services/platform_importers/huggingface.py new file mode 100644 index 00000000..17b81c88 --- /dev/null +++ b/api/services/platform_importers/huggingface.py @@ -0,0 +1,165 @@ +"""Hugging Face Hub dataset importer (metadata only). + +Two requests per import: the repo metadata and the README (dataset card). +Public datasets need no token; ``HF_TOKEN`` (settings) is sent when present so +gated/private repos the token can see also work. +""" + +from __future__ import annotations + +import re +from typing import Any, Dict, List +from urllib.parse import urlparse + +from django.conf import settings + +from api.services.platform_importers.base import ( + InvalidIdentifierError, + PlatformAuthError, + PlatformDatasetInfo, + PlatformImporter, + PlatformImportError, + parse_iso_datetime, +) +from api.utils.enums import ImportPlatform + +HF_HOSTS = ("huggingface.co", "www.huggingface.co", "hf.co") +HF_API = "https://huggingface.co/api/datasets" +HF_WEB = "https://huggingface.co/datasets" + +# "namespace/name" or a canonical "name"; HF ids allow letters, digits, - _ . +_ID_RE = re.compile(r"^(?:[A-Za-z0-9][A-Za-z0-9._-]*/)?[A-Za-z0-9][A-Za-z0-9._-]*$") + +MAX_DESCRIPTION = 1000 # Dataset.description column length + + +class HuggingFaceImporter(PlatformImporter): + platform = ImportPlatform.HUGGINGFACE + label = "Hugging Face" + hosts = HF_HOSTS + + def _headers(self) -> Dict[str, str]: + token = getattr(settings, "HF_TOKEN", None) + return {"Authorization": f"Bearer {token}"} if token else {} + + # -- identifier ---------------------------------------------------------- # + def parse_identifier(self, value: str) -> str: + raw = (value or "").strip() + if not raw: + raise InvalidIdentifierError("Enter a Hugging Face dataset id or URL") + + if "://" in raw or raw.startswith(tuple(HF_HOSTS)): + parsed = urlparse(raw if "://" in raw else f"https://{raw}") + if parsed.netloc.lower() not in HF_HOSTS: + raise InvalidIdentifierError("That is not a huggingface.co URL") + parts = [p for p in parsed.path.split("/") if p] + if not parts or parts[0] != "datasets": + raise InvalidIdentifierError( + "Expected a dataset URL like https://huggingface.co/datasets//" + ) + parts = parts[1:] + # Trim sub-paths such as /tree/main, /blob/main/..., /viewer. + if len(parts) >= 2 and parts[1] not in ( + "tree", + "blob", + "viewer", + "resolve", + "discussions", + ): + repo_id = f"{parts[0]}/{parts[1]}" + elif parts: + repo_id = parts[0] + else: + raise InvalidIdentifierError("Could not find a dataset id in that URL") + else: + repo_id = raw[len("datasets/") :] if raw.startswith("datasets/") else raw + + repo_id = repo_id.strip("/") + if not _ID_RE.match(repo_id): + raise InvalidIdentifierError( + "Hugging Face ids look like 'namespace/name' (letters, digits, '-', '_', '.')" + ) + return repo_id + + # -- fetch --------------------------------------------------------------- # + def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: + repo_id = self.parse_identifier(identifier) + headers = self._headers() + + try: + meta: Dict[str, Any] = self._get_json(f"{HF_API}/{repo_id}", headers=headers) + except PlatformAuthError as exc: + # Hugging Face answers 401 for repos that do not exist as well as + # for private/gated ones, so say both. + raise PlatformAuthError( + f"'{repo_id}' was not found on Hugging Face, or it is private/gated. " + "Check the spelling; only public datasets can be imported." + ) from exc + if meta.get("disabled"): + raise PlatformImportError("This dataset has been disabled on Hugging Face") + + card: Dict[str, Any] = meta.get("cardData") or {} + tags: List[str] = meta.get("tags") or [] + + title = card.get("pretty_name") or repo_id.split("/")[-1] + author = meta.get("author") or (repo_id.split("/")[0] if "/" in repo_id else "") + + return PlatformDatasetInfo( + platform=self.platform, + identifier=repo_id, + title=str(title)[:300], + description=self._description(repo_id, meta, headers), + source_url=f"{HF_WEB}/{repo_id}", + author=str(author)[:300], + license=self._license(card, tags), + tags=self._tags(tags), + last_updated=parse_iso_datetime(meta.get("lastModified")), + raw=meta, + ) + + # -- pieces -------------------------------------------------------------- # + @staticmethod + def _license(card: Dict[str, Any], tags: List[str]) -> str: + lic = card.get("license") + if isinstance(lic, list): + lic = lic[0] if lic else "" + if lic: + return str(lic) + for tag in tags: + if tag.startswith("license:"): + return tag[len("license:") :] + return "" + + @staticmethod + def _tags(tags: List[str]) -> List[str]: + """Keep human-meaningful tags: plain ones plus task/language values.""" + out: List[str] = [] + for tag in tags: + if ":" not in tag: + value = tag + else: + prefix, _, value = tag.partition(":") + if prefix not in ("task_categories", "language", "task_ids"): + continue + value = value.strip() + if value and value not in out: + out.append(value[:50]) + return out[:30] + + def _description(self, repo_id: str, meta: Dict[str, Any], headers: Dict[str, str]) -> str: + """Prefer the dataset card body (README) over the terse API field.""" + try: + response = self._get(f"{HF_WEB}/{repo_id}/resolve/main/README.md", headers=headers) + text = response.text + except PlatformImportError: + text = "" + if text: + # Strip YAML front matter. + if text.startswith("---"): + end = text.find("\n---", 3) + if end != -1: + text = text[end + 4 :] + text = text.strip() + if not text: + text = str(meta.get("description") or "") + return text[:MAX_DESCRIPTION] diff --git a/api/services/platform_importers/kaggle.py b/api/services/platform_importers/kaggle.py new file mode 100644 index 00000000..54bbef44 --- /dev/null +++ b/api/services/platform_importers/kaggle.py @@ -0,0 +1,121 @@ +"""Kaggle dataset importer (metadata only). + +One request per import: the dataset *view* endpoint, which answers anonymously +for public datasets and carries everything we store — title, description, +license, tags, owner and last updated. No key is needed. ``KAGGLE_USERNAME`` / +``KAGGLE_KEY`` (settings) are sent as basic auth when configured, which is only +useful for datasets the account can see but the public cannot. +""" + +from __future__ import annotations + +import re +from typing import Any, Dict, List, Optional, Tuple +from urllib.parse import urlparse + +from django.conf import settings + +from api.services.platform_importers.base import ( + InvalidIdentifierError, + PlatformAuthError, + PlatformDatasetInfo, + PlatformImporter, + parse_iso_datetime, +) +from api.utils.enums import ImportPlatform + +KAGGLE_HOSTS = ("www.kaggle.com", "kaggle.com") +KAGGLE_API = "https://www.kaggle.com/api/v1" +KAGGLE_WEB = "https://www.kaggle.com/datasets" + +_SLUG_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_-]*$") +MAX_DESCRIPTION = 1000 + + +class KaggleImporter(PlatformImporter): + platform = ImportPlatform.KAGGLE + label = "Kaggle" + hosts = KAGGLE_HOSTS + + def _auth(self) -> Optional[Tuple[str, str]]: + """Platform-level credentials if configured; None means anonymous.""" + username = getattr(settings, "KAGGLE_USERNAME", None) + key = getattr(settings, "KAGGLE_KEY", None) + return (username, key) if username and key else None + + # -- identifier ---------------------------------------------------------- # + def parse_identifier(self, value: str) -> str: + raw = (value or "").strip() + if not raw: + raise InvalidIdentifierError("Enter a Kaggle dataset ref (owner/dataset) or URL") + + if "://" in raw or raw.startswith(KAGGLE_HOSTS): + parsed = urlparse(raw if "://" in raw else f"https://{raw}") + if parsed.netloc.lower() not in KAGGLE_HOSTS: + raise InvalidIdentifierError("That is not a kaggle.com URL") + parts = [p for p in parsed.path.split("/") if p] + if parts and parts[0] == "datasets": + parts = parts[1:] + if len(parts) < 2: + raise InvalidIdentifierError( + "Expected a dataset URL like https://www.kaggle.com/datasets//" + ) + owner, slug = parts[0], parts[1] + else: + parts = raw.strip("/").split("/") + if len(parts) != 2: + raise InvalidIdentifierError("Kaggle refs look like 'owner/dataset-name'") + owner, slug = parts + + if not (_SLUG_RE.match(owner) and _SLUG_RE.match(slug)): + raise InvalidIdentifierError( + "Kaggle owner and dataset names use letters, digits, '-' and '_'" + ) + return f"{owner}/{slug}" + + # -- fetch --------------------------------------------------------------- # + def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: + ref = self.parse_identifier(identifier) + owner, slug = ref.split("/") + + meta: Dict[str, Any] = self._get_json( + f"{KAGGLE_API}/datasets/view/{owner}/{slug}", auth=self._auth() + ) + if meta.get("isPrivate"): + raise PlatformAuthError("This Kaggle dataset is private") + + title = meta.get("title") or slug + description = (meta.get("description") or meta.get("subtitle") or "").strip() + + return PlatformDatasetInfo( + platform=self.platform, + identifier=ref, + title=str(title)[:300], + description=description[:MAX_DESCRIPTION], + source_url=meta.get("url") or f"{KAGGLE_WEB}/{ref}", + author=str(meta.get("ownerName") or owner)[:300], + license=self._license(meta), + tags=self._tags(meta), + last_updated=parse_iso_datetime(meta.get("lastUpdated")), + raw=meta, + ) + + # -- pieces -------------------------------------------------------------- # + @staticmethod + def _license(meta: Dict[str, Any]) -> str: + licenses = meta.get("licenses") or [] + if licenses and isinstance(licenses[0], dict): + return str(licenses[0].get("name") or "") + return str(meta.get("licenseName") or "") + + @staticmethod + def _tags(meta: Dict[str, Any]) -> List[str]: + out: List[str] = [] + for entry in meta.get("keywords") or []: + if isinstance(entry, str) and entry.strip(): + out.append(entry.strip()[:50]) + for entry in meta.get("tags") or []: + name = entry.get("name") if isinstance(entry, dict) else entry + if isinstance(name, str) and name.strip() and name.strip() not in out: + out.append(name.strip()[:50]) + return out[:30] diff --git a/api/types/type_dataset.py b/api/types/type_dataset.py index b8e71ec3..dcc0ed3e 100644 --- a/api/types/type_dataset.py +++ b/api/types/type_dataset.py @@ -11,6 +11,7 @@ from api.models import Dataset, DatasetMetadata, PromptDataset, Resource, Tag from api.types.base_type import BaseType from api.types.type_dataset_metadata import TypeDatasetMetadata +from api.types.type_dataset_source import TypeDatasetSource from api.types.type_geo import TypeGeo from api.types.type_organization import TypeOrganization from api.types.type_resource import TypeResource @@ -65,6 +66,12 @@ class TypeDataset(BaseType): download_count: int user: Optional["TypeUser"] + @strawberry.field + def source(self) -> Optional["TypeDatasetSource"]: + """Provenance for datasets imported from a third-party platform (else null).""" + source = getattr(self, "source", None) + return TypeDatasetSource.from_django(source) if source is not None else None + @strawberry.field def sectors(self, info: Info) -> List["TypeSector"]: """Get sectors for this dataset. diff --git a/api/types/type_dataset_source.py b/api/types/type_dataset_source.py new file mode 100644 index 00000000..1655c2fb --- /dev/null +++ b/api/types/type_dataset_source.py @@ -0,0 +1,62 @@ +"""GraphQL types for third-party platform imports (link-only).""" + +from datetime import datetime +from typing import List, Optional + +import strawberry +import strawberry_django +from strawberry import auto +from strawberry.enum import EnumType + +from api.models import DatasetSource +from api.services.platform_importers import PlatformDatasetInfo +from api.types.base_type import BaseType +from api.utils.enums import ImportPlatform + +import_platform_enum: EnumType = strawberry.enum(ImportPlatform) # type: ignore + + +@strawberry_django.type(DatasetSource) +class TypeDatasetSource(BaseType): + """Where an imported dataset came from.""" + + id: auto + platform: import_platform_enum # type: ignore + source_identifier: auto + source_url: auto + source_author: auto + source_license: auto + source_last_updated: auto + imported_at: auto + last_synced_at: auto + + +@strawberry.type +class TypePlatformDatasetPreview: + """What an import *would* create — returned by the preview query, no side effects.""" + + platform: import_platform_enum # type: ignore + identifier: str + title: str + description: str + source_url: str + author: str + license: str + mapped_license: str + tags: List[str] + last_updated: Optional[datetime] + + @classmethod + def from_info(cls, info: PlatformDatasetInfo) -> "TypePlatformDatasetPreview": + return cls( + platform=ImportPlatform(info.platform), + identifier=info.identifier, + title=info.title, + description=info.description, + source_url=info.source_url, + author=info.author, + license=info.license, + mapped_license=info.mapped_license, + tags=list(info.tags), + last_updated=info.last_updated, + ) diff --git a/api/types/type_resource.py b/api/types/type_resource.py index cd6d09c3..e77d351e 100644 --- a/api/types/type_resource.py +++ b/api/types/type_resource.py @@ -68,6 +68,7 @@ class TypeResource(BaseType): preview_enabled: auto preview_details: Optional[TypePreviewDetails] download_count: auto + url: auto # @strawberry.field # def model_resources(self) -> List[TypeAccessModelResourceFields]: diff --git a/api/utils/enums.py b/api/utils/enums.py index 646749ff..49e1a610 100644 --- a/api/utils/enums.py +++ b/api/utils/enums.py @@ -324,3 +324,11 @@ class EndpointAuthType(models.TextChoices): OAUTH2 = "OAUTH2" CUSTOM = "CUSTOM" NONE = "NONE" + + +class ImportPlatform(models.TextChoices): + """Third-party platforms a dataset can be imported (link-only) from.""" + + KAGGLE = "KAGGLE" + HUGGINGFACE = "HUGGINGFACE" + GITHUB = "GITHUB" diff --git a/api/views/download_view.py b/api/views/download_view.py index cce801a0..5d106483 100644 --- a/api/views/download_view.py +++ b/api/views/download_view.py @@ -6,7 +6,7 @@ from asgiref.sync import sync_to_async from django.core.exceptions import ObjectDoesNotExist from django.core.files.uploadedfile import UploadedFile -from django.http import HttpRequest, HttpResponse, JsonResponse +from django.http import HttpRequest, HttpResponse, HttpResponseRedirect, JsonResponse from pyecharts.charts.chart import Chart from pyecharts.render import make_snapshot from selenium import webdriver @@ -16,6 +16,7 @@ from api.models import Resource, ResourceChartDetails, ResourceChartImage from api.types.type_resource_chart import chart_base +from api.utils.enums import DataType @sync_to_async @@ -47,13 +48,21 @@ def get_resource_response( resource: Resource, request: Optional[HttpRequest] = None ) -> HttpResponse: """Get file response for a resource.""" - file_details = resource.resourcefiledetails + # Link-only (platform-imported) resources: we hold no bytes, send the + # user to the file on the source platform. Still counts as a download. + if resource.type == DataType.EXTERNAL: + if not resource.url: + return JsonResponse({"error": "External resource has no URL"}, status=404) + resource.download_count += 1 + resource.save(update_fields=["download_count"]) + _track_download(resource, request) + return HttpResponseRedirect(resource.url) + + file_details = getattr(resource, "resourcefiledetails", None) if not file_details or not file_details.file: return JsonResponse({"error": "File not found"}, status=404) - response = HttpResponse( - file_details.file.read(), content_type="application/octet-stream" - ) + response = HttpResponse(file_details.file.read(), content_type="application/octet-stream") # Handle filename and basename explicitly default_name = f"resource_{resource.name}.csv" @@ -69,7 +78,14 @@ def get_resource_response( resource.download_count += 1 resource.save() - # Track the download activity if the user is authenticated + _track_download(resource, request) + + response["Content-Disposition"] = f'attachment; filename="{basename}"' + return response + + +def _track_download(resource: Resource, request: Optional[HttpRequest]) -> None: + """Record the download in the activity stream for authenticated users.""" if request and hasattr(request, "user") and request.user.is_authenticated: # Import here to avoid circular imports import asyncio @@ -80,9 +96,6 @@ def get_resource_response( sync_to_async(track_resource_downloaded)(request.user, resource, request) ) - response["Content-Disposition"] = f'attachment; filename="{basename}"' - return response - @sync_to_async def get_chart_image_response(chart_image: ResourceChartImage) -> HttpResponse: @@ -90,9 +103,7 @@ def get_chart_image_response(chart_image: ResourceChartImage) -> HttpResponse: if not chart_image.image: return JsonResponse({"error": "File not found"}, status=404) - response = HttpResponse( - chart_image.image.read(), content_type="application/octet-stream" - ) + response = HttpResponse(chart_image.image.read(), content_type="application/octet-stream") # Handle filename and basename explicitly default_name = f"chart_{chart_image.id}.png" @@ -180,9 +191,7 @@ def get_file_chart_image_response(chart_image: ResourceChartImage) -> HttpRespon file_obj.seek(0) # Reset file pointer response = HttpResponse(file_obj, content_type=mime_type) file_name = str(file_obj.name) - response["Content-Disposition"] = ( - f'attachment; filename="{os.path.basename(file_name)}"' - ) + response["Content-Disposition"] = f'attachment; filename="{os.path.basename(file_name)}"' else: response = HttpResponse("File doesn't exist", content_type="text/plain") return response @@ -192,9 +201,7 @@ def get_custom_webdriver() -> WebDriver: """Configure and return a custom Selenium WebDriver.""" chrome_options = Options() chrome_options.add_argument("--no-sandbox") # Bypass OS security model - chrome_options.add_argument( - "--disable-dev-shm-usage" - ) # Overcome limited resource problems + chrome_options.add_argument("--disable-dev-shm-usage") # Overcome limited resource problems chrome_options.add_argument("--headless") # Run headless browser chrome_options.add_argument("--disable-gpu") # Disable GPU for headless browser diff --git a/api/views/search_dataset.py b/api/views/search_dataset.py index 680d5093..b58b8115 100644 --- a/api/views/search_dataset.py +++ b/api/views/search_dataset.py @@ -81,6 +81,7 @@ class DatasetDocumentSerializer(serializers.ModelSerializer): tags = serializers.ListField() sectors = serializers.ListField() formats = serializers.ListField() + source_platform = serializers.CharField(required=False, allow_null=True) catalogs = serializers.ListField() geographies = serializers.ListField() has_charts = serializers.BooleanField() @@ -118,6 +119,7 @@ class Meta: "tags", "sectors", "formats", + "source_platform", "catalogs", "geographies", "has_charts", @@ -175,6 +177,7 @@ def get_searchable_and_aggregations(self) -> Tuple[List[str], Dict[str, str]]: "catalogs.raw": "terms", "geographies.raw": "terms", "dataset_type": "terms", + "source_platform": "terms", } for metadata in enabled_metadata: # type: Metadata if metadata.filterable: @@ -273,6 +276,13 @@ def add_filters(self, filters: Dict[str, str], search: Search) -> Search: elif filter == "dataset_type": # Filter by dataset type (DATA or PROMPT) search = search.filter("term", dataset_type=filters[filter]) + elif filter == "source_platform": + # Filter by import platform (HUGGINGFACE, GITHUB, KAGGLE); "NATIVE" + # selects datasets that were not imported at all. + if filters[filter] == "NATIVE": + search = search.exclude("exists", field="source_platform") + else: + search = search.filter("terms", source_platform=filters[filter].split(",")) elif filter == "task_type": # Filter by prompt task type (nested in prompt_metadata) search = search.filter( diff --git a/api/views/search_unified.py b/api/views/search_unified.py index e718d580..1323f5ec 100644 --- a/api/views/search_unified.py +++ b/api/views/search_unified.py @@ -64,6 +64,7 @@ class UserSerializer(serializers.Serializer): # Type-specific fields # Dataset specific formats = serializers.ListField(required=False) + source_platform = serializers.CharField(required=False, allow_null=True) has_charts = serializers.BooleanField(required=False) download_count = serializers.IntegerField(required=False) is_individual_dataset = serializers.BooleanField(required=False) diff --git a/search/documents/dataset_document.py b/search/documents/dataset_document.py index 83f8d6e7..559d5cf5 100644 --- a/search/documents/dataset_document.py +++ b/search/documents/dataset_document.py @@ -6,6 +6,7 @@ Catalog, Dataset, DatasetMetadata, + DatasetSource, Geography, Metadata, Organization, @@ -94,6 +95,10 @@ class DatasetDocument(Document): } ) + # Platform this dataset was imported from (KAGGLE, HUGGINGFACE) or null + # for datasets created natively. Lets listings badge/filter imports. + source_platform = fields.KeywordField(attr="source_platform_indexing") + formats = fields.TextField( attr="formats_indexing", analyzer=ngram_analyser, @@ -238,6 +243,8 @@ def get_instances_from_related( """Get Dataset instances from related models.""" if isinstance(related_instance, Resource): return related_instance.dataset + elif isinstance(related_instance, DatasetSource): + return related_instance.dataset elif isinstance(related_instance, Metadata): ds_metadata_objects = related_instance.datasetmetadata_set.all() return [obj.dataset for obj in ds_metadata_objects] # type: ignore @@ -271,6 +278,7 @@ class Django: related_models = [ Resource, + DatasetSource, Metadata, DatasetMetadata, PromptDataset, From 92e8487523296b897fbe7cd89898cbe3ed4f5f21 Mon Sep 17 00:00:00 2001 From: Anant Jain Date: Tue, 22 Sep 2026 01:56:05 +0530 Subject: [PATCH 2/8] fix(api): do not download or store the Hugging Face file list on import The Hub's dataset endpoint returns `siblings`, one entry per file, by default. For large repos that key dominates the response: 9.6 MB for an 85k-file repo against ~4 KB for everything else. The importer stored the whole response in DatasetSource.raw_metadata and fetched it on every preview and import, although files are never listed. Request fields by name with `expand[]` (every default field except `siblings`, plus `citation`), and drop `siblings` defensively before the payload is kept. --- .../platform_importers/huggingface.py | 32 +++++++++++++++++-- 1 file changed, 30 insertions(+), 2 deletions(-) diff --git a/api/services/platform_importers/huggingface.py b/api/services/platform_importers/huggingface.py index 17b81c88..2ab4a934 100644 --- a/api/services/platform_importers/huggingface.py +++ b/api/services/platform_importers/huggingface.py @@ -1,6 +1,7 @@ """Hugging Face Hub dataset importer (metadata only). -Two requests per import: the repo metadata and the README (dataset card). +Two small requests per import: the repo metadata (without its file list) and +the README (dataset card). Public datasets need no token; ``HF_TOKEN`` (settings) is sent when present so gated/private repos the token can see also work. """ @@ -32,6 +33,28 @@ MAX_DESCRIPTION = 1000 # Dataset.description column length +# Everything the Hub returns by default EXCEPT ``siblings`` (the per-file list). +# We never list files, and for large repos that one key is most of the payload: +# ~9.6 MB for an 85k-file repo versus ~4 KB without it. Asking for fields by +# name means the list is never downloaded, and never stored in raw_metadata. +_EXPAND_FIELDS = ( + "author", + "cardData", + "citation", + "createdAt", + "description", + "disabled", + "downloads", + "gated", + "lastModified", + "likes", + "paperswithcode_id", + "private", + "sha", + "tags", + "usedStorage", +) + class HuggingFaceImporter(PlatformImporter): platform = ImportPlatform.HUGGINGFACE @@ -87,7 +110,11 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: headers = self._headers() try: - meta: Dict[str, Any] = self._get_json(f"{HF_API}/{repo_id}", headers=headers) + meta: Dict[str, Any] = self._get_json( + f"{HF_API}/{repo_id}", + headers=headers, + params=[("expand[]", name) for name in _EXPAND_FIELDS], + ) except PlatformAuthError as exc: # Hugging Face answers 401 for repos that do not exist as well as # for private/gated ones, so say both. @@ -95,6 +122,7 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: f"'{repo_id}' was not found on Hugging Face, or it is private/gated. " "Check the spelling; only public datasets can be imported." ) from exc + meta.pop("siblings", None) # never keep the file list, whatever the API sends if meta.get("disabled"): raise PlatformImportError("This dataset has been disabled on Hugging Face") From 158f2656cce312e9003611794694f23e513120a5 Mon Sep 17 00:00:00 2001 From: Anant Jain Date: Thu, 24 Sep 2026 02:11:22 +0530 Subject: [PATCH 3/8] feat(api): store imported metadata in typed columns, drop the raw payload DatasetSource no longer keeps the platform's raw JSON. Every field we fetch now lands in a typed column, chosen by one rule: a metadata standard reads it on export (DCAT / Croissant / Dublin Core) or the platform itself reads it (attribution, duplicate check, licence review). New columns: revision (commit hash or version), source_created_at, source_readme (full card; Dataset.description keeps a 1,000-char cut), citation, languages, source_homepage, is_archived. Column definitions a platform declares (Hugging Face dataset_info) become ResourceSchema rows on the link resource, so the columns list works without fetching data. Importers fill what each platform provides: Hugging Face all of the above; GitHub adds one small call for the branch head SHA and reads homepage/archived; Kaggle uses the version number and the earliest version date. Fields nothing reads (downloads, likes, stars, size categories, task taxonomy) are no longer fetched. --- api/migrations/0048_platform_import.py | 10 +++- api/models/DatasetSource.py | 33 ++++++++++-- api/services/platform_import_service.py | 21 +++++++- api/services/platform_importers/__init__.py | 2 + api/services/platform_importers/base.py | 48 ++++++++++++++++- api/services/platform_importers/github.py | 30 ++++++++--- .../platform_importers/huggingface.py | 53 +++++++++++++++---- api/services/platform_importers/kaggle.py | 21 ++++++-- api/types/type_dataset_source.py | 18 ++++++- 9 files changed, 203 insertions(+), 33 deletions(-) diff --git a/api/migrations/0048_platform_import.py b/api/migrations/0048_platform_import.py index ac2e5107..9efe2d02 100644 --- a/api/migrations/0048_platform_import.py +++ b/api/migrations/0048_platform_import.py @@ -1,4 +1,4 @@ -# Generated by Django 5.0.4 on 2026-09-20 10:13 +# Generated by Django 5.0.4 on 2026-09-23 20:36 import uuid @@ -37,10 +37,16 @@ class Migration(migrations.Migration): ), ("source_identifier", models.CharField(max_length=300)), ("source_url", models.URLField(max_length=500)), + ("source_homepage", models.URLField(blank=True, max_length=500)), + ("revision", models.CharField(blank=True, max_length=64)), ("source_author", models.CharField(blank=True, max_length=300)), ("source_license", models.CharField(blank=True, max_length=300)), + ("source_readme", models.TextField(blank=True)), + ("citation", models.TextField(blank=True)), + ("languages", models.JSONField(blank=True, default=list)), + ("source_created_at", models.DateTimeField(blank=True, null=True)), ("source_last_updated", models.DateTimeField(blank=True, null=True)), - ("raw_metadata", models.JSONField(blank=True, default=dict)), + ("is_archived", models.BooleanField(default=False)), ("imported_at", models.DateTimeField(auto_now_add=True)), ("last_synced_at", models.DateTimeField(auto_now=True)), ( diff --git a/api/models/DatasetSource.py b/api/models/DatasetSource.py index e9aa7817..ccf0b2ad 100644 --- a/api/models/DatasetSource.py +++ b/api/models/DatasetSource.py @@ -9,25 +9,48 @@ class DatasetSource(models.Model): """Provenance record for a dataset imported from a third-party platform. Imports are link-only: DataSpace never copies the platform's files. This - row remembers where the dataset came from so the UI can attribute it, link - back to it, and (later) re-sync its metadata. The raw platform response is - kept in ``raw_metadata`` for debugging and future field mapping. + row holds what the platform told us about the dataset, in typed columns. + Every column here has a reader: either a metadata standard on export + (DCAT / Croissant / Dublin Core) or the platform itself (attribution, + duplicate detection, licence review). No raw payload is kept. """ id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) dataset = models.OneToOneField("api.Dataset", on_delete=models.CASCADE, related_name="source") platform = models.CharField(max_length=50, choices=ImportPlatform.choices) + + # --- identity on the platform ------------------------------------------ # Platform-native identifier, e.g. "owner/dataset-slug" (Kaggle) or # "namespace/name" (Hugging Face). Normalised by the importer. source_identifier = models.CharField(max_length=300) - # Human-facing page on the platform. + # Human-facing page on the platform. Export: schema:sameAs / prov:wasDerivedFrom. source_url = models.URLField(max_length=500) + # Home page declared by the source, if any (GitHub `homepage`). Export: dcat:landingPage. + source_homepage = models.URLField(max_length=500, blank=True) + # Commit hash (Hugging Face / GitHub) or version number (Kaggle) at import time. + # Export: Croissant `version`. Later: what a sync compares against. + revision = models.CharField(max_length=64, blank=True) + + # --- descriptive metadata the standards read ---------------------------- + # Who made the data on the platform. Export: dcterms:creator / Croissant creator. source_author = models.CharField(max_length=300, blank=True) # License string exactly as the platform reported it (may not map onto # DatasetLicense; the mapped value lives on Dataset.license). source_license = models.CharField(max_length=300, blank=True) + # Full dataset card / README. Dataset.description keeps a 1,000-char cut. + source_readme = models.TextField(blank=True) + # BibTeX or free-text citation, when the platform provides one. Export: Croissant citeAs. + citation = models.TextField(blank=True) + # Language codes of the data, e.g. ["en", "hi"]. Export: dcterms:language / inLanguage. + languages = models.JSONField(default=list, blank=True) + # When the dataset was first published on the platform. Export: dcterms:issued. + source_created_at = models.DateTimeField(null=True, blank=True) + # When the platform last changed it. Export: dcterms:modified. source_last_updated = models.DateTimeField(null=True, blank=True) - raw_metadata = models.JSONField(default=dict, blank=True) + # Source is frozen / read-only upstream (GitHub `archived`). Shown as a hint. + is_archived = models.BooleanField(default=False) + + # --- our side ------------------------------------------------------------- imported_by = models.ForeignKey( "authorization.User", on_delete=models.SET_NULL, diff --git a/api/services/platform_import_service.py b/api/services/platform_import_service.py index b30fe17b..e60e2304 100644 --- a/api/services/platform_import_service.py +++ b/api/services/platform_import_service.py @@ -24,6 +24,7 @@ Metadata, Organization, Resource, + ResourceSchema, Sector, Tag, ) @@ -202,7 +203,7 @@ def import_platform_dataset( # the dataset page, so "download" redirects there and people browse/fetch # files with the platform's own tooling. platform_label = PLATFORM_LABELS.get(str(info.platform), str(info.platform).title()) - Resource.objects.create( + link_resource = Resource.objects.create( dataset=dataset, type=DataType.EXTERNAL, name=f"Dataset on {platform_label}"[:200], @@ -215,13 +216,29 @@ def import_platform_dataset( platform=info.platform, source_identifier=info.identifier, source_url=info.source_url[:500], + source_homepage=(info.homepage or "")[:500], + revision=(info.revision or "")[:64], source_author=info.author[:300], source_license=info.license[:300], + source_readme=info.readme or "", + citation=info.citation or "", + languages=list(info.languages or []), + source_created_at=info.created_at, source_last_updated=info.last_updated, - raw_metadata=info.raw, + is_archived=bool(info.is_archived), imported_by=user, ) + # Column definitions the platform declared (Hugging Face dataset_info). + # They live on the link resource so "View All Columns" and the Croissant + # recordSet work without any data being fetched. + ResourceSchema.objects.bulk_create( + [ + ResourceSchema(resource=link_resource, field_name=col.name, format=col.field_type) + for col in info.columns + ] + ) + try: owner_role = Role.objects.get(name="owner") except Role.DoesNotExist as exc: diff --git a/api/services/platform_importers/__init__.py b/api/services/platform_importers/__init__.py index 246c6b08..be240499 100644 --- a/api/services/platform_importers/__init__.py +++ b/api/services/platform_importers/__init__.py @@ -10,6 +10,7 @@ from api.services.platform_importers.base import ( InvalidIdentifierError, PlatformAuthError, + PlatformColumn, PlatformDatasetInfo, PlatformDatasetNotFoundError, PlatformImporter, @@ -53,6 +54,7 @@ def detect_platform(value: str) -> Optional[str]: "detect_platform", "PlatformImporter", "PlatformDatasetInfo", + "PlatformColumn", "PlatformImportError", "InvalidIdentifierError", "PlatformDatasetNotFoundError", diff --git a/api/services/platform_importers/base.py b/api/services/platform_importers/base.py index bc6a671a..3efcd360 100644 --- a/api/services/platform_importers/base.py +++ b/api/services/platform_importers/base.py @@ -53,18 +53,36 @@ class PlatformUnavailableError(PlatformImportError): # --------------------------------------------------------------------------- # # Normalised result # --------------------------------------------------------------------------- # +@dataclass +class PlatformColumn: + """One column of the dataset as the platform describes it.""" + + name: str + field_type: str # a FieldTypes value: STRING / NUMBER / INTEGER / DATE / BOOLEAN + + @dataclass class PlatformDatasetInfo: + """Everything an importer returns. Each field lands in a typed column on + DatasetSource (or on Dataset / ResourceSchema); nothing raw is kept.""" + platform: str identifier: str title: str - description: str + description: str # short form, fits Dataset.description (1,000 chars) source_url: str author: str = "" license: str = "" tags: List[str] = field(default_factory=list) last_updated: Optional[datetime] = None - raw: Dict[str, Any] = field(default_factory=dict) + created_at: Optional[datetime] = None + revision: str = "" + readme: str = "" # full card / README, unbounded + citation: str = "" + languages: List[str] = field(default_factory=list) + homepage: str = "" + is_archived: bool = False + columns: List[PlatformColumn] = field(default_factory=list) @property def mapped_license(self) -> str: @@ -103,6 +121,29 @@ def map_license(raw: str) -> str: return LICENSE_ALIASES.get(key, DatasetLicense.CC_BY_4_0_ATTRIBUTION) +def field_type_for(dtype: Any) -> str: + """Map a platform column type (Hugging Face / Arrow style names) onto FieldTypes.""" + name = str(dtype if isinstance(dtype, str) else "").lower() + if name in ("bool", "boolean"): + return "BOOLEAN" + if name.startswith(("int", "uint")): + return "INTEGER" + if name.startswith(("float", "double", "decimal")): + return "NUMBER" + if name.startswith(("date", "timestamp", "time")): + return "DATE" + return "STRING" # strings, class labels, nested/binary types + + +def shorten(text: str, limit: int = 1000) -> str: + """Cut on a word boundary with an ellipsis; used for Dataset.description.""" + text = (text or "").strip() + if len(text) <= limit: + return text + cut = text[: limit - 1].rsplit(" ", 1)[0] + return cut + "…" + + def parse_iso_datetime(value: Optional[str]) -> Optional[datetime]: """Parse a platform timestamp. Django's parser copes with a trailing ``Z`` and any number of fractional digits (Kaggle sends e.g. ``...:04.7Z``), @@ -186,11 +227,14 @@ def _get_json(self, url: str, **kwargs: Any) -> Any: "ImportPlatform", "PlatformImporter", "PlatformDatasetInfo", + "PlatformColumn", "PlatformImportError", "InvalidIdentifierError", "PlatformDatasetNotFoundError", "PlatformAuthError", "PlatformUnavailableError", "map_license", + "field_type_for", + "shorten", "parse_iso_datetime", ] diff --git a/api/services/platform_importers/github.py b/api/services/platform_importers/github.py index c5202e31..8a525228 100644 --- a/api/services/platform_importers/github.py +++ b/api/services/platform_importers/github.py @@ -1,7 +1,8 @@ """GitHub repository importer (metadata only). -A repo (optionally a branch and sub-folder) is treated as a dataset. Two -requests per import: the repo metadata and the README. Public repos need no +A repo (optionally a branch and sub-folder) is treated as a dataset. Three +small requests per import: the repo metadata, the branch head (for the +revision) and the README. Public repos need no token; ``GITHUB_TOKEN`` (settings) is sent when present, mainly to lift the anonymous 60 requests/hour rate limit. """ @@ -20,6 +21,7 @@ PlatformImporter, PlatformImportError, parse_iso_datetime, + shorten, ) from api.utils.enums import ImportPlatform @@ -114,26 +116,28 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: if sub_path: source_url += f"/tree/{branch}/{sub_path}" - description = self._description(full_name, headers) or str(meta.get("description") or "") + readme = self._readme(full_name, headers) or str(meta.get("description") or "") license_info = meta.get("license") or {} - raw = dict(meta) - raw["_branch"] = branch return PlatformDatasetInfo( platform=self.platform, identifier=self.parse_identifier(identifier), title=title[:300], - description=description[:MAX_DESCRIPTION], + description=shorten(readme, MAX_DESCRIPTION), source_url=source_url, author=str((meta.get("owner") or {}).get("login") or owner)[:300], license=str(license_info.get("spdx_id") or license_info.get("name") or ""), tags=[t[:50] for t in (meta.get("topics") or []) if isinstance(t, str)][:30], last_updated=parse_iso_datetime(meta.get("pushed_at") or meta.get("updated_at")), - raw=raw, + created_at=parse_iso_datetime(meta.get("created_at")), + revision=self._head_sha(full_name, branch, headers), + readme=readme, + homepage=str(meta.get("homepage") or "")[:500], + is_archived=bool(meta.get("archived")), ) # -- pieces -------------------------------------------------------------- # - def _description(self, full_name: str, headers: Dict[str, str]) -> str: + def _readme(self, full_name: str, headers: Dict[str, str]) -> str: try: response = self._get( f"{GITHUB_API}/repos/{full_name}/readme", @@ -142,3 +146,13 @@ def _description(self, full_name: str, headers: Dict[str, str]) -> str: return response.text.strip() except PlatformImportError: return "" + + def _head_sha(self, full_name: str, branch: str, headers: Dict[str, str]) -> str: + """Commit SHA at the branch head; empty if the lookup fails.""" + try: + ref: Dict[str, Any] = self._get_json( + f"{GITHUB_API}/repos/{full_name}/git/ref/heads/{branch}", headers=headers + ) + return str((ref.get("object") or {}).get("sha") or "")[:64] + except PlatformImportError: + return "" diff --git a/api/services/platform_importers/huggingface.py b/api/services/platform_importers/huggingface.py index 2ab4a934..bf4ea994 100644 --- a/api/services/platform_importers/huggingface.py +++ b/api/services/platform_importers/huggingface.py @@ -17,10 +17,13 @@ from api.services.platform_importers.base import ( InvalidIdentifierError, PlatformAuthError, + PlatformColumn, PlatformDatasetInfo, PlatformImporter, PlatformImportError, + field_type_for, parse_iso_datetime, + shorten, ) from api.utils.enums import ImportPlatform @@ -36,7 +39,7 @@ # Everything the Hub returns by default EXCEPT ``siblings`` (the per-file list). # We never list files, and for large repos that one key is most of the payload: # ~9.6 MB for an 85k-file repo versus ~4 KB without it. Asking for fields by -# name means the list is never downloaded, and never stored in raw_metadata. +# name means the list is never downloaded, and never stored anywhere. _EXPAND_FIELDS = ( "author", "cardData", @@ -44,15 +47,11 @@ "createdAt", "description", "disabled", - "downloads", "gated", "lastModified", - "likes", - "paperswithcode_id", "private", "sha", "tags", - "usedStorage", ) @@ -131,18 +130,24 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: title = card.get("pretty_name") or repo_id.split("/")[-1] author = meta.get("author") or (repo_id.split("/")[0] if "/" in repo_id else "") + readme = self._readme(repo_id, meta, headers) return PlatformDatasetInfo( platform=self.platform, identifier=repo_id, title=str(title)[:300], - description=self._description(repo_id, meta, headers), + description=shorten(readme, MAX_DESCRIPTION), source_url=f"{HF_WEB}/{repo_id}", author=str(author)[:300], license=self._license(card, tags), tags=self._tags(tags), last_updated=parse_iso_datetime(meta.get("lastModified")), - raw=meta, + created_at=parse_iso_datetime(meta.get("createdAt")), + revision=str(meta.get("sha") or "")[:64], + readme=readme, + citation=str(meta.get("citation") or ""), + languages=self._languages(card, tags), + columns=self._columns(card), ) # -- pieces -------------------------------------------------------------- # @@ -174,8 +179,8 @@ def _tags(tags: List[str]) -> List[str]: out.append(value[:50]) return out[:30] - def _description(self, repo_id: str, meta: Dict[str, Any], headers: Dict[str, str]) -> str: - """Prefer the dataset card body (README) over the terse API field.""" + def _readme(self, repo_id: str, meta: Dict[str, Any], headers: Dict[str, str]) -> str: + """Full dataset card body (README) without its YAML front matter; else the API field.""" try: response = self._get(f"{HF_WEB}/{repo_id}/resolve/main/README.md", headers=headers) text = response.text @@ -190,4 +195,32 @@ def _description(self, repo_id: str, meta: Dict[str, Any], headers: Dict[str, st text = text.strip() if not text: text = str(meta.get("description") or "") - return text[:MAX_DESCRIPTION] + return text + + @staticmethod + def _languages(card: Dict[str, Any], tags: List[str]) -> List[str]: + langs = card.get("language") + if isinstance(langs, str): + langs = [langs] + out = [str(v).strip() for v in (langs or []) if str(v).strip()] + if not out: + out = [t[len("language:") :] for t in tags if t.startswith("language:")] + return out[:20] + + @staticmethod + def _columns(card: Dict[str, Any]) -> List[PlatformColumn]: + """Column names/types from the card's dataset_info (first config if several).""" + info = card.get("dataset_info") + if isinstance(info, list): + info = info[0] if info else None + if not isinstance(info, dict): + return [] + cols: List[PlatformColumn] = [] + for feat in info.get("features") or []: + if isinstance(feat, dict) and feat.get("name"): + cols.append( + PlatformColumn( + name=str(feat["name"])[:255], field_type=field_type_for(feat.get("dtype")) + ) + ) + return cols[:200] diff --git a/api/services/platform_importers/kaggle.py b/api/services/platform_importers/kaggle.py index 54bbef44..bb831254 100644 --- a/api/services/platform_importers/kaggle.py +++ b/api/services/platform_importers/kaggle.py @@ -21,6 +21,7 @@ PlatformDatasetInfo, PlatformImporter, parse_iso_datetime, + shorten, ) from api.utils.enums import ImportPlatform @@ -85,19 +86,22 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: raise PlatformAuthError("This Kaggle dataset is private") title = meta.get("title") or slug - description = (meta.get("description") or meta.get("subtitle") or "").strip() + readme = (meta.get("description") or meta.get("subtitle") or "").strip() + version = meta.get("currentVersionNumber") return PlatformDatasetInfo( platform=self.platform, identifier=ref, title=str(title)[:300], - description=description[:MAX_DESCRIPTION], + description=shorten(readme, MAX_DESCRIPTION), source_url=meta.get("url") or f"{KAGGLE_WEB}/{ref}", author=str(meta.get("ownerName") or owner)[:300], license=self._license(meta), tags=self._tags(meta), last_updated=parse_iso_datetime(meta.get("lastUpdated")), - raw=meta, + created_at=self._first_version_date(meta), + revision=str(version) if version is not None else "", + readme=readme, ) # -- pieces -------------------------------------------------------------- # @@ -119,3 +123,14 @@ def _tags(meta: Dict[str, Any]) -> List[str]: if isinstance(name, str) and name.strip() and name.strip() not in out: out.append(name.strip()[:50]) return out[:30] + + @staticmethod + def _first_version_date(meta: Dict[str, Any]): + """Kaggle has no created date; the earliest version's creation date is the same thing.""" + dates = [ + parse_iso_datetime(v.get("creationDate")) + for v in (meta.get("versions") or []) + if isinstance(v, dict) + ] + dates = [d for d in dates if d is not None] + return min(dates) if dates else None diff --git a/api/types/type_dataset_source.py b/api/types/type_dataset_source.py index 1655c2fb..00c62261 100644 --- a/api/types/type_dataset_source.py +++ b/api/types/type_dataset_source.py @@ -18,18 +18,28 @@ @strawberry_django.type(DatasetSource) class TypeDatasetSource(BaseType): - """Where an imported dataset came from.""" + """Where an imported dataset came from, and what the platform said about it.""" id: auto platform: import_platform_enum # type: ignore source_identifier: auto source_url: auto + source_homepage: auto + revision: auto source_author: auto source_license: auto + source_readme: auto + citation: auto + source_created_at: auto source_last_updated: auto + is_archived: auto imported_at: auto last_synced_at: auto + @strawberry.field + def languages(self) -> List[str]: + return list(getattr(self, "languages", None) or []) + @strawberry.type class TypePlatformDatasetPreview: @@ -45,6 +55,9 @@ class TypePlatformDatasetPreview: mapped_license: str tags: List[str] last_updated: Optional[datetime] + languages: List[str] + revision: str + column_count: int @classmethod def from_info(cls, info: PlatformDatasetInfo) -> "TypePlatformDatasetPreview": @@ -59,4 +72,7 @@ def from_info(cls, info: PlatformDatasetInfo) -> "TypePlatformDatasetPreview": mapped_license=info.mapped_license, tags=list(info.tags), last_updated=info.last_updated, + languages=list(info.languages), + revision=info.revision, + column_count=len(info.columns), ) From f7fe195aa39340192c4d22c78a6bf76a7564aed4 Mon Sep 17 00:00:00 2001 From: Anant Jain Date: Thu, 24 Sep 2026 19:08:08 +0530 Subject: [PATCH 4/8] fix(api): harden platform import after a live probe of 36 datasets Found by importing many real datasets and fuzzing identifiers: - Hugging Face: keep the platform's canonical id ("imdb" is really "stanfordnlp/imdb"), so the same dataset cannot be imported twice under a legacy name. Duplicate check re-runs under the canonical id. - GitHub: SPDX "NOASSERTION"/"other" means no detectable licence; treat as empty instead of a licence called NOASSERTION. Branch names and folder paths are validated (no "..", no whitespace, plain segments only). - Kaggle: a 403 covers "does not exist" as well as private, so say both. The created date is taken from the earliest version only when the view lists every version (it lists one of 2,324 for kaggle/meta-kaggle). - All: identifiers over 200 characters are rejected; README kept to 200 KB; citation to 20 KB. - Service: platform calls now happen before the transaction is opened, so a slow platform never holds a database connection. --- api/services/platform_import_service.py | 26 ++++++++++++++- api/services/platform_importers/base.py | 29 +++++++++++++++++ api/services/platform_importers/github.py | 23 +++++++++++-- .../platform_importers/huggingface.py | 11 +++++-- api/services/platform_importers/kaggle.py | 32 +++++++++++++------ 5 files changed, 105 insertions(+), 16 deletions(-) diff --git a/api/services/platform_import_service.py b/api/services/platform_import_service.py index e60e2304..ffcaf5d8 100644 --- a/api/services/platform_import_service.py +++ b/api/services/platform_import_service.py @@ -153,7 +153,6 @@ def _prefill_metadata(dataset: Dataset, info: PlatformDatasetInfo) -> None: logger.info("platform_import_metadata_skipped", label=field.label, error=str(exc)) -@transaction.atomic def import_platform_dataset( *, platform: str, @@ -172,9 +171,34 @@ def import_platform_dataset( if existing is not None: raise DuplicateImportError(existing) + # Network first, database second: the platform calls (up to a few seconds) + # happen before any transaction is opened, so a slow platform never holds a + # database connection. if info is None: info = importer.fetch_dataset_info(canonical_id) + # The platform may have canonicalised the id (Hugging Face "imdb" -> + # "stanfordnlp/imdb"); re-check duplicates under the canonical id too. + if info.identifier != canonical_id: + existing = find_existing_import(platform, info.identifier, organization, user) + if existing is not None: + raise DuplicateImportError(existing) + + with transaction.atomic(): + return _create_import( + info, user=user, organization=organization, dataspace=dataspace, title=title + ) + + +def _create_import( + info: PlatformDatasetInfo, + *, + user: User, + organization: Optional[Organization], + dataspace: Optional[DataSpace], + title: Optional[str], +) -> Dataset: + """All database writes for one import. Runs inside a transaction.""" # Publisher may choose the name shown on DataSpace; platform title otherwise. display_title = (title or "").strip()[:300] or info.title dataset = Dataset.objects.create( diff --git a/api/services/platform_importers/base.py b/api/services/platform_importers/base.py index 3efcd360..61518bfb 100644 --- a/api/services/platform_importers/base.py +++ b/api/services/platform_importers/base.py @@ -8,6 +8,7 @@ from __future__ import annotations +import re from abc import ABC, abstractmethod from dataclasses import dataclass, field from datetime import datetime @@ -121,6 +122,30 @@ def map_license(raw: str) -> str: return LICENSE_ALIASES.get(key, DatasetLicense.CC_BY_4_0_ATTRIBUTION) +# Hard limits: the platforms' own maxima are well under these, so anything +# longer is not a real identifier. Keeps hostile input out of URLs and columns. +MAX_IDENTIFIER_LEN = 200 +MAX_README_CHARS = 200_000 # ~200 KB; the longest real card seen is ~32 KB + +_SEGMENT_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$") + + +def check_identifier_length(value: str, platform_label: str) -> None: + if len(value) > MAX_IDENTIFIER_LEN: + raise InvalidIdentifierError(f"That {platform_label} identifier is too long to be real") + + +def check_path_segments(path: str, what: str) -> None: + """A branch name or sub-path must be plain segments: no '..', no whitespace, no odd characters.""" + for seg in (path or "").split("/"): + if seg and (seg in (".", "..") or not _SEGMENT_RE.match(seg)): + raise InvalidIdentifierError(f"'{path}' is not a valid {what}") + + +def cap_readme(text: str) -> str: + return (text or "")[:MAX_README_CHARS] + + def field_type_for(dtype: Any) -> str: """Map a platform column type (Hugging Face / Arrow style names) onto FieldTypes.""" name = str(dtype if isinstance(dtype, str) else "").lower() @@ -234,6 +259,10 @@ def _get_json(self, url: str, **kwargs: Any) -> Any: "PlatformAuthError", "PlatformUnavailableError", "map_license", + "check_identifier_length", + "check_path_segments", + "cap_readme", + "MAX_IDENTIFIER_LEN", "field_type_for", "shorten", "parse_iso_datetime", diff --git a/api/services/platform_importers/github.py b/api/services/platform_importers/github.py index 8a525228..e5792b18 100644 --- a/api/services/platform_importers/github.py +++ b/api/services/platform_importers/github.py @@ -20,6 +20,9 @@ PlatformDatasetInfo, PlatformImporter, PlatformImportError, + cap_readme, + check_identifier_length, + check_path_segments, parse_iso_datetime, shorten, ) @@ -94,7 +97,13 @@ def _parse(self, value: str) -> Tuple[str, str, Optional[str], str]: raise InvalidIdentifierError( "GitHub owner and repo names use letters, digits, '-', '_' and '.'" ) - return owner, repo, (branch or None), path.strip("/") + path = path.strip("/") + if branch: + check_path_segments(branch, "branch name") + if path: + check_path_segments(path, "folder path") + check_identifier_length(f"{owner}/{repo}@{branch or ''}:{path}", "GitHub") + return owner, repo, (branch or None), path # -- fetch --------------------------------------------------------------- # def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: @@ -126,12 +135,12 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: description=shorten(readme, MAX_DESCRIPTION), source_url=source_url, author=str((meta.get("owner") or {}).get("login") or owner)[:300], - license=str(license_info.get("spdx_id") or license_info.get("name") or ""), + license=self._license(license_info), tags=[t[:50] for t in (meta.get("topics") or []) if isinstance(t, str)][:30], last_updated=parse_iso_datetime(meta.get("pushed_at") or meta.get("updated_at")), created_at=parse_iso_datetime(meta.get("created_at")), revision=self._head_sha(full_name, branch, headers), - readme=readme, + readme=cap_readme(readme), homepage=str(meta.get("homepage") or "")[:500], is_archived=bool(meta.get("archived")), ) @@ -156,3 +165,11 @@ def _head_sha(self, full_name: str, branch: str, headers: Dict[str, str]) -> str return str((ref.get("object") or {}).get("sha") or "")[:64] except PlatformImportError: return "" + + @staticmethod + def _license(license_info: Dict[str, Any]) -> str: + """SPDX id when GitHub could detect one. 'NOASSERTION' / 'other' mean it could not.""" + spdx = str(license_info.get("spdx_id") or "") + if spdx.upper() in ("", "NOASSERTION", "OTHER"): + return "" + return spdx diff --git a/api/services/platform_importers/huggingface.py b/api/services/platform_importers/huggingface.py index bf4ea994..8a8b4c20 100644 --- a/api/services/platform_importers/huggingface.py +++ b/api/services/platform_importers/huggingface.py @@ -21,6 +21,8 @@ PlatformDatasetInfo, PlatformImporter, PlatformImportError, + cap_readme, + check_identifier_length, field_type_for, parse_iso_datetime, shorten, @@ -97,6 +99,7 @@ def parse_identifier(self, value: str) -> str: repo_id = raw[len("datasets/") :] if raw.startswith("datasets/") else raw repo_id = repo_id.strip("/") + check_identifier_length(repo_id, "Hugging Face") if not _ID_RE.match(repo_id): raise InvalidIdentifierError( "Hugging Face ids look like 'namespace/name' (letters, digits, '-', '_', '.')" @@ -125,6 +128,10 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: if meta.get("disabled"): raise PlatformImportError("This dataset has been disabled on Hugging Face") + # The Hub redirects legacy short names ("imdb") to their canonical id; keep + # the canonical one so the same dataset cannot be imported twice under two names. + repo_id = str(meta.get("id") or repo_id) + card: Dict[str, Any] = meta.get("cardData") or {} tags: List[str] = meta.get("tags") or [] @@ -144,8 +151,8 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: last_updated=parse_iso_datetime(meta.get("lastModified")), created_at=parse_iso_datetime(meta.get("createdAt")), revision=str(meta.get("sha") or "")[:64], - readme=readme, - citation=str(meta.get("citation") or ""), + readme=cap_readme(readme), + citation=str(meta.get("citation") or "")[:20_000], languages=self._languages(card, tags), columns=self._columns(card), ) diff --git a/api/services/platform_importers/kaggle.py b/api/services/platform_importers/kaggle.py index bb831254..45bbba44 100644 --- a/api/services/platform_importers/kaggle.py +++ b/api/services/platform_importers/kaggle.py @@ -20,6 +20,8 @@ PlatformAuthError, PlatformDatasetInfo, PlatformImporter, + cap_readme, + check_identifier_length, parse_iso_datetime, shorten, ) @@ -68,6 +70,7 @@ def parse_identifier(self, value: str) -> str: raise InvalidIdentifierError("Kaggle refs look like 'owner/dataset-name'") owner, slug = parts + check_identifier_length(f"{owner}/{slug}", "Kaggle") if not (_SLUG_RE.match(owner) and _SLUG_RE.match(slug)): raise InvalidIdentifierError( "Kaggle owner and dataset names use letters, digits, '-' and '_'" @@ -79,9 +82,16 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: ref = self.parse_identifier(identifier) owner, slug = ref.split("/") - meta: Dict[str, Any] = self._get_json( - f"{KAGGLE_API}/datasets/view/{owner}/{slug}", auth=self._auth() - ) + try: + meta: Dict[str, Any] = self._get_json( + f"{KAGGLE_API}/datasets/view/{owner}/{slug}", auth=self._auth() + ) + except PlatformAuthError as exc: + # Kaggle answers 403 for datasets that do not exist as well as private ones. + raise PlatformAuthError( + f"'{ref}' was not found on Kaggle, or it is private. " + "Check the spelling; only public datasets can be imported." + ) from exc if meta.get("isPrivate"): raise PlatformAuthError("This Kaggle dataset is private") @@ -101,7 +111,7 @@ def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: last_updated=parse_iso_datetime(meta.get("lastUpdated")), created_at=self._first_version_date(meta), revision=str(version) if version is not None else "", - readme=readme, + readme=cap_readme(readme), ) # -- pieces -------------------------------------------------------------- # @@ -126,11 +136,13 @@ def _tags(meta: Dict[str, Any]) -> List[str]: @staticmethod def _first_version_date(meta: Dict[str, Any]): - """Kaggle has no created date; the earliest version's creation date is the same thing.""" - dates = [ - parse_iso_datetime(v.get("creationDate")) - for v in (meta.get("versions") or []) - if isinstance(v, dict) - ] + """Kaggle has no created date. The earliest version's date is the same thing, + but the view only lists recent versions for datasets with many, so use it + only when the list is complete.""" + versions = [v for v in (meta.get("versions") or []) if isinstance(v, dict)] + current = meta.get("currentVersionNumber") + if not versions or (isinstance(current, int) and len(versions) < current): + return None + dates = [parse_iso_datetime(v.get("creationDate")) for v in versions] dates = [d for d in dates if d is not None] return min(dates) if dates else None From dee78041efd0fd9ce5add21dfe784e49f6b7f0ee Mon Sep 17 00:00:00 2001 From: Anant Jain Date: Mon, 28 Sep 2026 10:52:06 +0530 Subject: [PATCH 5/8] feat(api): export dataset metadata as DCAT, Croissant and Dublin Core GET /api/datasets//export/?standard=dcat|croissant|dublin_core &format=jsonld|turtle|rdfxml|ntriples (Croissant: jsonld only) GET /api/metadata/export-options/ Generated on request from the platform's own model; nothing is stored. Published datasets are public; owners can preview drafts. `report=1` returns the document together with the gap report (values the standard wanted as URIs but got as names, fields the standard cannot carry, mandatory properties we had nothing for). The mapping is data, not code. `contracts/crosswalk.json` binds each concept to a record field and, per standard, to a property with its value type and obligation; `contracts/*.csv` hold the allowed licences, sectors and geographies with URIs. Both are owned here and edited like any other file (see contracts/README.md). `crosswalk.py` is the engine that applies the file and knows no standard by name; `adapter.py` is the only module that reads Django models and resolves names to URIs. Serialisation the mapping cannot express lives in exporter.py: dates as xsd:date literals, media types as IANA IRIs, the licence repeated on each distribution, and the @id/contentUrl/encodingFormat Croissant requires on every FileObject. Imported datasets export with provenance to the source page and the platform author as creator; the link resource becomes a distribution with accessURL, never downloadURL. Not covered yet: Croissant's sha256 on file objects (no file hash is stored; separate change on the upload path). New dependency: rdflib (pure Python) for the non-JSON serialisations. --- api/services/metadata_export/__init__.py | 0 api/services/metadata_export/adapter.py | 202 + .../metadata_export/contracts/README.md | 20 + .../metadata_export/contracts/crosswalk.json | 4847 +++++++++++++++++ .../metadata_export/contracts/geographies.csv | 828 +++ .../metadata_export/contracts/licenses.csv | 61 + .../metadata_export/contracts/sectors.csv | 22 + api/services/metadata_export/crosswalk.py | 206 + api/services/metadata_export/exporter.py | 127 + api/services/metadata_export/formats.py | 33 + api/services/metadata_export/vocabularies.py | 107 + api/urls.py | 11 + api/views/metadata_export.py | 64 + requirements.txt | 1 + 14 files changed, 6529 insertions(+) create mode 100644 api/services/metadata_export/__init__.py create mode 100644 api/services/metadata_export/adapter.py create mode 100644 api/services/metadata_export/contracts/README.md create mode 100644 api/services/metadata_export/contracts/crosswalk.json create mode 100644 api/services/metadata_export/contracts/geographies.csv create mode 100644 api/services/metadata_export/contracts/licenses.csv create mode 100644 api/services/metadata_export/contracts/sectors.csv create mode 100644 api/services/metadata_export/crosswalk.py create mode 100644 api/services/metadata_export/exporter.py create mode 100644 api/services/metadata_export/formats.py create mode 100644 api/services/metadata_export/vocabularies.py create mode 100644 api/views/metadata_export.py diff --git a/api/services/metadata_export/__init__.py b/api/services/metadata_export/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/api/services/metadata_export/adapter.py b/api/services/metadata_export/adapter.py new file mode 100644 index 00000000..2b7018ad --- /dev/null +++ b/api/services/metadata_export/adapter.py @@ -0,0 +1,202 @@ +"""Turn a Dataset (and its resources / source) into the flat record the +crosswalk engine reads. This is the only module in the export that knows +Django models; the engine only ever sees plain dicts keyed by the contract's +``dataspace_field`` names. Names that a standard wants as identifiers +(licence, sector, geography, language, access rights) leave here already +resolved to URIs. +""" + +from __future__ import annotations + +from typing import Any, Dict, List, Optional + +from django.conf import settings + +from api.models import Dataset, Resource +from api.services import metadata_mapping +from api.services.metadata_export import vocabularies as vocab +from api.utils.enums import DataType + +EMPTY = (None, "", [], {}) + + +def _public_base() -> str: + return getattr(settings, "PUBLIC_SITE_URL", "https://civicdataspace.in").rstrip("/") + + +def _api_base() -> str: + return getattr(settings, "PUBLIC_API_URL", "https://api.civicdataspace.in").rstrip("/") + + +def _iso(dt: Any) -> Optional[str]: + return dt.date().isoformat() if dt else None + + +def _is_url(value: Any) -> bool: + return isinstance(value, str) and value.startswith(("http://", "https://")) + + +# DCAT-AP mandates the EU Access Right authority list for dcterms:accessRights. +ACCESS_RIGHTS_URI = { + "PUBLIC": "http://publications.europa.eu/resource/authority/access-right/PUBLIC", + "RESTRICTED": "http://publications.europa.eu/resource/authority/access-right/RESTRICTED", + "PRIVATE": "http://publications.europa.eu/resource/authority/access-right/NON_PUBLIC", +} + +# Languages are ISO 639-1 codes; the Library of Congress scheme gives each a URI. +LANGUAGE_SCHEME = "http://id.loc.gov/vocabulary/iso639-1/" + +# Our format labels -> IANA media types. DCAT-AP wants dcat:mediaType to be an +# IANA IRI and Croissant wants encodingFormat to be a MIME string; both are +# derived from this. Unknown labels pass through lower-cased so nothing is lost. +MEDIA_TYPES = { + "CSV": "text/csv", + "TSV": "text/tab-separated-values", + "TXT": "text/plain", + "JSON": "application/json", + "GEOJSON": "application/geo+json", + "XML": "application/xml", + "PDF": "application/pdf", + "XLSX": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", + "XLS": "application/vnd.ms-excel", + "ODS": "application/vnd.oasis.opendocument.spreadsheet", + "PARQUET": "application/vnd.apache.parquet", + "ZIP": "application/zip", + "HTML": "text/html", +} +IANA_BASE = "https://www.iana.org/assignments/media-types/" + + +def media_type(label: Optional[str]) -> Optional[str]: + if not label: + return None + key = str(label).strip().upper().lstrip(".") + return MEDIA_TYPES.get(key) or (label if "/" in str(label) else str(label).lower()) + + +def _dataset_type_concept(value: Optional[str]) -> Optional[dict]: + """Our own type vocabulary (DATA / PROMPT), identified under the platform namespace.""" + if not value: + return None + return vocab.concept(value, f"{_public_base()}/id/dataset-type/{str(value).lower()}") + + +def _agent( + name: Optional[str], uri: Optional[str] = None, kind: str = "foaf:Organization" +) -> Optional[dict]: + return {"name": name, "uri": uri, "@type": kind} if name else None + + +def _languages(codes: Any) -> List[dict]: + out: List[dict] = [] + for code in codes or []: + code = str(code).strip().lower() + if code: + out.append(vocab.concept(code, LANGUAGE_SCHEME + code)) + return out + + +def resource_to_record(resource: Resource) -> Dict[str, Any]: + file_details = getattr(resource, "resourcefiledetails", None) + rec: Dict[str, Any] = { + "name": resource.name, + "description": resource.description or None, + "format": media_type(file_details.format if file_details else None), + "size": (file_details.size if file_details else None) or None, + "sha256": None, # not stored yet; see the file-hash follow-up + "download_url": None, + "access_url": None, + "columns": [ + {"name": s.field_name, "type": s.format, "description": s.description or None} + for s in resource.resourceschema_set.all() + ], + "_id": str(resource.id), + "_kind": "link" if resource.type == DataType.EXTERNAL else "file", + } + if resource.type == DataType.EXTERNAL: + # Link-only: the platform page gives *access*; there is no direct file. + rec["access_url"] = resource.url or None + rec["format"] = rec["format"] or "text/html" + elif file_details and file_details.file: + rec["download_url"] = f"{_api_base()}/api/download/resource/{resource.id}" + rec["access_url"] = rec["download_url"] + return rec + + +def dataset_to_record(dataset: Dataset) -> Dict[str, Any]: + source = getattr(dataset, "source", None) + org = dataset.organization + user = dataset.user + landing = f"{_public_base()}/datasets/{dataset.slug}" + + sectors = [vocab.concept(s.name, vocab.sector_uri(s.name)) for s in dataset.sectors.all()] + geographies = [ + vocab.concept(g.name, vocab.geography_uri(g.name, getattr(g, "type", None))) + for g in dataset.geographies.all() + ] + license_value = dataset.license or None + license_uri = vocab.license_uri(license_value) or ( + vocab.license_uri(source.source_license) if source and source.source_license else None + ) + + # Who made the data. For an imported dataset that is the platform author, + # not the person who clicked import; our organisation stays the publisher. + creator = _agent((user.get_full_name() or user.username) if user else None, kind="foaf:Person") + if source and source.source_author: + creator = _agent(source.source_author, kind="foaf:Agent") + + record: Dict[str, Any] = { + "id": str(dataset.id), + "slug": dataset.slug, + "title": dataset.title, + "description": dataset.description or None, + "tags": [t.value for t in dataset.tags.all()], + "sectors": sectors, + "geographies": geographies, + "license": license_uri or license_value, + "organization": _agent(org.name if org else None), + "user": creator, + # when the record appeared (here, or on the platform it came from) + "issued": ( + _iso(source.source_created_at) + if source and source.source_created_at + else _iso(dataset.created) + ), + "modified": _iso(dataset.modified), + # when the data itself came into being: only a definition can say + "created": None, + "datasetType": _dataset_type_concept(dataset.dataset_type), + "accessType": vocab.concept( + dataset.access_type, ACCESS_RIGHTS_URI.get(str(dataset.access_type)) + ), + "status": dataset.status, + "resources": [resource_to_record(r) for r in dataset.resources.all()], + "landing_page": landing, + "in_catalog": ( + f"{_public_base()}/dataspaces/{dataset.dataspace.slug}" + if dataset.dataspace_id and getattr(dataset.dataspace, "slug", None) + else None + ), + # provenance and extras, filled for imports; definitions may fill the rest + "source": source.source_url if source and source.source_url else None, + "homepage": source.source_homepage if source and source.source_homepage else None, + "version": source.revision if source and source.revision else None, + "citation": source.citation if source and source.citation else None, + "language": _languages(source.languages) if source and source.languages else [], + } + + # Definition-backed values (admin-defined fields the publisher filled in), + # keyed by crosswalk concept. Core columns always win; a definition only + # supplies what the model has no column for. Unplaceable definitions are + # carried for the report. + defined, unmapped = metadata_mapping.definition_values(dataset) + for concept, value in defined.items(): + if record.get(concept) not in EMPTY: + continue + if concept == "language": + value = _languages(value) + elif concept in ("source", "homepage") and not _is_url(value): + continue + record[concept] = value + record["_unmapped_definitions"] = unmapped + return record diff --git a/api/services/metadata_export/contracts/README.md b/api/services/metadata_export/contracts/README.md new file mode 100644 index 00000000..298ea04f --- /dev/null +++ b/api/services/metadata_export/contracts/README.md @@ -0,0 +1,20 @@ +# Metadata contract + +These files are owned by DataSpaceBackend and are the source of truth for the +metadata export. Edit them here. + +- `crosswalk.json` — the mapping: one entry per concept, and for each standard + the property it becomes, its value type and obligation, plus the concepts the + standard cannot carry (`gaps`). The engine in `../crosswalk.py` reads it and + knows nothing about any standard by name. +- `licenses.csv`, `sectors.csv`, `geographies.csv` — allowed values with URIs. + `alt_codes` carries `dataspace_enum=` for licences, which is + how a platform value finds its row. Geographies cover regions, states, union + territories and districts. + +Changing a mapping decision means editing `crosswalk.json` in a PR, like any +other code. Adding a standard means adding a block under `standards`. + +Origin: the first version was copied from CivicDataLab/DataSpace-data-ecosystem +at commit 83e577752784 (Sep 2026) and has been edited here since. There is no +runtime or build-time link to that repository. diff --git a/api/services/metadata_export/contracts/crosswalk.json b/api/services/metadata_export/contracts/crosswalk.json new file mode 100644 index 00000000..f27369b7 --- /dev/null +++ b/api/services/metadata_export/contracts/crosswalk.json @@ -0,0 +1,4847 @@ +{ + "version": 2, + "value_types": { + "literal": "A plain string.", + "langstring": "A string that may carry a language tag.", + "date": "ISO 8601 date (xsd:date).", + "datetime": "ISO 8601 datetime (xsd:dateTime).", + "uri": "An absolute URI, serialised as a node reference not a string.", + "number": "A numeric literal.", + "bytes": "A non-negative integer count of bytes.", + "duration": "ISO 8601 duration (xsd:duration).", + "media_type": "An IANA media type, or a format token where none exists.", + "agent": "A person or organisation; a node with at least a name.", + "concept": "A term from a controlled scheme; prefer its URI.", + "location": "A place; prefer a resolvable URI over a name string.", + "period": "A time interval, expressed as start and end.", + "frequency": "A term from the Dublin Core Frequency vocabulary.", + "checksum": "A hash digest, with its algorithm.", + "vcard": "A contact, serialised as a vCard node.", + "structured": "A nested object whose shape the standard defines." + }, + "nodes": [ + "contact_point", + "dataset", + "distribution", + "period", + "record_set" + ], + "directions": { + "both": "Read on import, write on export.", + "export_only": "Platform-derived. Emit it; never let an import overwrite it.", + "import_only": "Accept from a source; the platform does not re-emit it.", + "none": "Platform-internal. Not metadata; ignore in both directions." + }, + "controlled_vocabularies": { + "license": { + "path": "licenses.csv", + "key_field": "key", + "uri_field": "uri" + }, + "sector": { + "path": "sectors.csv", + "key_field": "key", + "uri_field": "uri" + }, + "geography": { + "path": "geographies.csv", + "key_field": "key", + "uri_field": "uri" + }, + "language": { + "scheme": "http://id.loc.gov/vocabulary/iso639-1/", + "notes": "ISO 639-1 code appended to the scheme" + } + }, + "concepts": { + "identifier": { + "label": "Identifier", + "definition": "A unique identifier for the dataset (platform ID, DOI, catalog UUID).", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "id", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "export_only", + "notes": "Croissant uses @id for this. On import the incoming identifier is kept in source_identifier, never written over the platform's own ID." + }, + "source_identifier": { + "label": "Source identifier", + "definition": "The identifier this dataset carried on the platform it was imported from.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Not in the source workbook. Added because round-tripping a federated record is impossible without somewhere to keep its origin ID." + }, + "slug": { + "label": "Slug", + "definition": "URL-safe short name for the dataset on the platform.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "slug", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "export_only", + "notes": "No standard equivalent; it is the path component of landing_page." + }, + "title": { + "label": "Title", + "definition": "A name given to the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "title", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "alternative_title": { + "label": "Alternative title", + "definition": "An alternative name for the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "description": { + "label": "Description", + "definition": "A free-text account of the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "description", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "abstract": { + "label": "Abstract", + "definition": "A summary of the dataset, shorter than the description.", + "node": "dataset", + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "keyword": { + "label": "Keywords", + "definition": "Free-text tags describing the dataset, used for search.", + "node": "dataset", + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": "tags", + "dataspace_input_type": "semi_controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "theme": { + "label": "Theme / Sector", + "definition": "The category the dataset belongs to, from a controlled scheme.", + "node": "dataset", + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": "sectors", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": "sector", + "direction": "both", + "notes": "Partial match. CDL sectors are the platform's own scheme; export maps them onto EU data themes, which collapses several sectors onto one theme." + }, + "dataset_type": { + "label": "Dataset type", + "definition": "The nature or genre of the dataset.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "datasetType", + "dataspace_input_type": "controlled", + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The platform's set of types is undefined in the source workbook." + }, + "language": { + "label": "Language", + "definition": "Language of the dataset, as an ISO 639 code or URI.", + "node": "dataset", + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": "language", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "Requires normalisation to ISO codes/URIs on import." + }, + "publisher": { + "label": "Publisher", + "definition": "The entity responsible for making the dataset available.", + "node": "dataset", + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "organization", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The workbook maps organization to dct:publisher but Croissant creator, and user to dct:creator but Croissant publisher. That inversion is preserved here as recorded; see the open question in the README." + }, + "creator": { + "label": "Creator", + "definition": "The entity responsible for producing the dataset.", + "node": "dataset", + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "dataspace_field": "user", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "qualified_attribution": { + "label": "Qualified attribution", + "definition": "A role-qualified link to an agent, for multiple contributors.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Needed when a dataset has several publishers with distinct roles." + }, + "contact_point": { + "label": "Contact point", + "definition": "Contact information for the dataset, as a vCard.", + "node": "dataset", + "value_type": "vcard", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Sub-fields are the contact_* concepts on the contact_point node." + }, + "license": { + "label": "License", + "definition": "The legal document under which the dataset is made available.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "license", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": "license", + "direction": "both", + "notes": "Must serialise as a resolvable LicenseDocument URI, not the enum label. Resolve through the license vocabulary before emitting." + }, + "rights": { + "label": "Rights statement", + "definition": "A statement about rights held in and over the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "access_rights": { + "label": "Access rights", + "definition": "Whether the dataset is public, restricted, or non-public.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "accessType", + "dataspace_input_type": "controlled", + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The workbook calls accessType unclear and marks it unmappable. It lines up with dcterms:accessRights if it means public/restricted; confirm against the platform before relying on this binding." + }, + "issued": { + "label": "Date issued", + "definition": "The date the dataset was first published.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "issued", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The platform's \"created\" is mapped to issued, not dcterms:created. The workbook flags that it is unclear whether this is the original creation date or the upload date." + }, + "created": { + "label": "Date created", + "definition": "The date the dataset itself was created, before publication.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "created", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "modified": { + "label": "Date modified", + "definition": "The most recent date the dataset was changed.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "modified", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "Should be driven by the platform's versioning, per the workbook." + }, + "accrual_periodicity": { + "label": "Update frequency", + "definition": "How often the dataset is updated.", + "node": "dataset", + "value_type": "frequency", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Use the Dublin Core Frequency vocabulary." + }, + "accrual_method": { + "label": "Accrual method", + "definition": "How the dataset is added to.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "version": { + "label": "Version", + "definition": "The version designator of the dataset.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "version", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "DCAT has no single version field; the workbook records the combination of issued + modified + dcterms:hasVersion as the partial match." + }, + "spatial_coverage": { + "label": "Spatial coverage", + "definition": "The geographic region the dataset covers.", + "node": "dataset", + "value_type": "location", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": "geographies", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": "geography", + "direction": "both", + "notes": "Must serialise as a resolvable place URI. The platform stores name strings today; resolve through the geography vocabulary first." + }, + "spatial_resolution": { + "label": "Spatial resolution", + "definition": "Minimum spatial separation resolvable in the dataset.", + "node": "dataset", + "value_type": "number", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT expects metres. The registry sheet uses admin level (district, village), which needs a conversion or a separate literal field." + }, + "temporal_coverage_start": { + "label": "Temporal coverage (start)", + "definition": "Start of the period the dataset covers.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "temporal_coverage_start", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "Serialises inside a dcterms:temporal PeriodOfTime node." + }, + "temporal_coverage_end": { + "label": "Temporal coverage (end)", + "definition": "End of the period the dataset covers.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "temporal_coverage_end", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "temporal_resolution": { + "label": "Temporal resolution", + "definition": "Minimum time period resolvable in the dataset.", + "node": "dataset", + "value_type": "duration", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "landing_page": { + "label": "Landing page", + "definition": "A web page giving access to the dataset and its publisher.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "landing_page", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "export_only", + "notes": "Constructed from slug at export time." + }, + "homepage": { + "label": "Homepage", + "definition": "The publisher's home page for the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "homepage", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "source": { + "label": "Source", + "definition": "A related resource the dataset is derived from.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": "source", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "depiction": { + "label": "Depiction", + "definition": "An image representing the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "in_catalog": { + "label": "In catalog", + "definition": "The catalog the dataset is listed in.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "in_catalog", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "export_only", + "notes": null + }, + "in_series": { + "label": "In series", + "definition": "The dataset series this dataset belongs to.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "conforms_to": { + "label": "Conforms to", + "definition": "An established standard or schema the dataset conforms to.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Unbound. Definition values export by concept through metadata_mapping.py." + }, + "relation": { + "label": "Related resource", + "definition": "A resource with an unspecified relationship to the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "has_part": { + "label": "Has part", + "definition": "A resource included either physically or logically in the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "is_part_of": { + "label": "Is part of", + "definition": "A resource the dataset is included in.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "is_referenced_by": { + "label": "Is referenced by", + "definition": "A resource that references the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "replaces": { + "label": "Replaces", + "definition": "A resource the dataset supplants.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "is_replaced_by": { + "label": "Is replaced by", + "definition": "A resource that supplants the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "is_version_of": { + "label": "Is version of", + "definition": "A resource the dataset is a version of.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "has_version": { + "label": "Has version", + "definition": "A version of the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "provenance": { + "label": "Provenance", + "definition": "A statement of changes in ownership and custody of the dataset.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT needs the external PROV-O ontology for this." + }, + "jurisdiction_level": { + "label": "Jurisdiction level", + "definition": "The administrative level the dataset's authority operates at.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT-IN, not yet a stable published spec." + }, + "applicable_legislation": { + "label": "Applicable legislation", + "definition": "The legislation mandating the dataset's creation or publication.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT-IN." + }, + "hvd_category": { + "label": "High-value dataset category", + "definition": "The high-value dataset category the dataset falls in.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT-IN." + }, + "note": { + "label": "Note", + "definition": "A free-text note about the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT-IN." + }, + "distribution": { + "label": "Distribution", + "definition": "An accessible form of the dataset — a downloadable file or an API.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "dataspace_field": "resources", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The link from dataset to its distribution nodes." + }, + "access_url": { + "label": "Access URL", + "definition": "A URL giving access to the distribution.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "access_url", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "DCAT requires this on every Distribution. Where the platform has only a download URL, use it for both." + }, + "download_url": { + "label": "Download URL", + "definition": "A direct link to the downloadable file.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "download_url", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "media_type": { + "label": "Format / media type", + "definition": "The file format of the distribution.", + "node": "distribution", + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "format", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The platform's dataset-level \"formats\" is the set of its resources' formats, derived rather than entered." + }, + "byte_size": { + "label": "Byte size", + "definition": "The size of the distribution in bytes.", + "node": "distribution", + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "size", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "checksum": { + "label": "Checksum", + "definition": "A hash of the distribution's contents, for integrity checking.", + "node": "distribution", + "value_type": "checksum", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "sha256", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "Croissant requires sha256 on a FileObject; DCAT has no such property." + }, + "distribution_title": { + "label": "Distribution title", + "definition": "A name given to the distribution.", + "node": "distribution", + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "name", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "endpoint_url": { + "label": "Endpoint URL", + "definition": "The root location of an API-based distribution.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "endpoint_description": { + "label": "Endpoint description", + "definition": "A description of the services available at the endpoint.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "file_set": { + "label": "File set", + "definition": "A group of files described by a pattern rather than individually.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT has no grouping semantics; the nearest is many Distributions." + }, + "containment": { + "label": "Containment", + "definition": "Which file or archive a file is contained in.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT cannot model file hierarchy." + }, + "record_set": { + "label": "Record set", + "definition": "A table within the dataset — its columns and their semantics.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "The main DCAT gap. This is the concept that makes a dataset ML-readable, and it is also what schemas/data_dictionary_template.csv captures." + }, + "field": { + "label": "Field", + "definition": "One column of a record set, with a name, description and type.", + "node": "record_set", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "No column-level metadata exists in DCAT." + }, + "data_type": { + "label": "Data type", + "definition": "The semantic type of a field (including ML types like BoundingBox).", + "node": "record_set", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "transform": { + "label": "Transform", + "definition": "A transformation applied to source data to produce a field.", + "node": "record_set", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Expressible in DCAT only through external PROV-O." + }, + "annotation": { + "label": "Annotation", + "definition": "A structured ML annotation over the data.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_name": { + "label": "Contact name", + "definition": "Formatted name of the contact.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_title": { + "label": "Contact title", + "definition": "Job title of the contact.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_role": { + "label": "Contact role", + "definition": "Role the contact plays for this dataset.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_organization": { + "label": "Contact organisation", + "definition": "Organisation the contact belongs to.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_email": { + "label": "Contact email", + "definition": "Email address of the contact.", + "node": "contact_point", + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Serialises as a mailto: URI." + }, + "contact_address": { + "label": "Contact address", + "definition": "Postal address of the contact.", + "node": "contact_point", + "value_type": "structured", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Nests the contact_street_address / locality / postal_code concepts." + }, + "contact_street_address": { + "label": "Contact street address", + "definition": "Street address of the contact.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_locality": { + "label": "Contact locality", + "definition": "Town or city of the contact's address.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_postal_code": { + "label": "Contact postal code", + "definition": "Postal code of the contact's address.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "status": { + "label": "Status", + "definition": "Publication workflow state of the dataset on the platform.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "status", + "dataspace_input_type": "automated", + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Croissant has a status concept; the workbook records no DCAT equivalent and marks it not user-facing." + }, + "download_count": { + "label": "Download count", + "definition": "Number of times the dataset has been downloaded.", + "node": "dataset", + "value_type": "number", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "downloadCount", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Usage statistic, not metadata. No DCAT equivalent." + }, + "similar_datasets": { + "label": "Similar datasets", + "definition": "Datasets the platform computes as related.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": "similarDatasets", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Platform-level and dynamic; the workbook says it need not be metadata." + }, + "is_individual_dataset": { + "label": "Is individual dataset", + "definition": "Whether the record is a standalone dataset or part of a collection.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "isIndividualDataset", + "dataspace_input_type": "automated", + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "The workbook suspects this links to catalogs. If confirmed, it belongs with in_catalog / in_series rather than here." + }, + "dataspace": { + "label": "Dataspace", + "definition": "Unclear platform field.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "dataspace", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Marked unclear and unmappable in the source workbook." + }, + "prompt_metadata": { + "label": "Prompt metadata", + "definition": "Unclear platform field.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "promptMetadata", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Undefined in the source workbook." + }, + "citation": { + "label": "Citation", + "definition": "How to cite the dataset, as free text (BibTeX or a sentence).", + "dataspace_field": "citation", + "dataspace_input_type": "text", + "value_type": "literal", + "obligation": "optional", + "direction": "both", + "repeatable": false, + "node": "dataset", + "controlled_vocabulary": null, + "visible_on_dataspace": true, + "notes": "Croissant citeAs; Dublin Core / DCAT dcterms:bibliographicCitation." + } + }, + "platform_fields": { + "id": "identifier", + "slug": "slug", + "title": "title", + "description": "description", + "tags": "keyword", + "sectors": "theme", + "datasetType": "dataset_type", + "organization": "publisher", + "user": "creator", + "license": "license", + "accessType": "access_rights", + "created": "created", + "modified": "modified", + "geographies": "spatial_coverage", + "resources": "distribution", + "download_url": "download_url", + "format": "media_type", + "size": "byte_size", + "name": "distribution_title", + "status": "status", + "downloadCount": "download_count", + "similarDatasets": "similar_datasets", + "isIndividualDataset": "is_individual_dataset", + "dataspace": "dataspace", + "promptMetadata": "prompt_metadata", + "issued": "issued", + "source": "source", + "homepage": "homepage", + "language": "language", + "version": "version", + "landing_page": "landing_page", + "in_catalog": "in_catalog", + "access_url": "access_url", + "sha256": "checksum", + "citation": "citation", + "temporal_coverage_start": "temporal_coverage_start", + "temporal_coverage_end": "temporal_coverage_end" + }, + "standards": { + "croissant": { + "name": "Croissant (MLCommons ML-dataset metadata format)", + "version": "1.0", + "url": "https://mlcommons.org/croissant/", + "serialisation": "json-ld", + "uri_style": "string", + "context": { + "@vocab": "https://schema.org/", + "cr": "http://mlcommons.org/croissant/", + "sc": "https://schema.org/" + }, + "node_types": { + "dataset": "Dataset", + "distribution": "cr:FileObject", + "record_set": "cr:RecordSet" + }, + "export": [ + { + "concept": "identifier", + "property": "identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "partial", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": "Croissant also uses @id as the node identity; emit both." + }, + { + "concept": "title", + "property": "name", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "description", + "property": "description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "keyword", + "property": "keywords", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "Free text only; no controlled-scheme distinction." + }, + { + "concept": "publisher", + "property": "publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": "The source workbook maps the platform's organization to dct:publisher but to Croissant creator, and user the other way round. Preserved as recorded - see the open question in the README before relying on it." + }, + { + "concept": "creator", + "property": "creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": "See publisher; the inversion is as recorded in the workbook." + }, + { + "concept": "license", + "property": "license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": "A license URL, same steward URI as DCAT." + }, + { + "concept": "issued", + "property": "datePublished", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "modified", + "property": "dateModified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "version", + "property": "version", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "version", + "controlled_vocabulary": null, + "notes": "Croissant has a dedicated version field, which is the one place it is stricter than DCAT." + }, + { + "concept": "spatial_coverage", + "property": "spatialCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": "schema.org Place, not a dcterms:Location. Accepts a name string, so an export can succeed while carrying no resolvable URI - check the value." + }, + { + "concept": "temporal_coverage_start", + "property": "temporalCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "temporal_coverage_start", + "controlled_vocabulary": null, + "notes": "One ISO 8601 interval string (\"2019-01-01/2023-12-31\"), not two fields. Both start and end serialise into it." + }, + { + "concept": "temporal_coverage_end", + "property": "temporalCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "temporal_coverage_end", + "controlled_vocabulary": null, + "notes": "See temporal_coverage_start; both concepts share one property." + }, + { + "concept": "landing_page", + "property": "url", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "landing_page", + "controlled_vocabulary": null, + "notes": "Requires interpretation of intent - url is not specifically a landing page." + }, + { + "concept": "distribution", + "property": "distribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "resources", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "download_url", + "property": "contentUrl", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "media_type", + "property": "encodingFormat", + "node": "distribution", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "byte_size", + "property": "contentSize", + "node": "distribution", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "checksum", + "property": "sha256", + "node": "distribution", + "parent_property": null, + "value_type": "checksum", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "sha256", + "controlled_vocabulary": null, + "notes": "Required on a FileObject. DCAT has nothing equivalent." + }, + { + "concept": "distribution_title", + "property": "name", + "node": "distribution", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "name", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "language", + "property": "inLanguage", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "language", + "controlled_vocabulary": "language", + "notes": null + }, + { + "concept": "created", + "property": "dateCreated", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "created", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "source", + "property": "sameAs", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "source", + "controlled_vocabulary": null, + "notes": "For imported datasets: the same dataset on the source platform." + }, + { + "concept": "citation", + "property": "citeAs", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "citation", + "controlled_vocabulary": null, + "notes": null + } + ], + "import": { + "dataset": { + "identifier": [ + { + "concept": "identifier", + "property": "identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "partial", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": "Croissant also uses @id as the node identity; emit both." + } + ], + "name": [ + { + "concept": "title", + "property": "name", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + } + ], + "alternateName": [ + { + "concept": "alternative_title", + "property": "alternateName", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "description": [ + { + "concept": "description", + "property": "description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + } + ], + "keywords": [ + { + "concept": "keyword", + "property": "keywords", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "Free text only; no controlled-scheme distinction." + } + ], + "inLanguage": [ + { + "concept": "language", + "property": "inLanguage", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "creator": [ + { + "concept": "publisher", + "property": "creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": "The source workbook maps the platform's organization to dct:publisher but to Croissant creator, and user the other way round. Preserved as recorded - see the open question in the README before relying on it." + } + ], + "publisher": [ + { + "concept": "creator", + "property": "publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": "See publisher; the inversion is as recorded in the workbook." + } + ], + "license": [ + { + "concept": "license", + "property": "license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": "A license URL, same steward URI as DCAT." + } + ], + "datePublished": [ + { + "concept": "issued", + "property": "datePublished", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + } + ], + "dateCreated": [ + { + "concept": "created", + "property": "dateCreated", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dateModified": [ + { + "concept": "modified", + "property": "dateModified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + } + ], + "version": [ + { + "concept": "version", + "property": "version", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Croissant has a dedicated version field, which is the one place it is stricter than DCAT." + } + ], + "spatialCoverage": [ + { + "concept": "spatial_coverage", + "property": "spatialCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "mandatory", + "match": "partial", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": "schema.org Place, not a dcterms:Location. Accepts a name string, so an export can succeed while carrying no resolvable URI - check the value." + } + ], + "temporalCoverage": [ + { + "concept": "temporal_coverage_start", + "property": "temporalCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "One ISO 8601 interval string (\"2019-01-01/2023-12-31\"), not two fields. Both start and end serialise into it." + }, + { + "concept": "temporal_coverage_end", + "property": "temporalCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "See temporal_coverage_start; both concepts share one property." + } + ], + "url": [ + { + "concept": "landing_page", + "property": "url", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Requires interpretation of intent - url is not specifically a landing page." + }, + { + "concept": "homepage", + "property": "url", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "isBasedOn": [ + { + "concept": "source", + "property": "isBasedOn", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "conformsTo": [ + { + "concept": "conforms_to", + "property": "conformsTo", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "metadata", + "controlled_vocabulary": null, + "notes": "Croissant uses this to declare the Croissant spec version itself." + } + ], + "hasPart": [ + { + "concept": "has_part", + "property": "hasPart", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "isPartOf": [ + { + "concept": "is_part_of", + "property": "isPartOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "distribution": [ + { + "concept": "distribution", + "property": "distribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "resources", + "controlled_vocabulary": null, + "notes": null + } + ], + "cr:FileSet": [ + { + "concept": "file_set", + "property": "cr:FileSet", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "A group of files given by an includes/excludes pattern." + } + ], + "cr:recordSet": [ + { + "concept": "record_set", + "property": "cr:recordSet", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "cr:annotation": [ + { + "concept": "annotation", + "property": "cr:annotation", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "distribution": { + "contentUrl": [ + { + "concept": "download_url", + "property": "contentUrl", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": null + } + ], + "encodingFormat": [ + { + "concept": "media_type", + "property": "encodingFormat", + "node": "distribution", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": null + } + ], + "contentSize": [ + { + "concept": "byte_size", + "property": "contentSize", + "node": "distribution", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": null + } + ], + "sha256": [ + { + "concept": "checksum", + "property": "sha256", + "node": "distribution", + "parent_property": null, + "value_type": "checksum", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Required on a FileObject. DCAT has nothing equivalent." + } + ], + "name": [ + { + "concept": "distribution_title", + "property": "name", + "node": "distribution", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "name", + "controlled_vocabulary": null, + "notes": null + } + ], + "containedIn": [ + { + "concept": "containment", + "property": "containedIn", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "record_set": { + "cr:field": [ + { + "concept": "field", + "property": "cr:field", + "node": "record_set", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dataType": [ + { + "concept": "data_type", + "property": "dataType", + "node": "record_set", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Includes ML types such as cr:BoundingBox." + } + ], + "cr:transform": [ + { + "concept": "transform", + "property": "cr:transform", + "node": "record_set", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + } + }, + "gaps": [ + { + "concept": "source_identifier", + "reason": "unbound", + "notes": null + }, + { + "concept": "slug", + "reason": "unbound", + "notes": null + }, + { + "concept": "abstract", + "reason": "declared gap", + "notes": null + }, + { + "concept": "theme", + "reason": "declared gap", + "notes": "No controlled theme scheme. Sectors have to go out as keywords, which loses the vocabulary binding." + }, + { + "concept": "dataset_type", + "reason": "declared gap", + "notes": null + }, + { + "concept": "qualified_attribution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_point", + "reason": "declared gap", + "notes": null + }, + { + "concept": "rights", + "reason": "declared gap", + "notes": null + }, + { + "concept": "access_rights", + "reason": "declared gap", + "notes": "Not defined in Croissant. The workbook notes dcterms:accessRights exists only on the DCAT side." + }, + { + "concept": "accrual_periodicity", + "reason": "declared gap", + "notes": null + }, + { + "concept": "accrual_method", + "reason": "declared gap", + "notes": null + }, + { + "concept": "spatial_resolution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "temporal_resolution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "depiction", + "reason": "declared gap", + "notes": null + }, + { + "concept": "in_catalog", + "reason": "declared gap", + "notes": null + }, + { + "concept": "in_series", + "reason": "declared gap", + "notes": null + }, + { + "concept": "relation", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_referenced_by", + "reason": "unbound", + "notes": null + }, + { + "concept": "replaces", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_replaced_by", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_version_of", + "reason": "unbound", + "notes": null + }, + { + "concept": "has_version", + "reason": "unbound", + "notes": null + }, + { + "concept": "provenance", + "reason": "declared gap", + "notes": "cr:transform covers field derivation, not dataset custody history." + }, + { + "concept": "jurisdiction_level", + "reason": "declared gap", + "notes": null + }, + { + "concept": "applicable_legislation", + "reason": "declared gap", + "notes": null + }, + { + "concept": "hvd_category", + "reason": "declared gap", + "notes": null + }, + { + "concept": "note", + "reason": "declared gap", + "notes": null + }, + { + "concept": "endpoint_url", + "reason": "declared gap", + "notes": null + }, + { + "concept": "endpoint_description", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_name", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_title", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_role", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_organization", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_email", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_address", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_street_address", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_locality", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_postal_code", + "reason": "declared gap", + "notes": null + }, + { + "concept": "download_count", + "reason": "unbound", + "notes": null + }, + { + "concept": "similar_datasets", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_individual_dataset", + "reason": "unbound", + "notes": null + }, + { + "concept": "dataspace", + "reason": "unbound", + "notes": null + }, + { + "concept": "prompt_metadata", + "reason": "unbound", + "notes": null + }, + { + "concept": "homepage", + "reason": "declared gap", + "notes": "Croissant `url` is the landing page; there is no separate homepage property." + } + ] + }, + "dcat": { + "name": "DCAT v3 (Data Catalog Vocabulary)", + "version": "3", + "url": "https://www.w3.org/TR/vocab-dcat-3/", + "serialisation": "json-ld", + "uri_style": "node", + "context": { + "dcat": "http://www.w3.org/ns/dcat#", + "dcterms": "http://purl.org/dc/terms/", + "foaf": "http://xmlns.com/foaf/0.1/", + "prov": "http://www.w3.org/ns/prov#", + "xsd": "http://www.w3.org/2001/XMLSchema#", + "owl": "http://www.w3.org/2002/07/owl#" + }, + "node_types": { + "dataset": "dcat:Dataset", + "distribution": "dcat:Distribution", + "contact_point": "vcard:Kind" + }, + "export": [ + { + "concept": "identifier", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "title", + "property": "dcterms:title", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "description", + "property": "dcterms:description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "keyword", + "property": "dcat:keyword", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "dcat:keyword or dcterms:subject, depending on whether the values are free text or come from a controlled scheme. The platform's tags are semi-controlled, so keyword is the safer target." + }, + { + "concept": "theme", + "property": "dcat:theme", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "sectors", + "controlled_vocabulary": "sector", + "notes": "Expects a skos:Concept from a published scheme. CDL sectors collapse onto EU data themes many-to-one; the mapping is lossy in that direction and not reversible." + }, + { + "concept": "dataset_type", + "property": "dcterms:type", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "datasetType", + "controlled_vocabulary": null, + "notes": "Dataset-level only. DCAT has no field-level typing." + }, + { + "concept": "publisher", + "property": "dcterms:publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "creator", + "property": "dcterms:creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "license", + "property": "dcterms:license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": "Mandatory at Distribution in the registry profile. Emit the steward URI from the license vocabulary, never the platform's enum label." + }, + { + "concept": "access_rights", + "property": "dcterms:accessRights", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "accessType", + "controlled_vocabulary": null, + "notes": "Croissant has no equivalent; this is DCAT-only." + }, + { + "concept": "issued", + "property": "dcterms:issued", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "modified", + "property": "dcterms:modified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "version", + "property": "owl:versionInfo", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": "version", + "controlled_vocabulary": null, + "notes": "DCAT has no dedicated version field. The workbook records the combination issued + modified + dcterms:hasVersion / owl:versionInfo. owl:versionInfo is used here because dcterms:hasVersion means \"a version of this exists over there\" - binding both concepts to it would make an incoming hasVersion impossible to route." + }, + { + "concept": "spatial_coverage", + "property": "dcterms:spatial", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": null + }, + { + "concept": "temporal_coverage_start", + "property": "dcat:startDate", + "node": "period", + "parent_property": "dcterms:temporal", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "temporal_coverage_start", + "controlled_vocabulary": null, + "notes": "Nested inside a dcterms:PeriodOfTime node." + }, + { + "concept": "temporal_coverage_end", + "property": "dcat:endDate", + "node": "period", + "parent_property": "dcterms:temporal", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "temporal_coverage_end", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "landing_page", + "property": "dcat:landingPage", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "landing_page", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "in_catalog", + "property": "dcat:inCatalog", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "in_catalog", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "distribution", + "property": "dcat:distribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "resources", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "access_url", + "property": "dcat:accessURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "access_url", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "download_url", + "property": "dcat:downloadURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "media_type", + "property": "dcat:mediaType", + "node": "distribution", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": "dcterms:format is the looser alternative the workbook also records. Use mediaType when the value is a real MIME type, format otherwise." + }, + { + "concept": "byte_size", + "property": "dcat:byteSize", + "node": "distribution", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "distribution_title", + "property": "dcterms:title", + "node": "distribution", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "name", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "language", + "property": "dcterms:language", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "language", + "controlled_vocabulary": "language", + "notes": null + }, + { + "concept": "created", + "property": "dcterms:created", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "created", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "homepage", + "property": "foaf:homepage", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "homepage", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "source", + "property": "dcterms:source", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "source", + "controlled_vocabulary": null, + "notes": "For imported datasets: the dataset page on the source platform." + }, + { + "concept": "source", + "property": "prov:wasDerivedFrom", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "source", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "citation", + "property": "dcterms:bibliographicCitation", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "citation", + "controlled_vocabulary": null, + "notes": null + } + ], + "import": { + "dataset": { + "dcterms:identifier": [ + { + "concept": "identifier", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:title": [ + { + "concept": "title", + "property": "dcterms:title", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:alternative": [ + { + "concept": "alternative_title", + "property": "dcterms:alternative", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:description": [ + { + "concept": "description", + "property": "dcterms:description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:abstract": [ + { + "concept": "abstract", + "property": "dcterms:abstract", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:keyword": [ + { + "concept": "keyword", + "property": "dcat:keyword", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "dcat:keyword or dcterms:subject, depending on whether the values are free text or come from a controlled scheme. The platform's tags are semi-controlled, so keyword is the safer target." + } + ], + "dcat:theme": [ + { + "concept": "theme", + "property": "dcat:theme", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "sectors", + "controlled_vocabulary": "sector", + "notes": "Expects a skos:Concept from a published scheme. CDL sectors collapse onto EU data themes many-to-one; the mapping is lossy in that direction and not reversible." + } + ], + "dcterms:type": [ + { + "concept": "dataset_type", + "property": "dcterms:type", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "datasetType", + "controlled_vocabulary": null, + "notes": "Dataset-level only. DCAT has no field-level typing." + } + ], + "dcterms:language": [ + { + "concept": "language", + "property": "dcterms:language", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:publisher": [ + { + "concept": "publisher", + "property": "dcterms:publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:creator": [ + { + "concept": "creator", + "property": "dcterms:creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": null + } + ], + "prov:qualifiedAttribution": [ + { + "concept": "qualified_attribution", + "property": "prov:qualifiedAttribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Requires PROV-O, which is outside DCAT proper." + } + ], + "dcat:contactPoint": [ + { + "concept": "contact_point", + "property": "dcat:contactPoint", + "node": "dataset", + "parent_property": null, + "value_type": "vcard", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Nests a vcard:Kind node built from the contact_* concepts." + } + ], + "dcterms:license": [ + { + "concept": "license", + "property": "dcterms:license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": "Mandatory at Distribution in the registry profile. Emit the steward URI from the license vocabulary, never the platform's enum label." + } + ], + "dcterms:rights": [ + { + "concept": "rights", + "property": "dcterms:rights", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accessRights": [ + { + "concept": "access_rights", + "property": "dcterms:accessRights", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "accessType", + "controlled_vocabulary": null, + "notes": "Croissant has no equivalent; this is DCAT-only." + } + ], + "dcterms:issued": [ + { + "concept": "issued", + "property": "dcterms:issued", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:created": [ + { + "concept": "created", + "property": "dcterms:created", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:modified": [ + { + "concept": "modified", + "property": "dcterms:modified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accrualPeriodicity": [ + { + "concept": "accrual_periodicity", + "property": "dcterms:accrualPeriodicity", + "node": "dataset", + "parent_property": null, + "value_type": "frequency", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accrualMethod": [ + { + "concept": "accrual_method", + "property": "dcterms:accrualMethod", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "owl:versionInfo": [ + { + "concept": "version", + "property": "owl:versionInfo", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT has no dedicated version field. The workbook records the combination issued + modified + dcterms:hasVersion / owl:versionInfo. owl:versionInfo is used here because dcterms:hasVersion means \"a version of this exists over there\" - binding both concepts to it would make an incoming hasVersion impossible to route." + } + ], + "dcterms:spatial": [ + { + "concept": "spatial_coverage", + "property": "dcterms:spatial", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": null + } + ], + "dcat:spatialResolutionInMeters": [ + { + "concept": "spatial_resolution", + "property": "dcat:spatialResolutionInMeters", + "node": "dataset", + "parent_property": null, + "value_type": "number", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT wants metres. The registry records admin level instead, which does not convert cleanly — carry it as a literal or drop it." + } + ], + "dcat:temporalResolution": [ + { + "concept": "temporal_resolution", + "property": "dcat:temporalResolution", + "node": "dataset", + "parent_property": null, + "value_type": "duration", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:landingPage": [ + { + "concept": "landing_page", + "property": "dcat:landingPage", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "foaf:homepage": [ + { + "concept": "homepage", + "property": "foaf:homepage", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:source": [ + { + "concept": "source", + "property": "dcterms:source", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "foaf:depiction": [ + { + "concept": "depiction", + "property": "foaf:depiction", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:inCatalog": [ + { + "concept": "in_catalog", + "property": "dcat:inCatalog", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:inSeries": [ + { + "concept": "in_series", + "property": "dcat:inSeries", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:conformsTo": [ + { + "concept": "conforms_to", + "property": "dcterms:conformsTo", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": "metadata", + "controlled_vocabulary": null, + "notes": "DCAT's only way to point at an external schema such as CSVW." + } + ], + "dcterms:relation": [ + { + "concept": "relation", + "property": "dcterms:relation", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:hasPart": [ + { + "concept": "has_part", + "property": "dcterms:hasPart", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isPartOf": [ + { + "concept": "is_part_of", + "property": "dcterms:isPartOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isReferencedBy": [ + { + "concept": "is_referenced_by", + "property": "dcterms:isReferencedBy", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:replaces": [ + { + "concept": "replaces", + "property": "dcterms:replaces", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isReplacedBy": [ + { + "concept": "is_replaced_by", + "property": "dcterms:isReplacedBy", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isVersionOf": [ + { + "concept": "is_version_of", + "property": "dcterms:isVersionOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:hasVersion": [ + { + "concept": "has_version", + "property": "dcterms:hasVersion", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "prov:wasGeneratedBy": [ + { + "concept": "provenance", + "property": "prov:wasGeneratedBy", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Requires PROV-O." + } + ], + "dcatin:jurisdictionLevel": [ + { + "concept": "jurisdiction_level", + "property": "dcatin:jurisdictionLevel", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT-IN, proposed. Namespace is a placeholder until it publishes." + } + ], + "dcatin:applicableLegislation": [ + { + "concept": "applicable_legislation", + "property": "dcatin:applicableLegislation", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT-IN, proposed." + } + ], + "dcatin:hvdCategory": [ + { + "concept": "hvd_category", + "property": "dcatin:hvdCategory", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT-IN, proposed." + } + ], + "dcatin:note": [ + { + "concept": "note", + "property": "dcatin:note", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT-IN, proposed." + } + ], + "dcat:distribution": [ + { + "concept": "distribution", + "property": "dcat:distribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "resources", + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "period": { + "dcat:startDate": [ + { + "concept": "temporal_coverage_start", + "property": "dcat:startDate", + "node": "period", + "parent_property": "dcterms:temporal", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Nested inside a dcterms:PeriodOfTime node." + } + ], + "dcat:endDate": [ + { + "concept": "temporal_coverage_end", + "property": "dcat:endDate", + "node": "period", + "parent_property": "dcterms:temporal", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "distribution": { + "dcat:accessURL": [ + { + "concept": "access_url", + "property": "dcat:accessURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:downloadURL": [ + { + "concept": "download_url", + "property": "dcat:downloadURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:mediaType": [ + { + "concept": "media_type", + "property": "dcat:mediaType", + "node": "distribution", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": "dcterms:format is the looser alternative the workbook also records. Use mediaType when the value is a real MIME type, format otherwise." + } + ], + "dcat:byteSize": [ + { + "concept": "byte_size", + "property": "dcat:byteSize", + "node": "distribution", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:title": [ + { + "concept": "distribution_title", + "property": "dcterms:title", + "node": "distribution", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "name", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:endpointURL": [ + { + "concept": "endpoint_url", + "property": "dcat:endpointURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:endpointDescription": [ + { + "concept": "endpoint_description", + "property": "dcat:endpointDescription", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "contact_point": { + "vcard:fn": [ + { + "concept": "contact_name", + "property": "vcard:fn", + "node": "contact_point", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:title": [ + { + "concept": "contact_title", + "property": "vcard:title", + "node": "contact_point", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:role": [ + { + "concept": "contact_role", + "property": "vcard:role", + "node": "contact_point", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:organization-name": [ + { + "concept": "contact_organization", + "property": "vcard:organization-name", + "node": "contact_point", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:hasEmail": [ + { + "concept": "contact_email", + "property": "vcard:hasEmail", + "node": "contact_point", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:hasAddress": [ + { + "concept": "contact_address", + "property": "vcard:hasAddress", + "node": "contact_point", + "parent_property": null, + "value_type": "structured", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:street-address": [ + { + "concept": "contact_street_address", + "property": "vcard:street-address", + "node": "contact_point", + "parent_property": "vcard:hasAddress", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:locality": [ + { + "concept": "contact_locality", + "property": "vcard:locality", + "node": "contact_point", + "parent_property": "vcard:hasAddress", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:postal-code": [ + { + "concept": "contact_postal_code", + "property": "vcard:postal-code", + "node": "contact_point", + "parent_property": "vcard:hasAddress", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + } + }, + "gaps": [ + { + "concept": "source_identifier", + "reason": "unbound", + "notes": null + }, + { + "concept": "slug", + "reason": "unbound", + "notes": null + }, + { + "concept": "checksum", + "reason": "declared gap", + "notes": "No integrity-verification property in DCAT. Croissant requires sha256." + }, + { + "concept": "file_set", + "reason": "declared gap", + "notes": "No file grouping or pattern support; nearest is many Distributions." + }, + { + "concept": "containment", + "reason": "declared gap", + "notes": "No file hierarchy modelling." + }, + { + "concept": "record_set", + "reason": "declared gap", + "notes": "Cannot represent tables or rows. Use conforms_to to point outward." + }, + { + "concept": "field", + "reason": "declared gap", + "notes": "No column-level metadata." + }, + { + "concept": "data_type", + "reason": "declared gap", + "notes": "No semantic or ML typing." + }, + { + "concept": "transform", + "reason": "declared gap", + "notes": "Possible only via external PROV-O." + }, + { + "concept": "annotation", + "reason": "declared gap", + "notes": "No native annotation model." + }, + { + "concept": "status", + "reason": "declared gap", + "notes": "Publication workflow state is not a DCAT concept." + }, + { + "concept": "download_count", + "reason": "unbound", + "notes": null + }, + { + "concept": "similar_datasets", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_individual_dataset", + "reason": "unbound", + "notes": null + }, + { + "concept": "dataspace", + "reason": "unbound", + "notes": null + }, + { + "concept": "prompt_metadata", + "reason": "unbound", + "notes": null + } + ] + }, + "dublin_core": { + "name": "Dublin Core Metadata Terms (DCMI)", + "version": "DCMI Metadata Terms 2020-01-20", + "url": "https://www.dublincore.org/specifications/dublin-core/dcmi-terms/", + "serialisation": "json-ld", + "uri_style": "node", + "context": { + "dcterms": "http://purl.org/dc/terms/" + }, + "node_types": { + "dataset": "dcterms:Dataset" + }, + "export": [ + { + "concept": "identifier", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "title", + "property": "dcterms:title", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "description", + "property": "dcterms:description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "keyword", + "property": "dcterms:subject", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "Dublin Core has no free-text keyword element. subject is the nearest and is meant to carry controlled terms, so round-tripping tags through it loses the distinction between a tag and a theme." + }, + { + "concept": "theme", + "property": "dcterms:subject", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "sectors", + "controlled_vocabulary": "sector", + "notes": "Shares dcterms:subject with keyword. On import from Dublin Core there is no way to tell which concept a subject value belongs to; the workbook's DCAT sheet maps Subject to both dcat:theme and dcat:keyword for the same reason. Prefer DCAT when the distinction matters." + }, + { + "concept": "dataset_type", + "property": "dcterms:type", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "datasetType", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "publisher", + "property": "dcterms:publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "creator", + "property": "dcterms:creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "license", + "property": "dcterms:license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": null + }, + { + "concept": "access_rights", + "property": "dcterms:accessRights", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "accessType", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "issued", + "property": "dcterms:issued", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "modified", + "property": "dcterms:modified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "spatial_coverage", + "property": "dcterms:spatial", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": null + }, + { + "concept": "temporal_coverage_start", + "property": "dcterms:temporal", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "temporal_coverage_start", + "controlled_vocabulary": null, + "notes": "Dublin Core has no start/end structure - temporal takes a single literal period. Serialise start and end as one DCMI Period string; parsing it back into two dates is best-effort." + }, + { + "concept": "temporal_coverage_end", + "property": "dcterms:temporal", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "temporal_coverage_end", + "controlled_vocabulary": null, + "notes": "See temporal_coverage_start; both concepts share one property." + }, + { + "concept": "download_url", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": "Dublin Core has no download link. The workbook's Identifier row maps to dcat:landingPage and foaf:homePage for this reason. Lossy either way." + }, + { + "concept": "media_type", + "property": "dcterms:format", + "node": "dataset", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": "Flattened onto the dataset, because Dublin Core has no Distribution. For a multi-file dataset this collapses to the set of its formats." + }, + { + "concept": "byte_size", + "property": "dcterms:extent", + "node": "dataset", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": "Flattened onto the dataset; a literal, not a byte count." + }, + { + "concept": "language", + "property": "dcterms:language", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "language", + "controlled_vocabulary": "language", + "notes": null + }, + { + "concept": "created", + "property": "dcterms:created", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "created", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "source", + "property": "dcterms:source", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "source", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "citation", + "property": "dcterms:bibliographicCitation", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "citation", + "controlled_vocabulary": null, + "notes": null + } + ], + "import": { + "dataset": { + "dcterms:identifier": [ + { + "concept": "identifier", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "download_url", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": "Dublin Core has no download link. The workbook's Identifier row maps to dcat:landingPage and foaf:homePage for this reason. Lossy either way." + } + ], + "dcterms:title": [ + { + "concept": "title", + "property": "dcterms:title", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:alternative": [ + { + "concept": "alternative_title", + "property": "dcterms:alternative", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:description": [ + { + "concept": "description", + "property": "dcterms:description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:abstract": [ + { + "concept": "abstract", + "property": "dcterms:abstract", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:subject": [ + { + "concept": "keyword", + "property": "dcterms:subject", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "Dublin Core has no free-text keyword element. subject is the nearest and is meant to carry controlled terms, so round-tripping tags through it loses the distinction between a tag and a theme." + }, + { + "concept": "theme", + "property": "dcterms:subject", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "sectors", + "controlled_vocabulary": "sector", + "notes": "Shares dcterms:subject with keyword. On import from Dublin Core there is no way to tell which concept a subject value belongs to; the workbook's DCAT sheet maps Subject to both dcat:theme and dcat:keyword for the same reason. Prefer DCAT when the distinction matters." + } + ], + "dcterms:type": [ + { + "concept": "dataset_type", + "property": "dcterms:type", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "datasetType", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:language": [ + { + "concept": "language", + "property": "dcterms:language", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:publisher": [ + { + "concept": "publisher", + "property": "dcterms:publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:creator": [ + { + "concept": "creator", + "property": "dcterms:creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:license": [ + { + "concept": "license", + "property": "dcterms:license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": null + } + ], + "dcterms:rights": [ + { + "concept": "rights", + "property": "dcterms:rights", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accessRights": [ + { + "concept": "access_rights", + "property": "dcterms:accessRights", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "accessType", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:issued": [ + { + "concept": "issued", + "property": "dcterms:issued", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:created": [ + { + "concept": "created", + "property": "dcterms:created", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:modified": [ + { + "concept": "modified", + "property": "dcterms:modified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accrualPeriodicity": [ + { + "concept": "accrual_periodicity", + "property": "dcterms:accrualPeriodicity", + "node": "dataset", + "parent_property": null, + "value_type": "frequency", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accrualMethod": [ + { + "concept": "accrual_method", + "property": "dcterms:accrualMethod", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:spatial": [ + { + "concept": "spatial_coverage", + "property": "dcterms:spatial", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": null + } + ], + "dcterms:temporal": [ + { + "concept": "temporal_coverage_start", + "property": "dcterms:temporal", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Dublin Core has no start/end structure - temporal takes a single literal period. Serialise start and end as one DCMI Period string; parsing it back into two dates is best-effort." + }, + { + "concept": "temporal_coverage_end", + "property": "dcterms:temporal", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "See temporal_coverage_start; both concepts share one property." + } + ], + "dcterms:source": [ + { + "concept": "source", + "property": "dcterms:source", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:conformsTo": [ + { + "concept": "conforms_to", + "property": "dcterms:conformsTo", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "metadata", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:relation": [ + { + "concept": "relation", + "property": "dcterms:relation", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:hasPart": [ + { + "concept": "has_part", + "property": "dcterms:hasPart", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isPartOf": [ + { + "concept": "is_part_of", + "property": "dcterms:isPartOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isReferencedBy": [ + { + "concept": "is_referenced_by", + "property": "dcterms:isReferencedBy", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:replaces": [ + { + "concept": "replaces", + "property": "dcterms:replaces", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isReplacedBy": [ + { + "concept": "is_replaced_by", + "property": "dcterms:isReplacedBy", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isVersionOf": [ + { + "concept": "is_version_of", + "property": "dcterms:isVersionOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:hasVersion": [ + { + "concept": "has_version", + "property": "dcterms:hasVersion", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:provenance": [ + { + "concept": "provenance", + "property": "dcterms:provenance", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "A free-text statement, not the structured PROV-O graph." + } + ], + "dcterms:format": [ + { + "concept": "media_type", + "property": "dcterms:format", + "node": "dataset", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": "Flattened onto the dataset, because Dublin Core has no Distribution. For a multi-file dataset this collapses to the set of its formats." + } + ], + "dcterms:extent": [ + { + "concept": "byte_size", + "property": "dcterms:extent", + "node": "dataset", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": "Flattened onto the dataset; a literal, not a byte count." + } + ] + } + }, + "gaps": [ + { + "concept": "source_identifier", + "reason": "unbound", + "notes": null + }, + { + "concept": "slug", + "reason": "unbound", + "notes": null + }, + { + "concept": "qualified_attribution", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_point", + "reason": "declared gap", + "notes": "No contact structure; publisher is the only agent-ish element." + }, + { + "concept": "version", + "reason": "declared gap", + "notes": "No version element. dcterms:hasVersion points at a different resource that is a version of this one, which is has_version, not this concept." + }, + { + "concept": "spatial_resolution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "temporal_resolution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "landing_page", + "reason": "declared gap", + "notes": null + }, + { + "concept": "homepage", + "reason": "unbound", + "notes": null + }, + { + "concept": "depiction", + "reason": "unbound", + "notes": null + }, + { + "concept": "in_catalog", + "reason": "unbound", + "notes": null + }, + { + "concept": "in_series", + "reason": "unbound", + "notes": null + }, + { + "concept": "jurisdiction_level", + "reason": "unbound", + "notes": null + }, + { + "concept": "applicable_legislation", + "reason": "unbound", + "notes": null + }, + { + "concept": "hvd_category", + "reason": "unbound", + "notes": null + }, + { + "concept": "note", + "reason": "unbound", + "notes": null + }, + { + "concept": "distribution", + "reason": "declared gap", + "notes": "No Distribution resource. Every file-level fact flattens or is lost." + }, + { + "concept": "access_url", + "reason": "declared gap", + "notes": null + }, + { + "concept": "checksum", + "reason": "declared gap", + "notes": null + }, + { + "concept": "distribution_title", + "reason": "unbound", + "notes": null + }, + { + "concept": "endpoint_url", + "reason": "unbound", + "notes": null + }, + { + "concept": "endpoint_description", + "reason": "unbound", + "notes": null + }, + { + "concept": "file_set", + "reason": "declared gap", + "notes": null + }, + { + "concept": "containment", + "reason": "declared gap", + "notes": null + }, + { + "concept": "record_set", + "reason": "declared gap", + "notes": null + }, + { + "concept": "field", + "reason": "declared gap", + "notes": null + }, + { + "concept": "data_type", + "reason": "declared gap", + "notes": null + }, + { + "concept": "transform", + "reason": "declared gap", + "notes": null + }, + { + "concept": "annotation", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_name", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_title", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_role", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_organization", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_email", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_address", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_street_address", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_locality", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_postal_code", + "reason": "declared gap", + "notes": null + }, + { + "concept": "status", + "reason": "declared gap", + "notes": null + }, + { + "concept": "download_count", + "reason": "unbound", + "notes": null + }, + { + "concept": "similar_datasets", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_individual_dataset", + "reason": "unbound", + "notes": null + }, + { + "concept": "dataspace", + "reason": "unbound", + "notes": null + }, + { + "concept": "prompt_metadata", + "reason": "unbound", + "notes": null + }, + { + "concept": "homepage", + "reason": "declared gap", + "notes": "No homepage term in DCMI Metadata Terms." + } + ] + } + }, + "owner": "DataSpaceBackend (api/services/metadata_export/contracts)", + "origin": "Started as a copy of CivicDataLab/DataSpace-data-ecosystem @ 83e577752784 (data_model/metadata/metadata-standards/superset/out/crosswalk.json). Edited here since; this file is the source of truth." +} diff --git a/api/services/metadata_export/contracts/geographies.csv b/api/services/metadata_export/contracts/geographies.csv new file mode 100644 index 00000000..0dfe7481 --- /dev/null +++ b/api/services/metadata_export/contracts/geographies.csv @@ -0,0 +1,828 @@ +key,label,code,uri,tier,parent_key,visible_on_dataspace,is_active +region:north-east-india,North East India,north-east-india,https://civicdataspace.in/id/geography/region/north-east-india,REGION,,yes,True +region:southern-asia,Southern Asia,southern-asia,https://civicdataspace.in/id/geography/region/southern-asia,REGION,,yes,True +country:IN,India,IN,https://www.wikidata.org/entity/Q668,COUNTRY,,yes,True +state:1,Jammu and Kashmir,1,http://www.wikidata.org/entity/Q66278313,UT,country:IN,yes,True +state:10,Bihar,10,http://www.wikidata.org/entity/Q1165,STATE,country:IN,yes,True +state:11,Sikkim,11,http://www.wikidata.org/entity/Q1505,STATE,country:IN,yes,True +state:12,Arunachal Pradesh,12,http://www.wikidata.org/entity/Q1162,STATE,country:IN,yes,True +state:13,Nagaland,13,http://www.wikidata.org/entity/Q1599,STATE,country:IN,yes,True +state:14,Manipur,14,http://www.wikidata.org/entity/Q1193,STATE,country:IN,yes,True +state:15,Mizoram,15,http://www.wikidata.org/entity/Q1502,STATE,country:IN,yes,True +state:16,Tripura,16,http://www.wikidata.org/entity/Q1363,STATE,country:IN,yes,True +state:17,Meghalaya,17,http://www.wikidata.org/entity/Q1195,STATE,country:IN,yes,True +state:18,Assam,18,http://www.wikidata.org/entity/Q1164,STATE,country:IN,yes,True +state:19,West Bengal,19,http://www.wikidata.org/entity/Q1356,STATE,country:IN,yes,True +state:2,Himachal Pradesh,2,http://www.wikidata.org/entity/Q1177,STATE,country:IN,yes,True +state:20,Jharkhand,20,http://www.wikidata.org/entity/Q1184,STATE,country:IN,yes,True +state:21,Odisha,21,http://www.wikidata.org/entity/Q22048,STATE,country:IN,yes,True +state:22,Chhattisgarh,22,http://www.wikidata.org/entity/Q1168,STATE,country:IN,yes,True +state:23,Madhya Pradesh,23,http://www.wikidata.org/entity/Q1188,STATE,country:IN,yes,True +state:24,Gujarat,24,http://www.wikidata.org/entity/Q1061,STATE,country:IN,yes,True +state:27,Maharashtra,27,http://www.wikidata.org/entity/Q1191,STATE,country:IN,yes,True +state:28,Andhra Pradesh,28,http://www.wikidata.org/entity/Q1159,STATE,country:IN,yes,True +state:29,Karnataka,29,http://www.wikidata.org/entity/Q1185,STATE,country:IN,yes,True +state:3,Punjab,3,http://www.wikidata.org/entity/Q22424,STATE,country:IN,yes,True +state:30,Goa,30,http://www.wikidata.org/entity/Q1171,STATE,country:IN,yes,True +state:31,Lakshadweep,31,http://www.wikidata.org/entity/Q26927,UT,country:IN,yes,True +state:32,Kerala,32,http://www.wikidata.org/entity/Q1186,STATE,country:IN,yes,True +state:33,Tamil Nadu,33,http://www.wikidata.org/entity/Q1445,STATE,country:IN,yes,True +state:34,Puducherry,34,http://www.wikidata.org/entity/Q66743,UT,country:IN,yes,True +state:35,Andaman and Nicobar Islands,35,http://www.wikidata.org/entity/Q40888,UT,country:IN,yes,True +state:36,Telangana,36,http://www.wikidata.org/entity/Q677037,STATE,country:IN,yes,True +state:37,Ladakh,37,http://www.wikidata.org/entity/Q200667,UT,country:IN,yes,True +state:38,Dadra and Nagar Haveli and Daman and Diu,38,http://www.wikidata.org/entity/Q77997266,UT,country:IN,yes,True +state:4,Chandigarh,4,http://www.wikidata.org/entity/Q120971341,UT,country:IN,yes,True +state:5,Uttarakhand,5,http://www.wikidata.org/entity/Q1499,STATE,country:IN,yes,True +state:6,Haryana,6,http://www.wikidata.org/entity/Q1174,STATE,country:IN,yes,True +state:7,National Capital Territory of Delhi,7,http://www.wikidata.org/entity/Q9357528,UT,country:IN,yes,True +state:8,Rajasthan,8,http://www.wikidata.org/entity/Q1437,STATE,country:IN,yes,True +state:9,Uttar Pradesh,9,http://www.wikidata.org/entity/Q1498,STATE,country:IN,yes,True +district:1,Anantnag,1,http://www.wikidata.org/entity/Q2982349,DISTRICT,state:1,no,True +district:10,Poonch,10,http://www.wikidata.org/entity/Q2983134,DISTRICT,state:1,no,True +district:11,Pulwama,11,http://www.wikidata.org/entity/Q2085364,DISTRICT,state:1,no,True +district:12,Rajouri,12,http://www.wikidata.org/entity/Q544279,DISTRICT,state:1,no,True +district:13,Srinagar,13,http://www.wikidata.org/entity/Q1506029,DISTRICT,state:1,no,True +district:14,Udhampur,14,http://www.wikidata.org/entity/Q1947311,DISTRICT,state:1,no,True +district:2,Budgam,2,http://www.wikidata.org/entity/Q2594218,DISTRICT,state:1,no,True +district:3,Baramulla,3,http://www.wikidata.org/entity/Q1912057,DISTRICT,state:1,no,True +district:4,Doda,4,http://www.wikidata.org/entity/Q2298979,DISTRICT,state:1,no,True +district:5,Jammu,5,http://www.wikidata.org/entity/Q1947371,DISTRICT,state:1,no,True +district:620,Kishtwar,620,http://www.wikidata.org/entity/Q2321899,DISTRICT,state:1,no,True +district:621,Ramban,621,http://www.wikidata.org/entity/Q2321939,DISTRICT,state:1,no,True +district:622,Kulgam,622,http://www.wikidata.org/entity/Q2321867,DISTRICT,state:1,no,True +district:623,Bandipore,623,http://www.wikidata.org/entity/Q2983553,DISTRICT,state:1,no,True +district:624,Samba,624,http://www.wikidata.org/entity/Q1117086,DISTRICT,state:1,no,True +district:625,Shopian,625,http://www.wikidata.org/entity/Q2073646,DISTRICT,state:1,no,True +district:626,Ganderbal,626,http://www.wikidata.org/entity/Q2556028,DISTRICT,state:1,no,True +district:627,Reasi,627,http://www.wikidata.org/entity/Q2321956,DISTRICT,state:1,no,True +district:7,Kathua,7,http://www.wikidata.org/entity/Q2375700,DISTRICT,state:1,no,True +district:8,Kupwara,8,http://www.wikidata.org/entity/Q2297306,DISTRICT,state:1,no,True +district:188,Araria,188,http://www.wikidata.org/entity/Q42901,DISTRICT,state:10,no,True +district:189,Aurangabad,189,http://www.wikidata.org/entity/Q43086,DISTRICT,state:10,no,True +district:190,Banka,190,http://www.wikidata.org/entity/Q43097,DISTRICT,state:10,no,True +district:191,Begusarai,191,http://www.wikidata.org/entity/Q49157,DISTRICT,state:10,no,True +district:192,Bhagalpur,192,http://www.wikidata.org/entity/Q49155,DISTRICT,state:10,no,True +district:193,Bhojpur,193,http://www.wikidata.org/entity/Q49153,DISTRICT,state:10,no,True +district:194,Buxar,194,http://www.wikidata.org/entity/Q49161,DISTRICT,state:10,no,True +district:195,Darbhanga,195,http://www.wikidata.org/entity/Q49160,DISTRICT,state:10,no,True +district:196,Gaya,196,http://www.wikidata.org/entity/Q49173,DISTRICT,state:10,no,True +district:197,Gopalganj,197,http://www.wikidata.org/entity/Q49171,DISTRICT,state:10,no,True +district:198,Jamui,198,http://www.wikidata.org/entity/Q49168,DISTRICT,state:10,no,True +district:199,Jehanabad,199,http://www.wikidata.org/entity/Q49176,DISTRICT,state:10,no,True +district:200,Kaimur,200,http://www.wikidata.org/entity/Q77367,DISTRICT,state:10,no,True +district:201,Katihar,201,http://www.wikidata.org/entity/Q77568,DISTRICT,state:10,no,True +district:202,Khagaria,202,http://www.wikidata.org/entity/Q49175,DISTRICT,state:10,no,True +district:203,Kishanganj,203,http://www.wikidata.org/entity/Q77375,DISTRICT,state:10,no,True +district:204,Lakhisarai,204,http://www.wikidata.org/entity/Q77505,DISTRICT,state:10,no,True +district:205,Madhepura,205,http://www.wikidata.org/entity/Q77746,DISTRICT,state:10,no,True +district:206,Madhubani,206,http://www.wikidata.org/entity/Q77474,DISTRICT,state:10,no,True +district:207,Munger,207,http://www.wikidata.org/entity/Q77452,DISTRICT,state:10,no,True +district:208,Muzaffarpur,208,http://www.wikidata.org/entity/Q77731,DISTRICT,state:10,no,True +district:209,Nalanda,209,http://www.wikidata.org/entity/Q77633,DISTRICT,state:10,no,True +district:210,Nawada,210,http://www.wikidata.org/entity/Q100067,DISTRICT,state:10,no,True +district:211,West Champaran,211,http://www.wikidata.org/entity/Q100124,DISTRICT,state:10,no,True +district:212,Patna,212,http://www.wikidata.org/entity/Q100077,DISTRICT,state:10,no,True +district:213,East Champaran,213,http://www.wikidata.org/entity/Q49159,DISTRICT,state:10,no,True +district:214,Purnia,214,http://www.wikidata.org/entity/Q100082,DISTRICT,state:10,no,True +district:215,Rohtas,215,http://www.wikidata.org/entity/Q100085,DISTRICT,state:10,no,True +district:216,Saharsa,216,http://www.wikidata.org/entity/Q100120,DISTRICT,state:10,no,True +district:217,Samastipur,217,http://www.wikidata.org/entity/Q100117,DISTRICT,state:10,no,True +district:218,Saran,218,http://www.wikidata.org/entity/Q100146,DISTRICT,state:10,no,True +district:219,Sheikhpura,219,http://www.wikidata.org/entity/Q100093,DISTRICT,state:10,no,True +district:220,Sheohar,220,http://www.wikidata.org/entity/Q100095,DISTRICT,state:10,no,True +district:221,Sitamarhi,221,http://www.wikidata.org/entity/Q100144,DISTRICT,state:10,no,True +district:222,Siwan,222,http://www.wikidata.org/entity/Q100131,DISTRICT,state:10,no,True +district:223,Supaul,223,http://www.wikidata.org/entity/Q100139,DISTRICT,state:10,no,True +district:224,Vaishali,224,http://www.wikidata.org/entity/Q100130,DISTRICT,state:10,no,True +district:611,Arwal,611,http://www.wikidata.org/entity/Q42917,DISTRICT,state:10,no,True +district:225,Gangtok,225,http://www.wikidata.org/entity/Q113956160,DISTRICT,state:11,no,True +district:226,Mangan,226,http://www.wikidata.org/entity/Q1784149,DISTRICT,state:11,no,True +district:227,Namchi,227,http://www.wikidata.org/entity/Q1805051,DISTRICT,state:11,no,True +district:228,Gyalshing,228,http://www.wikidata.org/entity/Q611357,DISTRICT,state:11,no,True +district:741,Pakyong,741,http://www.wikidata.org/entity/Q108803704,DISTRICT,state:11,no,True +district:742,Soreng,742,http://www.wikidata.org/entity/Q112939132,DISTRICT,state:11,no,True +district:229,Changlang,229,http://www.wikidata.org/entity/Q15427,DISTRICT,state:12,no,True +district:230,Dibang Valley,230,http://www.wikidata.org/entity/Q15446,DISTRICT,state:12,no,True +district:231,East Kameng,231,http://www.wikidata.org/entity/Q15424,DISTRICT,state:12,no,True +district:232,East Siang,232,http://www.wikidata.org/entity/Q15419,DISTRICT,state:12,no,True +district:233,Kurung Kumey,233,http://www.wikidata.org/entity/Q2449506,DISTRICT,state:12,no,True +district:234,Lohit,234,http://www.wikidata.org/entity/Q15438,DISTRICT,state:12,no,True +district:235,Lower Dibang Valley,235,http://www.wikidata.org/entity/Q2373368,DISTRICT,state:12,no,True +district:236,Lower Subansiri,236,http://www.wikidata.org/entity/Q15436,DISTRICT,state:12,no,True +district:237,Papum Pare,237,http://www.wikidata.org/entity/Q15432,DISTRICT,state:12,no,True +district:238,Tawang,238,http://www.wikidata.org/entity/Q15449,DISTRICT,state:12,no,True +district:239,Tirap,239,http://www.wikidata.org/entity/Q15448,DISTRICT,state:12,no,True +district:240,Upper Siang,240,http://www.wikidata.org/entity/Q15465,DISTRICT,state:12,no,True +district:241,Upper Subansiri,241,http://www.wikidata.org/entity/Q15464,DISTRICT,state:12,no,True +district:242,West Kameng,242,http://www.wikidata.org/entity/Q15459,DISTRICT,state:12,no,True +district:243,West Siang,243,http://www.wikidata.org/entity/Q15453,DISTRICT,state:12,no,True +district:628,Anjaw,628,http://www.wikidata.org/entity/Q15413,DISTRICT,state:12,no,True +district:666,Longding,666,http://www.wikidata.org/entity/Q5627568,DISTRICT,state:12,no,True +district:677,Kra Daadi,677,http://www.wikidata.org/entity/Q21018627,DISTRICT,state:12,no,True +district:678,Namsai,678,http://www.wikidata.org/entity/Q21559824,DISTRICT,state:12,no,True +district:679,Siang,679,http://www.wikidata.org/entity/Q18642331,DISTRICT,state:12,no,True +district:718,Kamle,718,http://www.wikidata.org/entity/Q48731073,DISTRICT,state:12,no,True +district:719,Lower Siang,719,http://www.wikidata.org/entity/Q13602925,DISTRICT,state:12,no,True +district:723,Pakke-Kessang,723,http://www.wikidata.org/entity/Q61439260,DISTRICT,state:12,no,True +district:724,Lepa Rada,724,http://www.wikidata.org/entity/Q63563632,DISTRICT,state:12,no,True +district:725,Shi Yomi,725,http://www.wikidata.org/entity/Q63563625,DISTRICT,state:12,no,True +district:786,Keyi Panyor,786,http://www.wikidata.org/entity/Q124818540,DISTRICT,state:12,no,True +district:787,Bichom,787,http://www.wikidata.org/entity/Q124811465,DISTRICT,state:12,no,True +district:244,Dimapur,244,http://www.wikidata.org/entity/Q634262,DISTRICT,state:13,no,True +district:245,Kohima,245,http://www.wikidata.org/entity/Q953530,DISTRICT,state:13,no,True +district:246,Mokokchung,246,http://www.wikidata.org/entity/Q2175311,DISTRICT,state:13,no,True +district:247,Mon,247,http://www.wikidata.org/entity/Q2339648,DISTRICT,state:13,no,True +district:248,Phek,248,http://www.wikidata.org/entity/Q590882,DISTRICT,state:13,no,True +district:249,Tuensang,249,http://www.wikidata.org/entity/Q2571393,DISTRICT,state:13,no,True +district:250,Wokha,250,http://www.wikidata.org/entity/Q681821,DISTRICT,state:13,no,True +district:251,Zunheboto,251,http://www.wikidata.org/entity/Q2091461,DISTRICT,state:13,no,True +district:613,Peren,613,http://www.wikidata.org/entity/Q516294,DISTRICT,state:13,no,True +district:614,Kiphire,614,http://www.wikidata.org/entity/Q2597908,DISTRICT,state:13,no,True +district:615,Longleng,615,http://www.wikidata.org/entity/Q1426783,DISTRICT,state:13,no,True +district:736,Noklak,736,http://www.wikidata.org/entity/Q48731903,DISTRICT,state:13,no,True +district:757,Tseminyü,757,http://www.wikidata.org/entity/Q110223836,DISTRICT,state:13,no,True +district:758,Chümoukedima,758,http://www.wikidata.org/entity/Q110223837,DISTRICT,state:13,no,True +district:764,Niuland,764,http://www.wikidata.org/entity/Q110223839,DISTRICT,state:13,no,True +district:765,Shamator,765,http://www.wikidata.org/entity/Q111529435,DISTRICT,state:13,no,True +district:788,Meluri,788,http://www.wikidata.org/entity/Q131191793,DISTRICT,state:13,no,True +district:252,Bishnupur,252,http://www.wikidata.org/entity/Q938190,DISTRICT,state:14,no,True +district:253,Chandel,253,http://www.wikidata.org/entity/Q2301769,DISTRICT,state:14,no,True +district:254,Churachandpur,254,http://www.wikidata.org/entity/Q2577281,DISTRICT,state:14,no,True +district:255,Imphal East,255,http://www.wikidata.org/entity/Q1916666,DISTRICT,state:14,no,True +district:256,Imphal West,256,http://www.wikidata.org/entity/Q1822188,DISTRICT,state:14,no,True +district:257,Senapati,257,http://www.wikidata.org/entity/Q2301706,DISTRICT,state:14,no,True +district:258,Tamenglong,258,http://www.wikidata.org/entity/Q2301717,DISTRICT,state:14,no,True +district:259,Thoubal,259,http://www.wikidata.org/entity/Q2086198,DISTRICT,state:14,no,True +district:260,Ukhrul,260,http://www.wikidata.org/entity/Q735101,DISTRICT,state:14,no,True +district:711,Kakching,711,http://www.wikidata.org/entity/Q28173825,DISTRICT,state:14,no,True +district:712,Kangpokpi,712,http://www.wikidata.org/entity/Q28419386,DISTRICT,state:14,no,True +district:713,Jiribam,713,http://www.wikidata.org/entity/Q28419387,DISTRICT,state:14,no,True +district:714,Noney,714,http://www.wikidata.org/entity/Q28419389,DISTRICT,state:14,no,True +district:715,Pherzawl,715,http://www.wikidata.org/entity/Q28173809,DISTRICT,state:14,no,True +district:716,Tengnoupal,716,http://www.wikidata.org/entity/Q28419388,DISTRICT,state:14,no,True +district:717,Kamjong,717,http://www.wikidata.org/entity/Q28419390,DISTRICT,state:14,no,True +district:261,Aizawl,261,http://www.wikidata.org/entity/Q1947322,DISTRICT,state:15,no,True +district:262,Champhai,262,http://www.wikidata.org/entity/Q1965256,DISTRICT,state:15,no,True +district:263,Kolasib,263,http://www.wikidata.org/entity/Q1947343,DISTRICT,state:15,no,True +district:264,Lawngtlai,264,http://www.wikidata.org/entity/Q2086209,DISTRICT,state:15,no,True +district:265,Lunglei,265,http://www.wikidata.org/entity/Q1947352,DISTRICT,state:15,no,True +district:266,Mamit,266,http://www.wikidata.org/entity/Q751531,DISTRICT,state:15,no,True +district:267,Saiha,267,http://www.wikidata.org/entity/Q1821714,DISTRICT,state:15,no,True +district:268,Serchhip,268,http://www.wikidata.org/entity/Q2086190,DISTRICT,state:15,no,True +district:726,Hnahthial,726,http://www.wikidata.org/entity/Q86882590,DISTRICT,state:15,no,True +district:727,Saitual,727,http://www.wikidata.org/entity/Q86882593,DISTRICT,state:15,no,True +district:728,Khawzawl,728,http://www.wikidata.org/entity/Q86882591,DISTRICT,state:15,no,True +district:269,Dhalai,269,http://www.wikidata.org/entity/Q2086546,DISTRICT,state:16,no,True +district:270,North Tripura,270,http://www.wikidata.org/entity/Q1920978,DISTRICT,state:16,no,True +district:271,South Tripura,271,http://www.wikidata.org/entity/Q1822159,DISTRICT,state:16,no,True +district:272,West Tripura,272,http://www.wikidata.org/entity/Q1947570,DISTRICT,state:16,no,True +district:652,Khowai,652,http://www.wikidata.org/entity/Q16086680,DISTRICT,state:16,no,True +district:653,Sepahijala,653,http://www.wikidata.org/entity/Q16086076,DISTRICT,state:16,no,True +district:654,Gomati,654,http://www.wikidata.org/entity/Q16086497,DISTRICT,state:16,no,True +district:655,Unakoti,655,http://www.wikidata.org/entity/Q16087996,DISTRICT,state:16,no,True +district:273,East Garo Hills,273,http://www.wikidata.org/entity/Q2085455,DISTRICT,state:17,no,True +district:274,East Khasi Hills,274,http://www.wikidata.org/entity/Q1945304,DISTRICT,state:17,no,True +district:275,West Jaintia Hills,275,http://www.wikidata.org/entity/Q13181190,DISTRICT,state:17,no,True +district:276,Ri-Bhoi,276,http://www.wikidata.org/entity/Q1884672,DISTRICT,state:17,no,True +district:277,South Garo Hills,277,http://www.wikidata.org/entity/Q2329228,DISTRICT,state:17,no,True +district:278,West Garo Hills,278,http://www.wikidata.org/entity/Q2329181,DISTRICT,state:17,no,True +district:279,West Khasi Hills,279,http://www.wikidata.org/entity/Q2064752,DISTRICT,state:17,no,True +district:656,North Garo Hills,656,http://www.wikidata.org/entity/Q7055466,DISTRICT,state:17,no,True +district:657,East Jaintia Hills,657,http://www.wikidata.org/entity/Q15923776,DISTRICT,state:17,no,True +district:658,South West Khasi Hills,658,http://www.wikidata.org/entity/Q15923741,DISTRICT,state:17,no,True +district:663,South West Garo Hills,663,http://www.wikidata.org/entity/Q15961576,DISTRICT,state:17,no,True +district:740,Eastern West Khasi Hills,740,http://www.wikidata.org/entity/Q110442602,DISTRICT,state:17,no,True +district:280,Barpeta,280,http://www.wikidata.org/entity/Q41249,DISTRICT,state:18,no,True +district:281,Bongaigaon,281,http://www.wikidata.org/entity/Q42197,DISTRICT,state:18,no,True +district:282,Cachar,282,http://www.wikidata.org/entity/Q42209,DISTRICT,state:18,no,True +district:283,Darrang,283,http://www.wikidata.org/entity/Q42461,DISTRICT,state:18,no,True +district:284,Dhemaji,284,http://www.wikidata.org/entity/Q42473,DISTRICT,state:18,no,True +district:285,Dhubri,285,http://www.wikidata.org/entity/Q42485,DISTRICT,state:18,no,True +district:286,Dibrugarh,286,http://www.wikidata.org/entity/Q42479,DISTRICT,state:18,no,True +district:287,Goalpara,287,http://www.wikidata.org/entity/Q42522,DISTRICT,state:18,no,True +district:288,Golaghat,288,http://www.wikidata.org/entity/Q42517,DISTRICT,state:18,no,True +district:289,Hailakandi,289,http://www.wikidata.org/entity/Q42505,DISTRICT,state:18,no,True +district:290,Jorhat,290,http://www.wikidata.org/entity/Q42611,DISTRICT,state:18,no,True +district:291,Kamrup,291,http://www.wikidata.org/entity/Q2247441,DISTRICT,state:18,no,True +district:292,Karbi Anglong,292,http://www.wikidata.org/entity/Q29025081,DISTRICT,state:18,no,True +district:293,Sribhumi,293,http://www.wikidata.org/entity/Q42542,DISTRICT,state:18,no,True +district:294,Kokrajhar,294,http://www.wikidata.org/entity/Q42618,DISTRICT,state:18,no,True +district:295,Lakhimpur,295,http://www.wikidata.org/entity/Q42743,DISTRICT,state:18,no,True +district:296,Morigaon,296,http://www.wikidata.org/entity/Q42737,DISTRICT,state:18,no,True +district:297,,297,http://www.wikidata.org/entity/Q42686,DISTRICT,state:18,no,True +district:298,Nalbari,298,http://www.wikidata.org/entity/Q42779,DISTRICT,state:18,no,True +district:299,Dima Hasao,299,http://www.wikidata.org/entity/Q42774,DISTRICT,state:18,no,True +district:300,Sivasagar,300,http://www.wikidata.org/entity/Q42768,DISTRICT,state:18,no,True +district:301,Sonitpur,301,http://www.wikidata.org/entity/Q42765,DISTRICT,state:18,no,True +district:302,Tinsukia,302,http://www.wikidata.org/entity/Q42756,DISTRICT,state:18,no,True +district:612,Chirang,612,http://www.wikidata.org/entity/Q2574898,DISTRICT,state:18,no,True +district:616,Baksa,616,http://www.wikidata.org/entity/Q2360266,DISTRICT,state:18,no,True +district:617,Udalguri,617,http://www.wikidata.org/entity/Q321998,DISTRICT,state:18,no,True +district:618,Kamrup Metropolitan,618,http://www.wikidata.org/entity/Q2464674,DISTRICT,state:18,no,True +district:705,Bishwanath,705,http://www.wikidata.org/entity/Q22079836,DISTRICT,state:18,no,True +district:706,Majuli,706,http://www.wikidata.org/entity/Q28110729,DISTRICT,state:18,no,True +district:707,South Salmara-Mankachar,707,http://www.wikidata.org/entity/Q24907599,DISTRICT,state:18,no,True +district:708,Charaideo,708,http://www.wikidata.org/entity/Q24039029,DISTRICT,state:18,no,True +district:709,Hojai,709,http://www.wikidata.org/entity/Q24699407,DISTRICT,state:18,no,True +district:710,West Karbi Anglong,710,http://www.wikidata.org/entity/Q24949218,DISTRICT,state:18,no,True +district:739,Bajali,739,http://www.wikidata.org/entity/Q101088203,DISTRICT,state:18,no,True +district:756,Tamulpur,756,http://www.wikidata.org/entity/Q110661970,DISTRICT,state:18,no,True +district:303,North 24 Parganas,303,http://www.wikidata.org/entity/Q338425,DISTRICT,state:19,no,True +district:304,South 24 Parganas,304,http://www.wikidata.org/entity/Q2308319,DISTRICT,state:19,no,True +district:305,Bankura,305,http://www.wikidata.org/entity/Q2088458,DISTRICT,state:19,no,True +district:306,Purba Bardhaman,306,http://www.wikidata.org/entity/Q29257278,DISTRICT,state:19,no,True +district:307,Birbhum,307,http://www.wikidata.org/entity/Q2088440,DISTRICT,state:19,no,True +district:308,Cooch Behar,308,http://www.wikidata.org/entity/Q2728658,DISTRICT,state:19,no,True +district:309,Darjeeling,309,http://www.wikidata.org/entity/Q1134759,DISTRICT,state:19,no,True +district:310,Dakshin Dinajpur,310,http://www.wikidata.org/entity/Q533839,DISTRICT,state:19,no,True +district:311,Uttar Dinajpur,311,http://www.wikidata.org/entity/Q2019766,DISTRICT,state:19,no,True +district:312,Hooghly,312,http://www.wikidata.org/entity/Q548518,DISTRICT,state:19,no,True +district:313,Howrah,313,http://www.wikidata.org/entity/Q1478937,DISTRICT,state:19,no,True +district:314,Jalpaiguri,314,http://www.wikidata.org/entity/Q1351487,DISTRICT,state:19,no,True +district:315,Kolkata,315,http://www.wikidata.org/entity/Q2088496,DISTRICT,state:19,no,True +district:316,Malda,316,http://www.wikidata.org/entity/Q2049820,DISTRICT,state:19,no,True +district:317,Purba Medinipur,317,http://www.wikidata.org/entity/Q1431920,DISTRICT,state:19,no,True +district:318,Paschim Medinipur,318,http://www.wikidata.org/entity/Q1855537,DISTRICT,state:19,no,True +district:319,Murshidabad,319,http://www.wikidata.org/entity/Q1546240,DISTRICT,state:19,no,True +district:320,Nadia,320,http://www.wikidata.org/entity/Q1143880,DISTRICT,state:19,no,True +district:321,Purulia,321,http://www.wikidata.org/entity/Q307474,DISTRICT,state:19,no,True +district:664,Alipurduar,664,http://www.wikidata.org/entity/Q4726845,DISTRICT,state:19,no,True +district:702,Kalimpong,702,http://www.wikidata.org/entity/Q28769140,DISTRICT,state:19,no,True +district:703,Jhargram,703,http://www.wikidata.org/entity/Q29168456,DISTRICT,state:19,no,True +district:704,Paschim Bardhaman,704,http://www.wikidata.org/entity/Q29215602,DISTRICT,state:19,no,True +district:15,Bilaspur,15,http://www.wikidata.org/entity/Q1478939,DISTRICT,state:2,no,True +district:16,Chamba,16,http://www.wikidata.org/entity/Q1060614,DISTRICT,state:2,no,True +district:17,Hamirpur,17,http://www.wikidata.org/entity/Q2086180,DISTRICT,state:2,no,True +district:18,Kangra,18,http://www.wikidata.org/entity/Q727232,DISTRICT,state:2,no,True +district:19,Kinnaur,19,http://www.wikidata.org/entity/Q1862950,DISTRICT,state:2,no,True +district:20,Kullu,20,http://www.wikidata.org/entity/Q2980880,DISTRICT,state:2,no,True +district:21,Lahaul and Spiti,21,http://www.wikidata.org/entity/Q837595,DISTRICT,state:2,no,True +district:22,Mandi,22,http://www.wikidata.org/entity/Q1892161,DISTRICT,state:2,no,True +district:23,Shimla,23,http://www.wikidata.org/entity/Q1921404,DISTRICT,state:2,no,True +district:24,Sirmaur,24,http://www.wikidata.org/entity/Q654331,DISTRICT,state:2,no,True +district:25,Solan,25,http://www.wikidata.org/entity/Q2980937,DISTRICT,state:2,no,True +district:26,Una,26,http://www.wikidata.org/entity/Q2301741,DISTRICT,state:2,no,True +district:322,Bokaro,322,http://www.wikidata.org/entity/Q2295925,DISTRICT,state:20,no,True +district:323,Chatra,323,http://www.wikidata.org/entity/Q1979499,DISTRICT,state:20,no,True +district:324,Deoghar,324,http://www.wikidata.org/entity/Q2030017,DISTRICT,state:20,no,True +district:325,Dhanbad,325,http://www.wikidata.org/entity/Q2240791,DISTRICT,state:20,no,True +district:326,Dumka,326,http://www.wikidata.org/entity/Q2577657,DISTRICT,state:20,no,True +district:327,East Singhbhum,327,http://www.wikidata.org/entity/Q2452921,DISTRICT,state:20,no,True +district:328,Garhwa,328,http://www.wikidata.org/entity/Q2302076,DISTRICT,state:20,no,True +district:329,Giridih,329,http://www.wikidata.org/entity/Q2302065,DISTRICT,state:20,no,True +district:330,Godda,330,http://www.wikidata.org/entity/Q638980,DISTRICT,state:20,no,True +district:331,Gumla,331,http://www.wikidata.org/entity/Q2295865,DISTRICT,state:20,no,True +district:332,Hazaribagh,332,http://www.wikidata.org/entity/Q1945416,DISTRICT,state:20,no,True +district:333,Jamtara,333,http://www.wikidata.org/entity/Q2980986,DISTRICT,state:20,no,True +district:334,Koderma,334,http://www.wikidata.org/entity/Q2085480,DISTRICT,state:20,no,True +district:335,Latehar,335,http://www.wikidata.org/entity/Q2244762,DISTRICT,state:20,no,True +district:336,Lohardaga,336,http://www.wikidata.org/entity/Q1948301,DISTRICT,state:20,no,True +district:337,Pakur,337,http://www.wikidata.org/entity/Q2295930,DISTRICT,state:20,no,True +district:338,Palamu,338,http://www.wikidata.org/entity/Q1797254,DISTRICT,state:20,no,True +district:339,Ranchi,339,http://www.wikidata.org/entity/Q1947380,DISTRICT,state:20,no,True +district:340,Sahebganj,340,http://www.wikidata.org/entity/Q767878,DISTRICT,state:20,no,True +district:341,Seraikela Kharsawan,341,http://www.wikidata.org/entity/Q2362658,DISTRICT,state:20,no,True +district:342,Simdega,342,http://www.wikidata.org/entity/Q2597889,DISTRICT,state:20,no,True +district:343,West Singhbhum,343,http://www.wikidata.org/entity/Q1950527,DISTRICT,state:20,no,True +district:606,Khunti,606,http://www.wikidata.org/entity/Q367344,DISTRICT,state:20,no,True +district:607,Ramgarh,607,http://www.wikidata.org/entity/Q2663612,DISTRICT,state:20,no,True +district:344,Angul,344,http://www.wikidata.org/entity/Q1772807,DISTRICT,state:21,no,True +district:345,Balangir,345,http://www.wikidata.org/entity/Q804642,DISTRICT,state:21,no,True +district:346,Balasore,346,http://www.wikidata.org/entity/Q2022279,DISTRICT,state:21,no,True +district:347,Bargarh,347,http://www.wikidata.org/entity/Q808140,DISTRICT,state:21,no,True +district:348,Bhadrak,348,http://www.wikidata.org/entity/Q685638,DISTRICT,state:21,no,True +district:349,Boudh,349,http://www.wikidata.org/entity/Q2363639,DISTRICT,state:21,no,True +district:350,Cuttack,350,http://www.wikidata.org/entity/Q2022256,DISTRICT,state:21,no,True +district:351,Debagarh,351,http://www.wikidata.org/entity/Q2269639,DISTRICT,state:21,no,True +district:352,Dhenkanal,352,http://www.wikidata.org/entity/Q1948389,DISTRICT,state:21,no,True +district:353,Gajapati,353,http://www.wikidata.org/entity/Q1947292,DISTRICT,state:21,no,True +district:354,Ganjam,354,http://www.wikidata.org/entity/Q776213,DISTRICT,state:21,no,True +district:355,Jagatsinghpur,355,http://www.wikidata.org/entity/Q971581,DISTRICT,state:21,no,True +district:356,Jajpur,356,http://www.wikidata.org/entity/Q2087771,DISTRICT,state:21,no,True +district:357,Jharsuguda,357,http://www.wikidata.org/entity/Q569181,DISTRICT,state:21,no,True +district:358,Kalahandi,358,http://www.wikidata.org/entity/Q1876588,DISTRICT,state:21,no,True +district:359,Kandhamal,359,http://www.wikidata.org/entity/Q2085500,DISTRICT,state:21,no,True +district:360,Kendrapara,360,http://www.wikidata.org/entity/Q2299172,DISTRICT,state:21,no,True +district:361,Kendujhar,361,http://www.wikidata.org/entity/Q2085428,DISTRICT,state:21,no,True +district:362,Khordha,362,http://www.wikidata.org/entity/Q662818,DISTRICT,state:21,no,True +district:363,Koraput,363,http://www.wikidata.org/entity/Q1947300,DISTRICT,state:21,no,True +district:364,Malkangiri,364,http://www.wikidata.org/entity/Q5122619,DISTRICT,state:21,no,True +district:365,Mayurbhanj,365,http://www.wikidata.org/entity/Q1914546,DISTRICT,state:21,no,True +district:366,Nabarangpur,366,http://www.wikidata.org/entity/Q2396798,DISTRICT,state:21,no,True +district:367,Nayagarh,367,http://www.wikidata.org/entity/Q2367388,DISTRICT,state:21,no,True +district:368,Nuapada,368,http://www.wikidata.org/entity/Q1810550,DISTRICT,state:21,no,True +district:369,Puri,369,http://www.wikidata.org/entity/Q1817158,DISTRICT,state:21,no,True +district:370,Rayagada,370,http://www.wikidata.org/entity/Q2577997,DISTRICT,state:21,no,True +district:371,Sambalpur,371,http://www.wikidata.org/entity/Q1267306,DISTRICT,state:21,no,True +district:372,Subarnapur,372,http://www.wikidata.org/entity/Q1473957,DISTRICT,state:21,no,True +district:373,Sundargarh,373,http://www.wikidata.org/entity/Q2296047,DISTRICT,state:21,no,True +district:374,Bastar,374,http://www.wikidata.org/entity/Q100152,DISTRICT,state:22,no,True +district:375,Bilaspur,375,http://www.wikidata.org/entity/Q100157,DISTRICT,state:22,no,True +district:376,Dantewada,376,http://www.wikidata.org/entity/Q100211,DISTRICT,state:22,no,True +district:377,Dhamtari,377,http://www.wikidata.org/entity/Q100190,DISTRICT,state:22,no,True +district:378,Durg,378,http://www.wikidata.org/entity/Q100182,DISTRICT,state:22,no,True +district:379,Janjgir–Champa,379,http://www.wikidata.org/entity/Q2575633,DISTRICT,state:22,no,True +district:380,Jashpur,380,http://www.wikidata.org/entity/Q2577551,DISTRICT,state:22,no,True +district:381,Kanker,381,http://www.wikidata.org/entity/Q2310530,DISTRICT,state:22,no,True +district:382,Kabirdham,382,http://www.wikidata.org/entity/Q2450255,DISTRICT,state:22,no,True +district:383,Korba,383,http://www.wikidata.org/entity/Q2299121,DISTRICT,state:22,no,True +district:384,Koriya,384,http://www.wikidata.org/entity/Q2295896,DISTRICT,state:22,no,True +district:385,Mahasamund,385,http://www.wikidata.org/entity/Q2450240,DISTRICT,state:22,no,True +district:386,Raigarh,386,http://www.wikidata.org/entity/Q2286310,DISTRICT,state:22,no,True +district:387,Raipur,387,http://www.wikidata.org/entity/Q2295914,DISTRICT,state:22,no,True +district:388,Rajnandgaon,388,http://www.wikidata.org/entity/Q2341800,DISTRICT,state:22,no,True +district:389,Surguja,389,http://www.wikidata.org/entity/Q1805075,DISTRICT,state:22,no,True +district:636,Bijapur,636,http://www.wikidata.org/entity/Q100164,DISTRICT,state:22,no,True +district:637,Narayanpur,637,http://www.wikidata.org/entity/Q2322000,DISTRICT,state:22,no,True +district:642,Sukma,642,http://www.wikidata.org/entity/Q16933590,DISTRICT,state:22,no,True +district:643,Kondagaon,643,http://www.wikidata.org/entity/Q12420995,DISTRICT,state:22,no,True +district:644,Baloda Bazar,644,http://www.wikidata.org/entity/Q15663455,DISTRICT,state:22,no,True +district:645,Gariaband,645,http://www.wikidata.org/entity/Q16961365,DISTRICT,state:22,no,True +district:646,Balod,646,http://www.wikidata.org/entity/Q16056266,DISTRICT,state:22,no,True +district:647,Mungeli,647,http://www.wikidata.org/entity/Q13476249,DISTRICT,state:22,no,True +district:648,Surajpur,648,http://www.wikidata.org/entity/Q16938031,DISTRICT,state:22,no,True +district:649,Balrampur–Ramanujganj,649,http://www.wikidata.org/entity/Q16056268,DISTRICT,state:22,no,True +district:650,Bemetara,650,http://www.wikidata.org/entity/Q16254159,DISTRICT,state:22,no,True +district:734,Gaurela-Pendra-Marwahi,734,http://www.wikidata.org/entity/Q96584972,DISTRICT,state:22,no,True +district:759,Khairagarh-Chhuikhadan-Gandai,759,http://www.wikidata.org/entity/Q113485010,DISTRICT,state:22,no,True +district:760,Manendragarh-Chirmiri-Bharatpur,760,http://www.wikidata.org/entity/Q108427451,DISTRICT,state:22,no,True +district:761,Mohla-Manpur-Ambagarh Chowki,761,http://www.wikidata.org/entity/Q108569901,DISTRICT,state:22,no,True +district:762,Sakti,762,http://www.wikidata.org/entity/Q108569905,DISTRICT,state:22,no,True +district:763,Sarangarh-Bilaigarh,763,http://www.wikidata.org/entity/Q108569900,DISTRICT,state:22,no,True +district:390,Anuppur,390,http://www.wikidata.org/entity/Q2299093,DISTRICT,state:23,no,True +district:391,Ashoknagar,391,http://www.wikidata.org/entity/Q2246416,DISTRICT,state:23,no,True +district:392,Balaghat,392,http://www.wikidata.org/entity/Q641904,DISTRICT,state:23,no,True +district:393,Barwani,393,http://www.wikidata.org/entity/Q2126754,DISTRICT,state:23,no,True +district:394,Betul,394,http://www.wikidata.org/entity/Q1815279,DISTRICT,state:23,no,True +district:395,Bhind,395,http://www.wikidata.org/entity/Q2341700,DISTRICT,state:23,no,True +district:396,Bhopal,396,http://www.wikidata.org/entity/Q1797245,DISTRICT,state:23,no,True +district:397,Burhanpur,397,http://www.wikidata.org/entity/Q2125592,DISTRICT,state:23,no,True +district:398,Chhatarpur,398,http://www.wikidata.org/entity/Q2449785,DISTRICT,state:23,no,True +district:399,Chhindwara,399,http://www.wikidata.org/entity/Q1986096,DISTRICT,state:23,no,True +district:400,Damoh,400,http://www.wikidata.org/entity/Q2479331,DISTRICT,state:23,no,True +district:401,Datia,401,http://www.wikidata.org/entity/Q2206266,DISTRICT,state:23,no,True +district:402,Dewas,402,http://www.wikidata.org/entity/Q2025998,DISTRICT,state:23,no,True +district:403,Dhar,403,http://www.wikidata.org/entity/Q2299069,DISTRICT,state:23,no,True +district:404,Dindori,404,http://www.wikidata.org/entity/Q2398551,DISTRICT,state:23,no,True +district:405,Khandwa,405,http://www.wikidata.org/entity/Q2085436,DISTRICT,state:23,no,True +district:406,Guna,406,http://www.wikidata.org/entity/Q930027,DISTRICT,state:23,no,True +district:407,Gwalior,407,http://www.wikidata.org/entity/Q2085310,DISTRICT,state:23,no,True +district:408,Harda,408,http://www.wikidata.org/entity/Q2173003,DISTRICT,state:23,no,True +district:409,Narmadapuram,409,http://www.wikidata.org/entity/Q620801,DISTRICT,state:23,no,True +district:410,Indore,410,http://www.wikidata.org/entity/Q742938,DISTRICT,state:23,no,True +district:411,Jabalpur,411,http://www.wikidata.org/entity/Q632093,DISTRICT,state:23,no,True +district:412,Jhabua,412,http://www.wikidata.org/entity/Q2085336,DISTRICT,state:23,no,True +district:413,Katni,413,http://www.wikidata.org/entity/Q746441,DISTRICT,state:23,no,True +district:414,Khargone,414,http://www.wikidata.org/entity/Q2273900,DISTRICT,state:23,no,True +district:415,Mandla,415,http://www.wikidata.org/entity/Q2341670,DISTRICT,state:23,no,True +district:416,Mandsaur,416,http://www.wikidata.org/entity/Q1870014,DISTRICT,state:23,no,True +district:417,Morena,417,http://www.wikidata.org/entity/Q2341467,DISTRICT,state:23,no,True +district:418,Narsinghpur,418,http://www.wikidata.org/entity/Q2341616,DISTRICT,state:23,no,True +district:419,Neemuch,419,http://www.wikidata.org/entity/Q2341713,DISTRICT,state:23,no,True +district:420,Panna,420,http://www.wikidata.org/entity/Q2341630,DISTRICT,state:23,no,True +district:421,Raisen,421,http://www.wikidata.org/entity/Q1815223,DISTRICT,state:23,no,True +district:422,Rajgarh,422,http://www.wikidata.org/entity/Q1833306,DISTRICT,state:23,no,True +district:423,Ratlam,423,http://www.wikidata.org/entity/Q2299164,DISTRICT,state:23,no,True +district:424,Rewa,424,http://www.wikidata.org/entity/Q526862,DISTRICT,state:23,no,True +district:425,Sagar,425,http://www.wikidata.org/entity/Q2085421,DISTRICT,state:23,no,True +district:426,Satna,426,http://www.wikidata.org/entity/Q2577924,DISTRICT,state:23,no,True +district:427,Sehore,427,http://www.wikidata.org/entity/Q2299029,DISTRICT,state:23,no,True +district:428,Seoni,428,http://www.wikidata.org/entity/Q2221184,DISTRICT,state:23,no,True +district:429,Shahdol,429,http://www.wikidata.org/entity/Q2085464,DISTRICT,state:23,no,True +district:430,Shajapur,430,http://www.wikidata.org/entity/Q2449803,DISTRICT,state:23,no,True +district:431,Sheopur,431,http://www.wikidata.org/entity/Q620105,DISTRICT,state:23,no,True +district:432,Shivpuri,432,http://www.wikidata.org/entity/Q2299042,DISTRICT,state:23,no,True +district:433,Sidhi,433,http://www.wikidata.org/entity/Q2449793,DISTRICT,state:23,no,True +district:434,Tikamgarh,434,http://www.wikidata.org/entity/Q2449760,DISTRICT,state:23,no,True +district:435,Ujjain,435,http://www.wikidata.org/entity/Q892641,DISTRICT,state:23,no,True +district:436,Umaria,436,http://www.wikidata.org/entity/Q620297,DISTRICT,state:23,no,True +district:437,Vidisha,437,http://www.wikidata.org/entity/Q1815253,DISTRICT,state:23,no,True +district:638,Singrauli,638,http://www.wikidata.org/entity/Q2668638,DISTRICT,state:23,no,True +district:639,Alirajpur,639,http://www.wikidata.org/entity/Q2667586,DISTRICT,state:23,no,True +district:667,Agar Malwa,667,http://www.wikidata.org/entity/Q15732396,DISTRICT,state:23,no,True +district:722,Niwari,722,http://www.wikidata.org/entity/Q63563797,DISTRICT,state:23,no,True +district:766,Mauganj,766,http://www.wikidata.org/entity/Q122417864,DISTRICT,state:23,no,True +district:784,Maihar,784,http://www.wikidata.org/entity/Q111675213,DISTRICT,state:23,no,True +district:785,Pandhurna,785,http://www.wikidata.org/entity/Q123286184,DISTRICT,state:23,no,True +district:438,Ahmedabad,438,http://www.wikidata.org/entity/Q401686,DISTRICT,state:24,no,True +district:439,Amreli,439,http://www.wikidata.org/entity/Q257946,DISTRICT,state:24,no,True +district:440,Anand,440,http://www.wikidata.org/entity/Q485683,DISTRICT,state:24,no,True +district:441,Banaskantha,441,http://www.wikidata.org/entity/Q806125,DISTRICT,state:24,no,True +district:442,Bharuch,442,http://www.wikidata.org/entity/Q854900,DISTRICT,state:24,no,True +district:443,Bhavnagar,443,http://www.wikidata.org/entity/Q854963,DISTRICT,state:24,no,True +district:444,Dang,444,http://www.wikidata.org/entity/Q1135616,DISTRICT,state:24,no,True +district:445,Dahod,445,http://www.wikidata.org/entity/Q186518,DISTRICT,state:24,no,True +district:446,Gandhinagar,446,http://www.wikidata.org/entity/Q1772860,DISTRICT,state:24,no,True +district:447,Jamnagar,447,http://www.wikidata.org/entity/Q2982118,DISTRICT,state:24,no,True +district:448,Junagadh,448,http://www.wikidata.org/entity/Q1797344,DISTRICT,state:24,no,True +district:449,Kutch,449,http://www.wikidata.org/entity/Q1063417,DISTRICT,state:24,no,True +district:450,Kheda,450,http://www.wikidata.org/entity/Q1755463,DISTRICT,state:24,no,True +district:451,Mehsana,451,http://www.wikidata.org/entity/Q2019694,DISTRICT,state:24,no,True +district:452,Narmada,452,http://www.wikidata.org/entity/Q1797230,DISTRICT,state:24,no,True +district:453,Navsari,453,http://www.wikidata.org/entity/Q1797349,DISTRICT,state:24,no,True +district:454,Panchmahal,454,http://www.wikidata.org/entity/Q1781463,DISTRICT,state:24,no,True +district:455,Patan,455,http://www.wikidata.org/entity/Q1815269,DISTRICT,state:24,no,True +district:456,Porbandar,456,http://www.wikidata.org/entity/Q1772815,DISTRICT,state:24,no,True +district:457,Rajkot,457,http://www.wikidata.org/entity/Q1815245,DISTRICT,state:24,no,True +district:458,Sabarkantha,458,http://www.wikidata.org/entity/Q1772856,DISTRICT,state:24,no,True +district:459,Surat,459,http://www.wikidata.org/entity/Q1797317,DISTRICT,state:24,no,True +district:460,Surendranagar,460,http://www.wikidata.org/entity/Q237535,DISTRICT,state:24,no,True +district:461,Vadodara,461,http://www.wikidata.org/entity/Q578285,DISTRICT,state:24,no,True +district:462,Valsad,462,http://www.wikidata.org/entity/Q1946743,DISTRICT,state:24,no,True +district:641,Tapi,641,http://www.wikidata.org/entity/Q670165,DISTRICT,state:24,no,True +district:668,Chhota Udaipur,668,http://www.wikidata.org/entity/Q5979243,DISTRICT,state:24,no,True +district:669,Mahisagar,669,http://www.wikidata.org/entity/Q5706885,DISTRICT,state:24,no,True +district:672,Aravalli,672,http://www.wikidata.org/entity/Q12175285,DISTRICT,state:24,no,True +district:673,Morbi,673,http://www.wikidata.org/entity/Q5979727,DISTRICT,state:24,no,True +district:674,Devbhumi Dwarka,674,http://www.wikidata.org/entity/Q14594717,DISTRICT,state:24,no,True +district:675,Gir Somnath,675,http://www.wikidata.org/entity/Q15244465,DISTRICT,state:24,no,True +district:676,Botad,676,http://www.wikidata.org/entity/Q14505072,DISTRICT,state:24,no,True +district:789,Vav-Tharad,789,http://www.wikidata.org/entity/Q131621560,DISTRICT,state:24,no,True +district:466,Ahilyanagar,466,http://www.wikidata.org/entity/Q401744,DISTRICT,state:27,no,True +district:467,Akola,467,http://www.wikidata.org/entity/Q520510,DISTRICT,state:27,no,True +district:468,Amravati,468,http://www.wikidata.org/entity/Q1771774,DISTRICT,state:27,no,True +district:469,Aurangabad,469,http://www.wikidata.org/entity/Q592942,DISTRICT,state:27,no,True +district:470,Beed,470,http://www.wikidata.org/entity/Q814037,DISTRICT,state:27,no,True +district:471,Bhandara,471,http://www.wikidata.org/entity/Q1813857,DISTRICT,state:27,no,True +district:472,Buldhana,472,http://www.wikidata.org/entity/Q47929,DISTRICT,state:27,no,True +district:473,Chandrapur,473,http://www.wikidata.org/entity/Q1797274,DISTRICT,state:27,no,True +district:474,Dhule,474,http://www.wikidata.org/entity/Q1797383,DISTRICT,state:27,no,True +district:475,Gadchiroli,475,http://www.wikidata.org/entity/Q1804847,DISTRICT,state:27,no,True +district:476,Gondia,476,http://www.wikidata.org/entity/Q1917227,DISTRICT,state:27,no,True +district:477,Hingoli,477,http://www.wikidata.org/entity/Q2087615,DISTRICT,state:27,no,True +district:478,Jalgaon,478,http://www.wikidata.org/entity/Q1797291,DISTRICT,state:27,no,True +district:479,Jalna,479,http://www.wikidata.org/entity/Q1804863,DISTRICT,state:27,no,True +district:480,Kolhapur,480,http://www.wikidata.org/entity/Q1797312,DISTRICT,state:27,no,True +district:481,Latur,481,http://www.wikidata.org/entity/Q1948713,DISTRICT,state:27,no,True +district:482,Mumbai City,482,http://www.wikidata.org/entity/Q2341660,DISTRICT,state:27,no,True +district:483,Mumbai Suburban,483,http://www.wikidata.org/entity/Q2085374,DISTRICT,state:27,no,True +district:484,Nagpur,484,http://www.wikidata.org/entity/Q1797367,DISTRICT,state:27,no,True +district:485,Nanded,485,http://www.wikidata.org/entity/Q692389,DISTRICT,state:27,no,True +district:486,Nandurbar,486,http://www.wikidata.org/entity/Q1623525,DISTRICT,state:27,no,True +district:487,Nashik,487,http://www.wikidata.org/entity/Q1797269,DISTRICT,state:27,no,True +district:488,Dharashiv,488,http://www.wikidata.org/entity/Q1647186,DISTRICT,state:27,no,True +district:489,Parbhani,489,http://www.wikidata.org/entity/Q1797389,DISTRICT,state:27,no,True +district:490,Pune,490,http://www.wikidata.org/entity/Q1797336,DISTRICT,state:27,no,True +district:491,Raigad,491,http://www.wikidata.org/entity/Q2019683,DISTRICT,state:27,no,True +district:492,Ratnagiri,492,http://www.wikidata.org/entity/Q1771768,DISTRICT,state:27,no,True +district:493,Sangli,493,http://www.wikidata.org/entity/Q1425060,DISTRICT,state:27,no,True +district:494,Satara,494,http://www.wikidata.org/entity/Q1135612,DISTRICT,state:27,no,True +district:495,Sindhudurg,495,http://www.wikidata.org/entity/Q768332,DISTRICT,state:27,no,True +district:496,Solapur,496,http://www.wikidata.org/entity/Q1797263,DISTRICT,state:27,no,True +district:497,Thane,497,http://www.wikidata.org/entity/Q943099,DISTRICT,state:27,no,True +district:498,Wardha,498,http://www.wikidata.org/entity/Q980608,DISTRICT,state:27,no,True +district:499,Washim,499,http://www.wikidata.org/entity/Q1804858,DISTRICT,state:27,no,True +district:500,Yavatmal,500,http://www.wikidata.org/entity/Q1804852,DISTRICT,state:27,no,True +district:665,Palghar,665,http://www.wikidata.org/entity/Q18003119,DISTRICT,state:27,no,True +district:502,Anantapuramu,502,http://www.wikidata.org/entity/Q15212,DISTRICT,state:28,no,True +district:503,Chittoor,503,http://www.wikidata.org/entity/Q15213,DISTRICT,state:28,no,True +district:504,YSR Kadapa,504,http://www.wikidata.org/entity/Q15342,DISTRICT,state:28,no,True +district:505,East Godavari,505,http://www.wikidata.org/entity/Q15338,DISTRICT,state:28,no,True +district:506,Guntur,506,http://www.wikidata.org/entity/Q15341,DISTRICT,state:28,no,True +district:510,Krishna,510,http://www.wikidata.org/entity/Q15382,DISTRICT,state:28,no,True +district:511,Kurnool,511,http://www.wikidata.org/entity/Q15381,DISTRICT,state:28,no,True +district:515,Sri Potti Sri Ramulu Nellore,515,http://www.wikidata.org/entity/Q15383,DISTRICT,state:28,no,True +district:517,Prakasam,517,http://www.wikidata.org/entity/Q15390,DISTRICT,state:28,no,True +district:519,Srikakulam,519,http://www.wikidata.org/entity/Q15395,DISTRICT,state:28,no,True +district:520,Visakhapatnam,520,http://www.wikidata.org/entity/Q15394,DISTRICT,state:28,no,True +district:521,Vizianagaram,521,http://www.wikidata.org/entity/Q15392,DISTRICT,state:28,no,True +district:523,West Godavari,523,http://www.wikidata.org/entity/Q15404,DISTRICT,state:28,no,True +district:743,Parvathipuram Manyam,743,http://www.wikidata.org/entity/Q110714856,DISTRICT,state:28,no,True +district:744,Anakapalli,744,http://www.wikidata.org/entity/Q110714857,DISTRICT,state:28,no,True +district:745,Alluri Sitharama Raju,745,http://www.wikidata.org/entity/Q110714850,DISTRICT,state:28,no,True +district:746,Kakinada,746,http://www.wikidata.org/entity/Q110714860,DISTRICT,state:28,no,True +district:747,Dr. B.R. Ambedkar Konaseema,747,http://www.wikidata.org/entity/Q110714859,DISTRICT,state:28,no,True +district:748,Eluru,748,http://www.wikidata.org/entity/Q110714851,DISTRICT,state:28,no,True +district:749,NTR,749,http://www.wikidata.org/entity/Q110876763,DISTRICT,state:28,no,True +district:750,Bapatla,750,http://www.wikidata.org/entity/Q110876712,DISTRICT,state:28,no,True +district:751,Palnadu,751,http://www.wikidata.org/entity/Q110714862,DISTRICT,state:28,no,True +district:752,Tirupati,752,http://www.wikidata.org/entity/Q110714853,DISTRICT,state:28,no,True +district:753,Annamayya,753,http://www.wikidata.org/entity/Q110714854,DISTRICT,state:28,no,True +district:754,Sri Sathya Sai,754,http://www.wikidata.org/entity/Q110714863,DISTRICT,state:28,no,True +district:755,Nandyal,755,http://www.wikidata.org/entity/Q110714861,DISTRICT,state:28,no,True +district:524,Bagalkot,524,http://www.wikidata.org/entity/Q1910231,DISTRICT,state:29,no,True +district:525,Bengaluru Urban,525,http://www.wikidata.org/entity/Q806463,DISTRICT,state:29,no,True +district:526,Bengaluru North,526,http://www.wikidata.org/entity/Q806464,DISTRICT,state:29,no,True +district:527,Belagavi,527,http://www.wikidata.org/entity/Q815464,DISTRICT,state:29,no,True +district:528,Ballari,528,http://www.wikidata.org/entity/Q1791926,DISTRICT,state:29,no,True +district:529,Bidar,529,http://www.wikidata.org/entity/Q1790568,DISTRICT,state:29,no,True +district:530,Vijaypura,530,http://www.wikidata.org/entity/Q83108,DISTRICT,state:29,no,True +district:531,Chamarajanagar,531,http://www.wikidata.org/entity/Q862912,DISTRICT,state:29,no,True +district:532,Chikmagalur,532,http://www.wikidata.org/entity/Q743077,DISTRICT,state:29,no,True +district:533,Chitradurga,533,http://www.wikidata.org/entity/Q165264,DISTRICT,state:29,no,True +district:534,Dakshina Kannada,534,http://www.wikidata.org/entity/Q950571,DISTRICT,state:29,no,True +district:535,Davanagere,535,http://www.wikidata.org/entity/Q1863214,DISTRICT,state:29,no,True +district:536,Dharwad,536,http://www.wikidata.org/entity/Q1790904,DISTRICT,state:29,no,True +district:537,Gadag,537,http://www.wikidata.org/entity/Q2353931,DISTRICT,state:29,no,True +district:538,Kalaburgi,538,http://www.wikidata.org/entity/Q2641873,DISTRICT,state:29,no,True +district:539,Hassan,539,http://www.wikidata.org/entity/Q956732,DISTRICT,state:29,no,True +district:540,Haveri,540,http://www.wikidata.org/entity/Q765481,DISTRICT,state:29,no,True +district:541,Kodagu,541,http://www.wikidata.org/entity/Q1553185,DISTRICT,state:29,no,True +district:542,Kolar,542,http://www.wikidata.org/entity/Q2509866,DISTRICT,state:29,no,True +district:543,Koppal,543,http://www.wikidata.org/entity/Q956387,DISTRICT,state:29,no,True +district:544,Mandya,544,http://www.wikidata.org/entity/Q2768290,DISTRICT,state:29,no,True +district:545,Mysuru,545,http://www.wikidata.org/entity/Q591781,DISTRICT,state:29,no,True +district:546,Raichur,546,http://www.wikidata.org/entity/Q1430830,DISTRICT,state:29,no,True +district:547,Shimoga,547,http://www.wikidata.org/entity/Q2981389,DISTRICT,state:29,no,True +district:548,Tumkur,548,http://www.wikidata.org/entity/Q1301635,DISTRICT,state:29,no,True +district:549,Udupi,549,http://www.wikidata.org/entity/Q1483337,DISTRICT,state:29,no,True +district:550,Uttara Kannada,550,http://www.wikidata.org/entity/Q579205,DISTRICT,state:29,no,True +district:630,Chikkaballapura,630,http://www.wikidata.org/entity/Q1072629,DISTRICT,state:29,no,True +district:631,Bengaluru South,631,http://www.wikidata.org/entity/Q427679,DISTRICT,state:29,no,True +district:635,Yadgir,635,http://www.wikidata.org/entity/Q1786949,DISTRICT,state:29,no,True +district:738,Vijayanagara,738,http://www.wikidata.org/entity/Q104876850,DISTRICT,state:29,no,True +district:27,Amritsar,27,http://www.wikidata.org/entity/Q202822,DISTRICT,state:3,no,True +district:28,Bathinda,28,http://www.wikidata.org/entity/Q172488,DISTRICT,state:3,no,True +district:29,Faridkot,29,http://www.wikidata.org/entity/Q172494,DISTRICT,state:3,no,True +district:30,Fatehgarh Sahib,30,http://www.wikidata.org/entity/Q172485,DISTRICT,state:3,no,True +district:31,Firozpur,31,http://www.wikidata.org/entity/Q172385,DISTRICT,state:3,no,True +district:32,Gurdaspur,32,http://www.wikidata.org/entity/Q146708,DISTRICT,state:3,no,True +district:33,Hoshiarpur,33,http://www.wikidata.org/entity/Q304800,DISTRICT,state:3,no,True +district:34,Jalandhar,34,http://www.wikidata.org/entity/Q1817425,DISTRICT,state:3,no,True +district:35,Kapurthala,35,http://www.wikidata.org/entity/Q172363,DISTRICT,state:3,no,True +district:36,Ludhiana,36,http://www.wikidata.org/entity/Q172482,DISTRICT,state:3,no,True +district:37,Mansa,37,http://www.wikidata.org/entity/Q172387,DISTRICT,state:3,no,True +district:38,Moga,38,http://www.wikidata.org/entity/Q1946896,DISTRICT,state:3,no,True +district:39,Sri Muktsar Sahib,39,http://www.wikidata.org/entity/Q1947359,DISTRICT,state:3,no,True +district:40,Shaheed Bhagat Singh Nagar,40,http://www.wikidata.org/entity/Q202710,DISTRICT,state:3,no,True +district:41,Patiala,41,http://www.wikidata.org/entity/Q172391,DISTRICT,state:3,no,True +district:42,Rupnagar,42,http://www.wikidata.org/entity/Q196508,DISTRICT,state:3,no,True +district:43,Sangrur,43,http://www.wikidata.org/entity/Q1945515,DISTRICT,state:3,no,True +district:605,Barnala,605,http://www.wikidata.org/entity/Q2353293,DISTRICT,state:3,no,True +district:608,Sahibzada Ajit Singh Nagar,608,http://www.wikidata.org/entity/Q2037672,DISTRICT,state:3,no,True +district:609,Tarn Taran,609,http://www.wikidata.org/entity/Q2298993,DISTRICT,state:3,no,True +district:651,Fazilka,651,http://www.wikidata.org/entity/Q188702,DISTRICT,state:3,no,True +district:662,Pathankot,662,http://www.wikidata.org/entity/Q172269,DISTRICT,state:3,no,True +district:737,Malerkotla,737,http://www.wikidata.org/entity/Q107016021,DISTRICT,state:3,no,True +district:551,North Goa,551,http://www.wikidata.org/entity/Q108234,DISTRICT,state:30,no,True +district:552,South Goa,552,http://www.wikidata.org/entity/Q108244,DISTRICT,state:30,no,True +district:553,Lakshadweep,553,http://www.wikidata.org/entity/Q10784153,DISTRICT,state:31,no,True +district:554,Alappuzha,554,http://www.wikidata.org/entity/Q928959,DISTRICT,state:32,no,True +district:555,Ernakulam,555,http://www.wikidata.org/entity/Q1356097,DISTRICT,state:32,no,True +district:556,Idukki,556,http://www.wikidata.org/entity/Q301821,DISTRICT,state:32,no,True +district:557,Kannur,557,http://www.wikidata.org/entity/Q2980652,DISTRICT,state:32,no,True +district:558,Kasaragod,558,http://www.wikidata.org/entity/Q1419703,DISTRICT,state:32,no,True +district:559,Kollam,559,http://www.wikidata.org/entity/Q1356124,DISTRICT,state:32,no,True +district:560,Kottayam,560,http://www.wikidata.org/entity/Q1353354,DISTRICT,state:32,no,True +district:561,Kozhikode,561,http://www.wikidata.org/entity/Q1142979,DISTRICT,state:32,no,True +district:562,Malappuram,562,http://www.wikidata.org/entity/Q1030918,DISTRICT,state:32,no,True +district:563,Palakkad,563,http://www.wikidata.org/entity/Q1535742,DISTRICT,state:32,no,True +district:564,Pathanamthitta,564,http://www.wikidata.org/entity/Q634935,DISTRICT,state:32,no,True +district:565,Thiruvananthapuram,565,http://www.wikidata.org/entity/Q162612,DISTRICT,state:32,no,True +district:566,Thrissur,566,http://www.wikidata.org/entity/Q2429655,DISTRICT,state:32,no,True +district:567,Wayanad,567,http://www.wikidata.org/entity/Q1364427,DISTRICT,state:32,no,True +district:568,Chennai,568,http://www.wikidata.org/entity/Q15116,DISTRICT,state:33,no,True +district:569,Coimbatore,569,http://www.wikidata.org/entity/Q15136,DISTRICT,state:33,no,True +district:570,Cuddalore,570,http://www.wikidata.org/entity/Q15150,DISTRICT,state:33,no,True +district:571,Dharmapuri,571,http://www.wikidata.org/entity/Q15152,DISTRICT,state:33,no,True +district:572,Dindigul,572,http://www.wikidata.org/entity/Q15154,DISTRICT,state:33,no,True +district:573,Erode,573,http://www.wikidata.org/entity/Q15155,DISTRICT,state:33,no,True +district:574,Kanchipuram,574,http://www.wikidata.org/entity/Q15157,DISTRICT,state:33,no,True +district:575,Kanniyakumari,575,http://www.wikidata.org/entity/Q15158,DISTRICT,state:33,no,True +district:576,Karur,576,http://www.wikidata.org/entity/Q15182,DISTRICT,state:33,no,True +district:577,Krishnagiri,577,http://www.wikidata.org/entity/Q15183,DISTRICT,state:33,no,True +district:578,Madurai,578,http://www.wikidata.org/entity/Q15184,DISTRICT,state:33,no,True +district:579,Nagapattinam,579,http://www.wikidata.org/entity/Q15185,DISTRICT,state:33,no,True +district:580,Namakkal,580,http://www.wikidata.org/entity/Q15187,DISTRICT,state:33,no,True +district:581,Perambalur,581,http://www.wikidata.org/entity/Q15186,DISTRICT,state:33,no,True +district:582,Pudukkottai,582,http://www.wikidata.org/entity/Q15190,DISTRICT,state:33,no,True +district:583,Ramanathapuram,583,http://www.wikidata.org/entity/Q15191,DISTRICT,state:33,no,True +district:584,Salem,584,http://www.wikidata.org/entity/Q15192,DISTRICT,state:33,no,True +district:585,Sivaganga,585,http://www.wikidata.org/entity/Q15195,DISTRICT,state:33,no,True +district:586,Thanjavur,586,http://www.wikidata.org/entity/Q15194,DISTRICT,state:33,no,True +district:587,Nilgiris,587,http://www.wikidata.org/entity/Q15188,DISTRICT,state:33,no,True +district:588,Theni,588,http://www.wikidata.org/entity/Q15196,DISTRICT,state:33,no,True +district:589,Tiruvallur,589,http://www.wikidata.org/entity/Q15204,DISTRICT,state:33,no,True +district:590,Tiruvarur,590,http://www.wikidata.org/entity/Q15197,DISTRICT,state:33,no,True +district:591,Tiruchirappalli,591,http://www.wikidata.org/entity/Q15201,DISTRICT,state:33,no,True +district:592,Tirunelveli,592,http://www.wikidata.org/entity/Q15200,DISTRICT,state:33,no,True +district:593,Tiruvannamalai,593,http://www.wikidata.org/entity/Q15207,DISTRICT,state:33,no,True +district:594,Thoothukudi,594,http://www.wikidata.org/entity/Q15198,DISTRICT,state:33,no,True +district:595,Vellore,595,http://www.wikidata.org/entity/Q15206,DISTRICT,state:33,no,True +district:596,Viluppuram,596,http://www.wikidata.org/entity/Q15205,DISTRICT,state:33,no,True +district:597,Virudhunagar,597,http://www.wikidata.org/entity/Q15209,DISTRICT,state:33,no,True +district:610,Ariyalur,610,http://www.wikidata.org/entity/Q15112,DISTRICT,state:33,no,True +district:634,Tiruppur,634,http://www.wikidata.org/entity/Q15202,DISTRICT,state:33,no,True +district:729,Kallakurichi,729,http://www.wikidata.org/entity/Q60493360,DISTRICT,state:33,no,True +district:730,Chengalpattu,730,http://www.wikidata.org/entity/Q65976177,DISTRICT,state:33,no,True +district:731,Ranipet,731,http://www.wikidata.org/entity/Q66659623,DISTRICT,state:33,no,True +district:732,Tirupattur,732,http://www.wikidata.org/entity/Q66659621,DISTRICT,state:33,no,True +district:733,Tenkasi,733,http://www.wikidata.org/entity/Q75094121,DISTRICT,state:33,no,True +district:735,Mayiladuthurai,735,http://www.wikidata.org/entity/Q89918869,DISTRICT,state:33,no,True +district:598,Karaikal,598,http://www.wikidata.org/entity/Q639264,DISTRICT,state:34,no,True +district:600,Puducherry,600,http://www.wikidata.org/entity/Q984035,DISTRICT,state:34,no,True +district:602,South Andaman,602,http://www.wikidata.org/entity/Q796979,DISTRICT,state:35,no,True +district:603,Nicobar,603,http://www.wikidata.org/entity/Q797295,DISTRICT,state:35,no,True +district:632,North and Middle Andaman,632,http://www.wikidata.org/entity/Q796983,DISTRICT,state:35,no,True +district:501,Adilabad,501,http://www.wikidata.org/entity/Q15211,DISTRICT,state:36,no,True +district:507,Hyderabad,507,http://www.wikidata.org/entity/Q15340,DISTRICT,state:36,no,True +district:508,Karimnagar,508,http://www.wikidata.org/entity/Q15373,DISTRICT,state:36,no,True +district:509,Khammam,509,http://www.wikidata.org/entity/Q15371,DISTRICT,state:36,no,True +district:512,Mahabubnagar,512,http://www.wikidata.org/entity/Q15380,DISTRICT,state:36,no,True +district:513,Medak,513,http://www.wikidata.org/entity/Q15386,DISTRICT,state:36,no,True +district:514,Nalgonda,514,http://www.wikidata.org/entity/Q15384,DISTRICT,state:36,no,True +district:516,Nizamabad,516,http://www.wikidata.org/entity/Q15391,DISTRICT,state:36,no,True +district:518,Ranga Reddy,518,http://www.wikidata.org/entity/Q15388,DISTRICT,state:36,no,True +district:522,Warangal,522,http://www.wikidata.org/entity/Q15399,DISTRICT,state:36,no,True +district:680,Nirmal,680,http://www.wikidata.org/entity/Q28169750,DISTRICT,state:36,no,True +district:681,Jagtial,681,http://www.wikidata.org/entity/Q28169780,DISTRICT,state:36,no,True +district:682,Peddapalli,682,http://www.wikidata.org/entity/Q27614797,DISTRICT,state:36,no,True +district:683,Rajanna Sircilla,683,http://www.wikidata.org/entity/Q28172781,DISTRICT,state:36,no,True +district:684,Mancherial,684,http://www.wikidata.org/entity/Q28169747,DISTRICT,state:36,no,True +district:685,Kamareddy,685,http://www.wikidata.org/entity/Q27956125,DISTRICT,state:36,no,True +district:686,Hanamkonda,686,http://www.wikidata.org/entity/Q213077,DISTRICT,state:36,no,True +district:687,Jayashankar Bhupalpally,687,http://www.wikidata.org/entity/Q28169775,DISTRICT,state:36,no,True +district:688,Mahabubabad,688,http://www.wikidata.org/entity/Q28169761,DISTRICT,state:36,no,True +district:689,Jangaon,689,http://www.wikidata.org/entity/Q28170170,DISTRICT,state:36,no,True +district:690,Bhadradri Kothagudem,690,http://www.wikidata.org/entity/Q28169767,DISTRICT,state:36,no,True +district:691,Sangareddy,691,http://www.wikidata.org/entity/Q28169753,DISTRICT,state:36,no,True +district:692,Siddipet,692,http://www.wikidata.org/entity/Q28169756,DISTRICT,state:36,no,True +district:693,Wanaparthy,693,http://www.wikidata.org/entity/Q28172504,DISTRICT,state:36,no,True +district:694,Nagarkurnool,694,http://www.wikidata.org/entity/Q28169773,DISTRICT,state:36,no,True +district:695,Jogulamba Gadwal,695,http://www.wikidata.org/entity/Q27897618,DISTRICT,state:36,no,True +district:696,Suryapet,696,http://www.wikidata.org/entity/Q28169770,DISTRICT,state:36,no,True +district:697,Yadadri Bhuvanagiri,697,http://www.wikidata.org/entity/Q28169764,DISTRICT,state:36,no,True +district:698,Vikarabad,698,http://www.wikidata.org/entity/Q28170173,DISTRICT,state:36,no,True +district:699,Kumaram Bheem Asifabad,699,http://www.wikidata.org/entity/Q28170184,DISTRICT,state:36,no,True +district:700,Medchal-Malkajgiri,700,http://www.wikidata.org/entity/Q27614841,DISTRICT,state:36,no,True +district:720,Mulugu,720,http://www.wikidata.org/entity/Q61746006,DISTRICT,state:36,no,True +district:721,Narayanpet,721,http://www.wikidata.org/entity/Q61746013,DISTRICT,state:36,no,True +district:6,Kargil,6,http://www.wikidata.org/entity/Q1650798,DISTRICT,state:37,no,True +district:9,Leh,9,http://www.wikidata.org/entity/Q1921210,DISTRICT,state:37,no,True +district:463,Daman,463,http://www.wikidata.org/entity/Q1158197,DISTRICT,state:38,no,True +district:464,Diu,464,http://www.wikidata.org/entity/Q2552347,DISTRICT,state:38,no,True +district:465,Dadra and Nagar Haveli,465,http://www.wikidata.org/entity/Q46107,DISTRICT,state:38,no,True +district:44,Chandigarh,44,http://www.wikidata.org/entity/Q5071071,DISTRICT,state:4,no,True +district:45,Almora,45,http://www.wikidata.org/entity/Q1805066,DISTRICT,state:5,no,True +district:46,Bageshwar,46,http://www.wikidata.org/entity/Q1815313,DISTRICT,state:5,no,True +district:47,Chamoli,47,http://www.wikidata.org/entity/Q1797372,DISTRICT,state:5,no,True +district:48,Champawat,48,http://www.wikidata.org/entity/Q288278,DISTRICT,state:5,no,True +district:49,Dehradun,49,http://www.wikidata.org/entity/Q1815740,DISTRICT,state:5,no,True +district:50,Haridwar,50,http://www.wikidata.org/entity/Q2270438,DISTRICT,state:5,no,True +district:51,Nainital,51,http://www.wikidata.org/entity/Q1797306,DISTRICT,state:5,no,True +district:52,Pauri Garhwal,52,http://www.wikidata.org/entity/Q2085474,DISTRICT,state:5,no,True +district:53,Pithoragarh,53,http://www.wikidata.org/entity/Q1945425,DISTRICT,state:5,no,True +district:54,Rudraprayag,54,http://www.wikidata.org/entity/Q1805059,DISTRICT,state:5,no,True +district:55,Tehri Garhwal,55,http://www.wikidata.org/entity/Q1357107,DISTRICT,state:5,no,True +district:56,Udham Singh Nagar,56,http://www.wikidata.org/entity/Q1805082,DISTRICT,state:5,no,True +district:57,Uttarkashi,57,http://www.wikidata.org/entity/Q1773437,DISTRICT,state:5,no,True +district:58,Ambala,58,http://www.wikidata.org/entity/Q2086226,DISTRICT,state:6,no,True +district:59,Bhiwani,59,http://www.wikidata.org/entity/Q1852857,DISTRICT,state:6,no,True +district:60,Faridabad,60,http://www.wikidata.org/entity/Q2086173,DISTRICT,state:6,no,True +district:604,Nuh,604,http://www.wikidata.org/entity/Q2216696,DISTRICT,state:6,no,True +district:61,Fatehabad,61,http://www.wikidata.org/entity/Q2301753,DISTRICT,state:6,no,True +district:619,Palwal,619,http://www.wikidata.org/entity/Q2724926,DISTRICT,state:6,no,True +district:62,Gurugram,62,http://www.wikidata.org/entity/Q1815766,DISTRICT,state:6,no,True +district:63,Hisar,63,http://www.wikidata.org/entity/Q1815773,DISTRICT,state:6,no,True +district:64,Jhajjar,64,http://www.wikidata.org/entity/Q1948260,DISTRICT,state:6,no,True +district:65,Jind,65,http://www.wikidata.org/entity/Q268605,DISTRICT,state:6,no,True +district:66,Kaithal,66,http://www.wikidata.org/entity/Q614037,DISTRICT,state:6,no,True +district:67,Karnal,67,http://www.wikidata.org/entity/Q607915,DISTRICT,state:6,no,True +district:68,Kurukshetra,68,http://www.wikidata.org/entity/Q980118,DISTRICT,state:6,no,True +district:69,Mahendragarh,69,http://www.wikidata.org/entity/Q684019,DISTRICT,state:6,no,True +district:70,Panchkula,70,http://www.wikidata.org/entity/Q1898143,DISTRICT,state:6,no,True +district:701,Charkhi Dadri,701,http://www.wikidata.org/entity/Q28172110,DISTRICT,state:6,no,True +district:71,Panipat,71,http://www.wikidata.org/entity/Q2086163,DISTRICT,state:6,no,True +district:72,Rewari,72,http://www.wikidata.org/entity/Q2301759,DISTRICT,state:6,no,True +district:73,Rohtak,73,http://www.wikidata.org/entity/Q967388,DISTRICT,state:6,no,True +district:74,Sirsa,74,http://www.wikidata.org/entity/Q526101,DISTRICT,state:6,no,True +district:75,Sonipat,75,http://www.wikidata.org/entity/Q2241746,DISTRICT,state:6,no,True +district:76,Yamunanagar,76,http://www.wikidata.org/entity/Q1873644,DISTRICT,state:6,no,True +district:670,South East Delhi,670,http://www.wikidata.org/entity/Q25553535,DISTRICT,state:7,no,True +district:671,Shahdara,671,http://www.wikidata.org/entity/Q83486,DISTRICT,state:7,no,True +district:77,Central Delhi,77,http://www.wikidata.org/entity/Q107941,DISTRICT,state:7,no,True +district:78,East Delhi,78,http://www.wikidata.org/entity/Q107960,DISTRICT,state:7,no,True +district:79,New Delhi,79,http://www.wikidata.org/entity/Q8560886,DISTRICT,state:7,no,True +district:794,Outer North Delhi,794,http://www.wikidata.org/entity/Q140805546,DISTRICT,state:7,no,True +district:795,Old Delhi,795,http://www.wikidata.org/entity/Q140721912,DISTRICT,state:7,no,True +district:796,Central North Delhi,796,http://www.wikidata.org/entity/Q140804026,DISTRICT,state:7,no,True +district:80,North Delhi,80,http://www.wikidata.org/entity/Q693367,DISTRICT,state:7,no,True +district:81,North East Delhi,81,http://www.wikidata.org/entity/Q429329,DISTRICT,state:7,no,True +district:82,North West Delhi,82,http://www.wikidata.org/entity/Q766125,DISTRICT,state:7,no,True +district:83,South Delhi,83,http://www.wikidata.org/entity/Q2061938,DISTRICT,state:7,no,True +district:84,South West Delhi,84,http://www.wikidata.org/entity/Q2379189,DISTRICT,state:7,no,True +district:85,West Delhi,85,http://www.wikidata.org/entity/Q549807,DISTRICT,state:7,no,True +district:100,Sri Ganganagar,100,http://www.wikidata.org/entity/Q1419696,DISTRICT,state:8,no,True +district:101,Hanumangarh,101,http://www.wikidata.org/entity/Q1356112,DISTRICT,state:8,no,True +district:102,Jaipur,102,http://www.wikidata.org/entity/Q1134781,DISTRICT,state:8,no,True +district:103,Jaisalmer,103,http://www.wikidata.org/entity/Q1419708,DISTRICT,state:8,no,True +district:104,kishangarh sub division.Ajmer,104,http://www.wikidata.org/entity/Q1460832,DISTRICT,state:8,no,True +district:105,Jhalawar,105,http://www.wikidata.org/entity/Q1471417,DISTRICT,state:8,no,True +district:106,Jhunjhunu,106,http://www.wikidata.org/entity/Q1471427,DISTRICT,state:8,no,True +district:107,Jodhpur,107,http://www.wikidata.org/entity/Q1434965,DISTRICT,state:8,no,True +district:108,Karauli,108,http://www.wikidata.org/entity/Q1419668,DISTRICT,state:8,no,True +district:109,Kota,109,http://www.wikidata.org/entity/Q999432,DISTRICT,state:8,no,True +district:110,Nagaur,110,http://www.wikidata.org/entity/Q1507174,DISTRICT,state:8,no,True +district:111,Pali,111,http://www.wikidata.org/entity/Q46925,DISTRICT,state:8,no,True +district:112,Rajsamand,112,http://www.wikidata.org/entity/Q596693,DISTRICT,state:8,no,True +district:113,Sawai Madhopur,113,http://www.wikidata.org/entity/Q1507166,DISTRICT,state:8,no,True +district:114,Sikar,114,http://www.wikidata.org/entity/Q12945777,DISTRICT,state:8,no,True +district:115,Sirohi,115,http://www.wikidata.org/entity/Q205719,DISTRICT,state:8,no,True +district:116,Tonk,116,http://www.wikidata.org/entity/Q915880,DISTRICT,state:8,no,True +district:117,Udaipur,117,http://www.wikidata.org/entity/Q1321577,DISTRICT,state:8,no,True +district:629,Pratapgarh,629,http://www.wikidata.org/entity/Q1585433,DISTRICT,state:8,no,True +district:767,Deeg,767,http://www.wikidata.org/entity/Q122766908,DISTRICT,state:8,no,True +district:768,Didwana-Kuchaman,768,http://www.wikidata.org/entity/Q122971176,DISTRICT,state:8,no,True +district:769,Dudu,769,http://www.wikidata.org/entity/Q122971166,DISTRICT,state:8,no,True +district:770,Khairthal-Tijara,770,http://www.wikidata.org/entity/Q122971175,DISTRICT,state:8,no,True +district:771,Gangapur City,771,http://www.wikidata.org/entity/Q121607665,DISTRICT,state:8,no,True +district:772,Phalodi,772,http://www.wikidata.org/entity/Q122971167,DISTRICT,state:8,no,True +district:773,Neem Ka Thana,773,http://www.wikidata.org/entity/Q122212229,DISTRICT,state:8,no,True +district:774,Beawar,774,http://www.wikidata.org/entity/Q122971174,DISTRICT,state:8,no,True +district:775,Balotra,775,http://www.wikidata.org/entity/Q121335556,DISTRICT,state:8,no,True +district:776,Anupgarh,776,http://www.wikidata.org/entity/Q117230972,DISTRICT,state:8,no,True +district:777,Salumbar,777,http://www.wikidata.org/entity/Q122971173,DISTRICT,state:8,no,True +district:778,Jodhpur Gramin,778,http://www.wikidata.org/entity/Q122971177,DISTRICT,state:8,no,True +district:779,Sanchore,779,http://www.wikidata.org/entity/Q122276059,DISTRICT,state:8,no,True +district:781,Kekri,781,http://www.wikidata.org/entity/Q122971178,DISTRICT,state:8,no,True +district:782,Kotputli-Behror,782,http://www.wikidata.org/entity/Q117313512,DISTRICT,state:8,no,True +district:86,Ajmer,86,http://www.wikidata.org/entity/Q413037,DISTRICT,state:8,no,True +district:87,Alwar,87,http://www.wikidata.org/entity/Q449690,DISTRICT,state:8,no,True +district:88,Banswara,88,http://www.wikidata.org/entity/Q806969,DISTRICT,state:8,no,True +district:89,Baran,89,http://www.wikidata.org/entity/Q2329717,DISTRICT,state:8,no,True +district:90,Barmer,90,http://www.wikidata.org/entity/Q42016,DISTRICT,state:8,no,True +district:91,Bharatpur,91,http://www.wikidata.org/entity/Q854861,DISTRICT,state:8,no,True +district:92,Bhilwara,92,http://www.wikidata.org/entity/Q41991,DISTRICT,state:8,no,True +district:93,Bikaner,93,http://www.wikidata.org/entity/Q778996,DISTRICT,state:8,no,True +district:94,Bundi,94,http://www.wikidata.org/entity/Q670405,DISTRICT,state:8,no,True +district:95,Chittorgarh,95,http://www.wikidata.org/entity/Q1075011,DISTRICT,state:8,no,True +district:96,Churu,96,http://www.wikidata.org/entity/Q1090006,DISTRICT,state:8,no,True +district:97,Dausa,97,http://www.wikidata.org/entity/Q1173042,DISTRICT,state:8,no,True +district:98,Dholpur,98,http://www.wikidata.org/entity/Q1207709,DISTRICT,state:8,no,True +district:99,Dungarpur,99,http://www.wikidata.org/entity/Q1265687,DISTRICT,state:8,no,True +district:118,Agra,118,http://www.wikidata.org/entity/Q606343,DISTRICT,state:9,no,True +district:119,Aligarh,119,http://www.wikidata.org/entity/Q766918,DISTRICT,state:9,no,True +district:120,Prayagraj,120,http://www.wikidata.org/entity/Q1773426,DISTRICT,state:9,no,True +district:121,Ambedkar Nagar,121,http://www.wikidata.org/entity/Q456764,DISTRICT,state:9,no,True +district:122,Auraiya,122,http://www.wikidata.org/entity/Q1812533,DISTRICT,state:9,no,True +district:123,Azamgarh,123,http://www.wikidata.org/entity/Q793553,DISTRICT,state:9,no,True +district:124,Bagpat,124,http://www.wikidata.org/entity/Q1797363,DISTRICT,state:9,no,True +district:125,Bahraich,125,http://www.wikidata.org/entity/Q1812548,DISTRICT,state:9,no,True +district:126,Ballia,126,http://www.wikidata.org/entity/Q584644,DISTRICT,state:9,no,True +district:127,Balrampur,127,http://www.wikidata.org/entity/Q1948380,DISTRICT,state:9,no,True +district:128,Banda,128,http://www.wikidata.org/entity/Q2131759,DISTRICT,state:9,no,True +district:129,Barabanki,129,http://www.wikidata.org/entity/Q633114,DISTRICT,state:9,no,True +district:130,Bareilly,130,http://www.wikidata.org/entity/Q1797378,DISTRICT,state:9,no,True +district:131,Basti,131,http://www.wikidata.org/entity/Q715267,DISTRICT,state:9,no,True +district:132,Bijnor,132,http://www.wikidata.org/entity/Q1937865,DISTRICT,state:9,no,True +district:133,Budaun,133,http://www.wikidata.org/entity/Q1815262,DISTRICT,state:9,no,True +district:134,Bulandshahr,134,http://www.wikidata.org/entity/Q1752328,DISTRICT,state:9,no,True +district:135,Chandauli,135,http://www.wikidata.org/entity/Q2733369,DISTRICT,state:9,no,True +district:136,Chitrakoot,136,http://www.wikidata.org/entity/Q2089141,DISTRICT,state:9,no,True +district:137,Deoria,137,http://www.wikidata.org/entity/Q731746,DISTRICT,state:9,no,True +district:138,Etah,138,http://www.wikidata.org/entity/Q1773429,DISTRICT,state:9,no,True +district:139,Etawah,139,http://www.wikidata.org/entity/Q1815288,DISTRICT,state:9,no,True +district:140,Ayodhya,140,http://www.wikidata.org/entity/Q1814132,DISTRICT,state:9,no,True +district:141,Farrukhabad,141,http://www.wikidata.org/entity/Q1897251,DISTRICT,state:9,no,True +district:142,Fatehpur,142,http://www.wikidata.org/entity/Q1946829,DISTRICT,state:9,no,True +district:143,Firozabad,143,http://www.wikidata.org/entity/Q1946950,DISTRICT,state:9,no,True +district:144,Gautam Buddh Nagar,144,http://www.wikidata.org/entity/Q1785950,DISTRICT,state:9,no,True +district:145,Ghaziabad,145,http://www.wikidata.org/entity/Q1773444,DISTRICT,state:9,no,True +district:146,Ghazipur,146,http://www.wikidata.org/entity/Q1287993,DISTRICT,state:9,no,True +district:147,Gonda,147,http://www.wikidata.org/entity/Q1937857,DISTRICT,state:9,no,True +district:148,Gorakhpur,148,http://www.wikidata.org/entity/Q1144349,DISTRICT,state:9,no,True +district:149,Hamirpur,149,http://www.wikidata.org/entity/Q2019757,DISTRICT,state:9,no,True +district:150,Hardoi,150,http://www.wikidata.org/entity/Q1772822,DISTRICT,state:9,no,True +district:151,Jalaun,151,http://www.wikidata.org/entity/Q2089115,DISTRICT,state:9,no,True +district:152,Jaunpur,152,http://www.wikidata.org/entity/Q1356060,DISTRICT,state:9,no,True +district:153,Jhansi,153,http://www.wikidata.org/entity/Q1937885,DISTRICT,state:9,no,True +district:154,Amroha,154,http://www.wikidata.org/entity/Q1891677,DISTRICT,state:9,no,True +district:155,Kannauj,155,http://www.wikidata.org/entity/Q627979,DISTRICT,state:9,no,True +district:156,Kanpur Dehat,156,http://www.wikidata.org/entity/Q610612,DISTRICT,state:9,no,True +district:157,Kanpur Nagar,157,http://www.wikidata.org/entity/Q2089152,DISTRICT,state:9,no,True +district:158,Kaushambi,158,http://www.wikidata.org/entity/Q1946937,DISTRICT,state:9,no,True +district:159,Lakhimpur Kheri,159,http://www.wikidata.org/entity/Q1755447,DISTRICT,state:9,no,True +district:160,Kushinagar,160,http://www.wikidata.org/entity/Q1840355,DISTRICT,state:9,no,True +district:161,Lalitpur,161,http://www.wikidata.org/entity/Q1947336,DISTRICT,state:9,no,True +district:162,Lucknow,162,http://www.wikidata.org/entity/Q1773416,DISTRICT,state:9,no,True +district:163,Hathras,163,http://www.wikidata.org/entity/Q1814892,DISTRICT,state:9,no,True +district:164,Maharajganj,164,http://www.wikidata.org/entity/Q1356139,DISTRICT,state:9,no,True +district:165,Mahoba,165,http://www.wikidata.org/entity/Q1815322,DISTRICT,state:9,no,True +district:166,Mainpuri,166,http://www.wikidata.org/entity/Q1816657,DISTRICT,state:9,no,True +district:167,Mathura,167,http://www.wikidata.org/entity/Q1773422,DISTRICT,state:9,no,True +district:168,Mau,168,http://www.wikidata.org/entity/Q1518847,DISTRICT,state:9,no,True +district:169,Meerut,169,http://www.wikidata.org/entity/Q1764627,DISTRICT,state:9,no,True +district:170,Mirzapur,170,http://www.wikidata.org/entity/Q1143894,DISTRICT,state:9,no,True +district:171,Moradabad,171,http://www.wikidata.org/entity/Q1345006,DISTRICT,state:9,no,True +district:172,Muzaffarnagar,172,http://www.wikidata.org/entity/Q2365710,DISTRICT,state:9,no,True +district:173,Pilibhit,173,http://www.wikidata.org/entity/Q2980705,DISTRICT,state:9,no,True +district:174,Pratapgarh,174,http://www.wikidata.org/entity/Q1473962,DISTRICT,state:9,no,True +district:175,Raebareli,175,http://www.wikidata.org/entity/Q1321157,DISTRICT,state:9,no,True +district:176,Rampur,176,http://www.wikidata.org/entity/Q1815331,DISTRICT,state:9,no,True +district:177,Saharanpur,177,http://www.wikidata.org/entity/Q1797326,DISTRICT,state:9,no,True +district:178,Sant Kabir Nagar,178,http://www.wikidata.org/entity/Q1945445,DISTRICT,state:9,no,True +district:179,Bhadohi,179,http://www.wikidata.org/entity/Q127533,DISTRICT,state:9,no,True +district:180,Shahjahanpur,180,http://www.wikidata.org/entity/Q1812557,DISTRICT,state:9,no,True +district:181,Shravasti,181,http://www.wikidata.org/entity/Q1945458,DISTRICT,state:9,no,True +district:182,Siddharthnagar,182,http://www.wikidata.org/entity/Q1815339,DISTRICT,state:9,no,True +district:183,Sitapur,183,http://www.wikidata.org/entity/Q1812539,DISTRICT,state:9,no,True +district:184,Sonbhadra,184,http://www.wikidata.org/entity/Q607798,DISTRICT,state:9,no,True +district:185,Sultanpur,185,http://www.wikidata.org/entity/Q1356154,DISTRICT,state:9,no,True +district:186,Unnao,186,http://www.wikidata.org/entity/Q1937875,DISTRICT,state:9,no,True +district:187,Varanasi,187,http://www.wikidata.org/entity/Q1321140,DISTRICT,state:9,no,True +district:633,Kasganj,633,http://www.wikidata.org/entity/Q890800,DISTRICT,state:9,no,True +district:640,Amethi,640,http://www.wikidata.org/entity/Q1071494,DISTRICT,state:9,no,True +district:659,Sambhal,659,http://www.wikidata.org/entity/Q3000436,DISTRICT,state:9,no,True +district:660,Shamli,660,http://www.wikidata.org/entity/Q2999938,DISTRICT,state:9,no,True +district:661,Hapur,661,http://www.wikidata.org/entity/Q5653340,DISTRICT,state:9,no,True diff --git a/api/services/metadata_export/contracts/licenses.csv b/api/services/metadata_export/contracts/licenses.csv new file mode 100644 index 00000000..cba256f4 --- /dev/null +++ b/api/services/metadata_export/contracts/licenses.csv @@ -0,0 +1,61 @@ +key,label,code,uri,is_open,visible_on_dataspace,alt_codes +GODL-India,Government Open Data License – India,GODL-India,https://civicdataspace.in/id/licence/godl-india,True,yes,dataspace_enum=GOVERNMENT_OPEN_DATA_LICENSE +CC-PDM-1.0,Creative Commons Public Domain Mark 1.0 Universal,CC-PDM-1.0,https://spdx.org/licenses/CC-PDM-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-PDM-1.0.json +CC-BY-1.0,Creative Commons Attribution 1.0 Generic,CC-BY-1.0,https://spdx.org/licenses/CC-BY-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-1.0.json +CC-BY-2.0,Creative Commons Attribution 2.0 Generic,CC-BY-2.0,https://spdx.org/licenses/CC-BY-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-2.0.json +CC-BY-2.5-AU,Creative Commons Attribution 2.5 Australia,CC-BY-2.5-AU,https://spdx.org/licenses/CC-BY-2.5-AU,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-2.5-AU.json +CC-BY-2.5,Creative Commons Attribution 2.5 Generic,CC-BY-2.5,https://spdx.org/licenses/CC-BY-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-2.5.json +CC-BY-3.0-AU,Creative Commons Attribution 3.0 Australia,CC-BY-3.0-AU,https://spdx.org/licenses/CC-BY-3.0-AU,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-AU.json +CC-BY-3.0-AT,Creative Commons Attribution 3.0 Austria,CC-BY-3.0-AT,https://spdx.org/licenses/CC-BY-3.0-AT,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-AT.json +CC-BY-3.0-DE,Creative Commons Attribution 3.0 Germany,CC-BY-3.0-DE,https://spdx.org/licenses/CC-BY-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-DE.json +CC-BY-3.0-IGO,Creative Commons Attribution 3.0 IGO,CC-BY-3.0-IGO,https://spdx.org/licenses/CC-BY-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-IGO.json +CC-BY-3.0-NL,Creative Commons Attribution 3.0 Netherlands,CC-BY-3.0-NL,https://spdx.org/licenses/CC-BY-3.0-NL,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-NL.json +CC-BY-3.0-US,Creative Commons Attribution 3.0 United States,CC-BY-3.0-US,https://spdx.org/licenses/CC-BY-3.0-US,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-US.json +CC-BY-3.0,Creative Commons Attribution 3.0 Unported,CC-BY-3.0,https://spdx.org/licenses/CC-BY-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0.json +CC-BY-4.0,Creative Commons Attribution 4.0,CC-BY-4.0,http://creativecommons.org/licenses/by/4.0/,True,yes,authority_label=Creative Commons Attribution 4.0 International|dataspace_enum=CC_BY_4_0_ATTRIBUTION|spdx_detail=https://spdx.org/licenses/CC-BY-4.0.json +CC-BY-ND-1.0,Creative Commons Attribution No Derivatives 1.0 Generic,CC-BY-ND-1.0,https://spdx.org/licenses/CC-BY-ND-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-1.0.json +CC-BY-ND-2.0,Creative Commons Attribution No Derivatives 2.0 Generic,CC-BY-ND-2.0,https://spdx.org/licenses/CC-BY-ND-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-2.0.json +CC-BY-ND-2.5,Creative Commons Attribution No Derivatives 2.5 Generic,CC-BY-ND-2.5,https://spdx.org/licenses/CC-BY-ND-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-2.5.json +CC-BY-ND-3.0-DE,Creative Commons Attribution No Derivatives 3.0 Germany,CC-BY-ND-3.0-DE,https://spdx.org/licenses/CC-BY-ND-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-3.0-DE.json +CC-BY-ND-3.0,Creative Commons Attribution No Derivatives 3.0 Unported,CC-BY-ND-3.0,https://spdx.org/licenses/CC-BY-ND-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-3.0.json +CC-BY-ND-4.0,Creative Commons Attribution No Derivatives 4.0 International,CC-BY-ND-4.0,https://spdx.org/licenses/CC-BY-ND-4.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-4.0.json +CC-BY-NC-1.0,Creative Commons Attribution Non Commercial 1.0 Generic,CC-BY-NC-1.0,https://spdx.org/licenses/CC-BY-NC-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-1.0.json +CC-BY-NC-2.0,Creative Commons Attribution Non Commercial 2.0 Generic,CC-BY-NC-2.0,https://spdx.org/licenses/CC-BY-NC-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-2.0.json +CC-BY-NC-2.5,Creative Commons Attribution Non Commercial 2.5 Generic,CC-BY-NC-2.5,https://spdx.org/licenses/CC-BY-NC-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-2.5.json +CC-BY-NC-3.0-DE,Creative Commons Attribution Non Commercial 3.0 Germany,CC-BY-NC-3.0-DE,https://spdx.org/licenses/CC-BY-NC-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-3.0-DE.json +CC-BY-NC-3.0-IGO,Creative Commons Attribution Non Commercial 3.0 IGO,CC-BY-NC-3.0-IGO,https://spdx.org/licenses/CC-BY-NC-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-3.0-IGO.json +CC-BY-NC-3.0,Creative Commons Attribution Non Commercial 3.0 Unported,CC-BY-NC-3.0,https://spdx.org/licenses/CC-BY-NC-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-3.0.json +CC-BY-NC-4.0,Creative Commons Attribution Non Commercial 4.0 International,CC-BY-NC-4.0,http://creativecommons.org/licenses/by-nc/4.0/,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-4.0.json +CC-BY-NC-ND-1.0,Creative Commons Attribution Non Commercial No Derivatives 1.0 Generic,CC-BY-NC-ND-1.0,https://spdx.org/licenses/CC-BY-NC-ND-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-1.0.json +CC-BY-NC-ND-2.0,Creative Commons Attribution Non Commercial No Derivatives 2.0 Generic,CC-BY-NC-ND-2.0,https://spdx.org/licenses/CC-BY-NC-ND-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-2.0.json +CC-BY-NC-ND-2.5,Creative Commons Attribution Non Commercial No Derivatives 2.5 Generic,CC-BY-NC-ND-2.5,https://spdx.org/licenses/CC-BY-NC-ND-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-2.5.json +CC-BY-NC-ND-3.0-DE,Creative Commons Attribution Non Commercial No Derivatives 3.0 Germany,CC-BY-NC-ND-3.0-DE,https://spdx.org/licenses/CC-BY-NC-ND-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-3.0-DE.json +CC-BY-NC-ND-3.0-IGO,Creative Commons Attribution Non Commercial No Derivatives 3.0 IGO,CC-BY-NC-ND-3.0-IGO,https://spdx.org/licenses/CC-BY-NC-ND-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-3.0-IGO.json +CC-BY-NC-ND-3.0,Creative Commons Attribution Non Commercial No Derivatives 3.0 Unported,CC-BY-NC-ND-3.0,https://spdx.org/licenses/CC-BY-NC-ND-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-3.0.json +CC-BY-NC-ND-4.0,Creative Commons Attribution Non Commercial No Derivatives 4.0 International,CC-BY-NC-ND-4.0,https://spdx.org/licenses/CC-BY-NC-ND-4.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-4.0.json +CC-BY-NC-SA-1.0,Creative Commons Attribution Non Commercial Share Alike 1.0 Generic,CC-BY-NC-SA-1.0,https://spdx.org/licenses/CC-BY-NC-SA-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-1.0.json +CC-BY-NC-SA-2.0-UK,Creative Commons Attribution Non Commercial Share Alike 2.0 England and Wales,CC-BY-NC-SA-2.0-UK,https://spdx.org/licenses/CC-BY-NC-SA-2.0-UK,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.0-UK.json +CC-BY-NC-SA-2.0,Creative Commons Attribution Non Commercial Share Alike 2.0 Generic,CC-BY-NC-SA-2.0,https://spdx.org/licenses/CC-BY-NC-SA-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.0.json +CC-BY-NC-SA-2.0-DE,Creative Commons Attribution Non Commercial Share Alike 2.0 Germany,CC-BY-NC-SA-2.0-DE,https://spdx.org/licenses/CC-BY-NC-SA-2.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.0-DE.json +CC-BY-NC-SA-2.5,Creative Commons Attribution Non Commercial Share Alike 2.5 Generic,CC-BY-NC-SA-2.5,https://spdx.org/licenses/CC-BY-NC-SA-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.5.json +CC-BY-NC-SA-3.0-DE,Creative Commons Attribution Non Commercial Share Alike 3.0 Germany,CC-BY-NC-SA-3.0-DE,https://spdx.org/licenses/CC-BY-NC-SA-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-3.0-DE.json +CC-BY-NC-SA-3.0-IGO,Creative Commons Attribution Non Commercial Share Alike 3.0 IGO,CC-BY-NC-SA-3.0-IGO,https://spdx.org/licenses/CC-BY-NC-SA-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-3.0-IGO.json +CC-BY-NC-SA-3.0,Creative Commons Attribution Non Commercial Share Alike 3.0 Unported,CC-BY-NC-SA-3.0,https://spdx.org/licenses/CC-BY-NC-SA-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-3.0.json +CC-BY-NC-SA-4.0,Creative Commons Attribution Non Commercial Share Alike 4.0 International,CC-BY-NC-SA-4.0,https://spdx.org/licenses/CC-BY-NC-SA-4.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-4.0.json +CC-BY-SA-1.0,Creative Commons Attribution Share Alike 1.0 Generic,CC-BY-SA-1.0,https://spdx.org/licenses/CC-BY-SA-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-1.0.json +CC-BY-SA-2.0-UK,Creative Commons Attribution Share Alike 2.0 England and Wales,CC-BY-SA-2.0-UK,https://spdx.org/licenses/CC-BY-SA-2.0-UK,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-2.0-UK.json +CC-BY-SA-2.0,Creative Commons Attribution Share Alike 2.0 Generic,CC-BY-SA-2.0,https://spdx.org/licenses/CC-BY-SA-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-2.0.json +CC-BY-SA-2.1-JP,Creative Commons Attribution Share Alike 2.1 Japan,CC-BY-SA-2.1-JP,https://spdx.org/licenses/CC-BY-SA-2.1-JP,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-2.1-JP.json +CC-BY-SA-2.5,Creative Commons Attribution Share Alike 2.5 Generic,CC-BY-SA-2.5,https://spdx.org/licenses/CC-BY-SA-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-2.5.json +CC-BY-SA-3.0-AT,Creative Commons Attribution Share Alike 3.0 Austria,CC-BY-SA-3.0-AT,https://spdx.org/licenses/CC-BY-SA-3.0-AT,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-3.0-AT.json +CC-BY-SA-3.0-DE,Creative Commons Attribution Share Alike 3.0 Germany,CC-BY-SA-3.0-DE,https://spdx.org/licenses/CC-BY-SA-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-3.0-DE.json +CC-BY-SA-3.0,Creative Commons Attribution Share Alike 3.0 Unported,CC-BY-SA-3.0,https://spdx.org/licenses/CC-BY-SA-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-3.0.json +CC-BY-NC-SA-2.0-FR,Creative Commons Attribution-NonCommercial-ShareAlike 2.0 France,CC-BY-NC-SA-2.0-FR,https://spdx.org/licenses/CC-BY-NC-SA-2.0-FR,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.0-FR.json +CC-BY-SA-3.0-IGO,Creative Commons Attribution-ShareAlike 3.0 IGO,CC-BY-SA-3.0-IGO,https://spdx.org/licenses/CC-BY-SA-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-3.0-IGO.json +CC-BY-SA-4.0,Creative Commons Attribution-ShareAlike 4.0,CC-BY-SA-4.0,http://creativecommons.org/licenses/by-sa/4.0/,True,yes,authority_label=Creative Commons Attribution Share Alike 4.0 International|dataspace_enum=CC_BY_SA_4_0_ATTRIBUTION_SHARE_ALIKE|spdx_detail=https://spdx.org/licenses/CC-BY-SA-4.0.json +CC-PDDC,Creative Commons Public Domain Dedication and Certification,CC-PDDC,https://spdx.org/licenses/CC-PDDC,False,no,spdx_detail=https://spdx.org/licenses/CC-PDDC.json +CC-SA-1.0,Creative Commons Share Alike 1.0 Generic,CC-SA-1.0,https://spdx.org/licenses/CC-SA-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-SA-1.0.json +CC0-1.0,Creative Commons Zero v1.0 Universal,CC0-1.0,https://creativecommons.org/publicdomain/zero/1.0/,True,yes,spdx_detail=https://spdx.org/licenses/CC0-1.0.json +ODC-By-1.0,Open Data Commons Attribution License 1.0,ODC-By-1.0,https://opendatacommons.org/licenses/by/1-0/,True,yes,authority_label=Open Data Commons Attribution License v1.0|dataspace_enum=OPEN_DATA_COMMONS_BY_ATTRIBUTION|spdx_detail=https://spdx.org/licenses/ODC-By-1.0.json +PDDL-1.0,Open Data Commons Public Domain Dedication & License 1.0,PDDL-1.0,https://opendatacommons.org/licenses/pddl/1-0/,True,yes,spdx_detail=https://spdx.org/licenses/PDDL-1.0.json +ODbL-1.0,Open Database License 1.0,ODbL-1.0,https://opendatacommons.org/licenses/odbl/1-0/,True,yes,authority_label=Open Data Commons Open Database License v1.0|dataspace_enum=OPEN_DATABASE_LICENSE|spdx_detail=https://spdx.org/licenses/ODbL-1.0.json diff --git a/api/services/metadata_export/contracts/sectors.csv b/api/services/metadata_export/contracts/sectors.csv new file mode 100644 index 00000000..747ebb43 --- /dev/null +++ b/api/services/metadata_export/contracts/sectors.csv @@ -0,0 +1,22 @@ +key,label,code,uri,is_open,visible_on_dataspace,alt_codes +cdl:child-rights,Child Rights,child-rights,https://civicdataspace.in/id/sector/child-rights,,yes,dac_crs=16010|eu_data_theme=SOCI|eu_match_type=broadMatch|sdg= +cdl:climate-action,Climate Action,climate-action,https://civicdataspace.in/id/sector/climate-action,,yes,dac_crs=|eu_data_theme=ENVI|eu_match_type=broadMatch|sdg=13 +cdl:coastal,Coastal,coastal,https://civicdataspace.in/id/sector/coastal,,yes,dac_crs=|eu_data_theme=ENVI|eu_match_type=broadMatch|sdg=14 +cdl:disaster-risk-reduction,Disaster Risk Reduction,disaster-risk-reduction,https://civicdataspace.in/id/sector/disaster-risk-reduction,,yes,dac_crs=74020|eu_data_theme=ENVI|eu_match_type=broadMatch|sdg= +cdl:gender,Gender,gender,https://civicdataspace.in/id/sector/gender,,yes,dac_crs=15170|eu_data_theme=SOCI|eu_match_type=broadMatch|sdg=5 +cdl:law-and-justice,Law and Justice,law-and-justice,https://civicdataspace.in/id/sector/law-and-justice,,yes,dac_crs=|eu_data_theme=JUST|eu_match_type=closeMatch|sdg=16 +cdl:public-finance,Public Finance,public-finance,https://civicdataspace.in/id/sector/public-finance,,yes,dac_crs=|eu_data_theme=GOVE|eu_match_type=closeMatch|sdg= +cdl:urban-development,Urban Development,urban-development,https://civicdataspace.in/id/sector/urban-development,,yes,dac_crs=|eu_data_theme=REGI|eu_match_type=closeMatch|sdg=11 +eu-theme:AGRI,"Agriculture, fisheries, forestry and food",AGRI,http://publications.europa.eu/resource/authority/data-theme/AGRI,,no, +eu-theme:ECON,Economy and finance,ECON,http://publications.europa.eu/resource/authority/data-theme/ECON,,no, +eu-theme:EDUC,"Education, culture and sport",EDUC,http://publications.europa.eu/resource/authority/data-theme/EDUC,,no, +eu-theme:ENER,Energy,ENER,http://publications.europa.eu/resource/authority/data-theme/ENER,,no, +eu-theme:ENVI,Environment,ENVI,http://publications.europa.eu/resource/authority/data-theme/ENVI,,no, +eu-theme:GOVE,Government and public sector,GOVE,http://publications.europa.eu/resource/authority/data-theme/GOVE,,no, +eu-theme:HEAL,Health,HEAL,http://publications.europa.eu/resource/authority/data-theme/HEAL,,no, +eu-theme:INTR,International issues,INTR,http://publications.europa.eu/resource/authority/data-theme/INTR,,no, +eu-theme:JUST,"Justice, legal system and public safety",JUST,http://publications.europa.eu/resource/authority/data-theme/JUST,,no, +eu-theme:SOCI,Population and society,SOCI,http://publications.europa.eu/resource/authority/data-theme/SOCI,,no, +eu-theme:REGI,Regions and cities,REGI,http://publications.europa.eu/resource/authority/data-theme/REGI,,no, +eu-theme:TECH,Science and technology,TECH,http://publications.europa.eu/resource/authority/data-theme/TECH,,no, +eu-theme:TRAN,Transport,TRAN,http://publications.europa.eu/resource/authority/data-theme/TRAN,,no, diff --git a/api/services/metadata_export/crosswalk.py b/api/services/metadata_export/crosswalk.py new file mode 100644 index 00000000..7a43eec4 --- /dev/null +++ b/api/services/metadata_export/crosswalk.py @@ -0,0 +1,206 @@ +"""Standards crosswalk engine. + +Nothing here knows the name of a standard: every property, node type, context +and obligation comes from ``contracts/crosswalk.json`` (owned by this repo; see +``contracts/README.md``). Add a standard to that file and this module exports +it unchanged. The record it reads is a plain dict keyed by the contract's +``dataspace_field`` names, produced by ``adapter.py``. +""" + +from __future__ import annotations + +import json +from functools import lru_cache +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +CONTRACT_PATH = Path(__file__).resolve().parent / "contracts" / "crosswalk.json" + +# Value types that must be a URI on the way out; a bare name is reported. +MUST_RESOLVE = {"uri", "concept", "location"} +# Value types written as {"@id": …} when the standard's uri_style is "node". +REFERENCE_TYPES = {"uri", "concept", "location", "agent"} +# FOAF agent classes -> schema.org classes, for standards whose uri_style is not "node". +SCHEMA_AGENT_TYPES = { + "foaf:Organization": "Organization", + "foaf:Person": "Person", + "foaf:Agent": "Organization", +} + +EMPTY = (None, "", [], {}) + + +def _is_uri(value: Any) -> bool: + return isinstance(value, str) and (value.startswith("http://") or value.startswith("https://")) + + +class Crosswalk: + def __init__(self, data: dict): + self.data = data + concepts = data.get("concepts", []) + # The contract ships concepts as a list (each with a "key") or a dict keyed by concept. + if isinstance(concepts, dict): + self.concepts: Dict[str, dict] = {k: dict(v, key=k) for k, v in concepts.items()} + else: + self.concepts = {c["key"]: c for c in concepts} + self.standards: Dict[str, dict] = data["standards"] + + @classmethod + @lru_cache(maxsize=1) + def load(cls, path: Optional[str] = None) -> "Crosswalk": + with open(path or CONTRACT_PATH, encoding="utf-8") as fh: + return cls(json.load(fh)) + + # ------------------------------------------------------------------ info + def standard_ids(self) -> List[str]: + return list(self.standards.keys()) + + def standard(self, sid: str) -> dict: + try: + return self.standards[sid] + except KeyError as exc: + raise KeyError(f"Unknown standard '{sid}'. Known: {', '.join(self.standards)}") from exc + + def gaps(self, sid: str) -> List[dict]: + """Concepts this standard cannot carry.""" + return list(self.standard(sid).get("gaps", [])) + + # ---------------------------------------------------------------- export + def export(self, record: dict, sid: str) -> dict: + doc, _ = self.export_with_report(record, sid) + return doc + + def export_with_report(self, record: dict, sid: str) -> Tuple[dict, dict]: + return self._export(record, self.standard(sid)) + + def _export(self, record: dict, std: dict) -> Tuple[dict, dict]: + node_types = std["node_types"] + uri_style = std.get("uri_style", "node") + + doc: Dict[str, Any] = {"@context": std["context"]} + if "dataset" in node_types: + doc["@type"] = node_types["dataset"] + + report: Dict[str, list] = { + "dropped": [], # platform has a value, standard has no binding + "unresolved": [], # needs a URI, got a name string + "missing_mandatory": [], # standard wants it, platform had nothing + } + + # Entries nested under another property (a period, a vCard) are + # collected per parent and attached once at the end. + nested: Dict[Tuple[str, str], Dict[str, Any]] = {} + by_node: Dict[str, List[dict]] = {} + for entry in std["export"]: + by_node.setdefault(entry["node"], []).append(entry) + + # --- dataset-level, plus anything nested under it ------------------- + for entry in by_node.get("dataset", []) + by_node.get("period", []): + value = self._platform_value(record, entry) + if value in EMPTY: + if entry["obligation"] == "mandatory": + report["missing_mandatory"].append(entry["concept"]) + continue + rendered = self._render(value, entry, uri_style, report) + if entry.get("parent_property"): + nested.setdefault(("dataset", entry["parent_property"]), {})[ + entry["property"] + ] = rendered + else: + doc[entry["property"]] = rendered + + for (owner, parent_property), payload in nested.items(): + if owner == "dataset": + doc[parent_property] = payload + + # --- distributions (one per resource) ------------------------------- + dist_entries = by_node.get("distribution", []) + link = next((e for e in std["export"] if e["concept"] == "distribution"), None) + if dist_entries and link: + distributions = [] + for resource in record.get("resources") or []: + dist: Dict[str, Any] = {} + if "distribution" in node_types: + dist["@type"] = node_types["distribution"] + for entry in dist_entries: + if entry["concept"] == "distribution": + continue + value = self._platform_value(resource, entry) + # DCAT requires accessURL on every Distribution; fall back to + # the download URL rather than emit an invalid Distribution. + if value in EMPTY and entry["concept"] == "access_url": + value = resource.get("download_url") + if value in EMPTY: + if entry["obligation"] == "mandatory": + report["missing_mandatory"].append(f"{entry['concept']} (distribution)") + continue + dist[entry["property"]] = self._render(value, entry, uri_style, report) + if len(dist) > (1 if "@type" in dist else 0): + distributions.append(dist) + if distributions: + doc[link["property"]] = distributions + + # --- what this standard cannot carry -------------------------------- + for gap in std.get("gaps", []): + concept = self.concepts.get(gap["concept"], {}) + field = concept.get("dataspace_field") + if field and record.get(field) not in EMPTY: + report["dropped"].append( + { + "concept": gap["concept"], + "dataspace_field": field, + "reason": gap.get("notes") or gap.get("reason"), + } + ) + + return doc, report + + def _platform_value(self, record: dict, entry: dict) -> Any: + """Value for an export entry, by the field the contract binds it to.""" + field = entry.get("dataspace_field") + return record.get(field) if field else None + + def _render(self, value: Any, entry: dict, uri_style: str, report: dict) -> Any: + vtype = entry["value_type"] + if isinstance(value, list): + return [self._render_one(v, entry, vtype, uri_style, report) for v in value] + rendered = self._render_one(value, entry, vtype, uri_style, report) + return [rendered] if entry.get("repeatable") else rendered + + def _render_one(self, value: Any, entry: dict, vtype: str, uri_style: str, report: dict) -> Any: + # Vocabulary values arrive as {"name": …, "uri": …} from the adapter. + if isinstance(value, dict) and vtype in ("concept", "location", "uri", "agent"): + if vtype == "agent" and uri_style == "node": + agent = {"@type": value.get("@type", "foaf:Agent"), "foaf:name": value.get("name")} + if value.get("uri"): + agent["@id"] = value["uri"] + return agent + if vtype == "agent": + kind = SCHEMA_AGENT_TYPES.get(value.get("@type", ""), "Organization") + return {"@type": kind, "name": value.get("name")} | ( + {"url": value["uri"]} if value.get("uri") else {} + ) + value = value.get("uri") or value.get("name") or value.get("label") + elif isinstance(value, dict) and vtype in ("literal", "langstring"): + # A vocabulary value written where the standard wants its plain name. + value = value.get("label") or value.get("name") or value.get("uri") + + if vtype in MUST_RESOLVE and not _is_uri(value): + report["unresolved"].append( + { + "concept": entry["concept"], + "value": value, + "vocabulary": entry.get("controlled_vocabulary"), + "expected": f"{vtype} — a resolvable URI", + } + ) + return value # emit what we have; the report says it is not a URI + + if vtype in REFERENCE_TYPES and uri_style == "node" and _is_uri(value): + return {"@id": value} + if vtype == "bytes": + try: + return int(value) + except (TypeError, ValueError): + return value + return value diff --git a/api/services/metadata_export/exporter.py b/api/services/metadata_export/exporter.py new file mode 100644 index 00000000..45352600 --- /dev/null +++ b/api/services/metadata_export/exporter.py @@ -0,0 +1,127 @@ +"""Orchestrate: model -> record -> crosswalk document -> serialisation fixes -> format. + +Which property a value becomes is decided by the contract. What is left here +is serialisation the contract cannot express: typed date literals, IANA +media-type IRIs, and the few structural rules the specs impose on a +Distribution or FileObject. +""" + +from __future__ import annotations + +from typing import Any, Dict, Tuple + +from api.models import Dataset +from api.services.metadata_export.adapter import IANA_BASE, dataset_to_record +from api.services.metadata_export.crosswalk import Crosswalk +from api.services.metadata_export.formats import allowed_formats, serialise + +CROISSANT_CONFORMS_TO = "http://mlcommons.org/croissant/1.0" +XSD_DATE = "http://www.w3.org/2001/XMLSchema#date" +DATE_PROPERTIES = ("dcterms:issued", "dcterms:modified", "dcterms:created") +PERIOD_PROPERTIES = ("dcat:startDate", "dcat:endDate") + + +def _typed_date(value: Any) -> Any: + """DCAT-AP expects xsd:date literals; JSON-LD needs the type spelled out.""" + if isinstance(value, str) and len(value) >= 10 and value[4] == "-" and value[7] == "-": + return {"@value": value[:10], "@type": XSD_DATE} + return value + + +def _type_dates(doc: Dict[str, Any]) -> None: + for prop in DATE_PROPERTIES: + if prop in doc: + doc[prop] = _typed_date(doc[prop]) + period = doc.get("dcterms:temporal") + if isinstance(period, dict): + for prop in PERIOD_PROPERTIES: + if prop in period: + period[prop] = _typed_date(period[prop]) + + +def _by_name(record: Dict[str, Any]) -> Dict[str, dict]: + return {r.get("name"): r for r in record.get("resources") or [] if r.get("name")} + + +def _fix_dcat_distributions(doc: Dict[str, Any]) -> None: + """Media type as an IANA IRI; licence on every distribution (DCAT-AP).""" + licence = doc.get("dcterms:license") + for dist in doc.get("dcat:distribution") or []: + mt = dist.get("dcat:mediaType") + if isinstance(mt, str) and "/" in mt: + dist["dcat:mediaType"] = {"@id": IANA_BASE + mt} + if licence and "dcterms:license" not in dist: + dist["dcterms:license"] = licence + + +def _fix_croissant_file_objects(doc: Dict[str, Any], record: Dict[str, Any]) -> None: + """Croissant 1.0 requires @id, contentUrl and encodingFormat on a FileObject. + + Croissant has no accessURL, so for a link-only resource the platform page + (what a reader can actually fetch) becomes the contentUrl. sha256 is also + required by the spec and is emitted only when the record carries it. + """ + resources = _by_name(record) + landing = record.get("landing_page") or "" + for n, dist in enumerate(doc.get("distribution") or [], start=1): + res = resources.get(dist.get("name"), {}) + if "contentUrl" not in dist and res.get("access_url"): + dist["contentUrl"] = res["access_url"] + if "encodingFormat" not in dist and res.get("format"): + dist["encodingFormat"] = res["format"] + dist.setdefault( + "@id", + f"{landing}#resource-{res['_id']}" if res.get("_id") else f"{landing}#resource-{n}", + ) + + +def _finish(doc: Dict[str, Any], record: Dict[str, Any], standard: str) -> None: + if standard == "croissant": + doc.setdefault("conformsTo", CROISSANT_CONFORMS_TO) # required by the spec + _fix_croissant_file_objects(doc, record) + elif standard == "dcat": + _fix_dcat_distributions(doc) + _type_dates(doc) + elif standard == "dublin_core": + _type_dates(doc) + + +def export_dataset( + dataset: Dataset, standard: str, fmt: str = "jsonld" +) -> Tuple[str, str, str, dict]: + """Return (body, content_type, extension, report).""" + crosswalk = Crosswalk.load() + if standard not in crosswalk.standard_ids(): + raise ValueError( + f"Unknown standard '{standard}'. Known: {', '.join(crosswalk.standard_ids())}" + ) + if fmt not in allowed_formats(standard): + raise ValueError( + f"Format '{fmt}' is not available for {standard}. " + f"Allowed: {', '.join(allowed_formats(standard))}" + ) + + record = dataset_to_record(dataset) + doc, report = crosswalk.export_with_report(record, standard) + doc.setdefault("@id", record["landing_page"]) + _finish(doc, record, standard) + for definition in record.get("_unmapped_definitions") or []: + report["dropped"].append( + { + "concept": None, + "dataspace_field": f"metadata:{definition.get('urn') or definition.get('label')}", + "reason": "definition has no crosswalk concept; give it a recognised URN", + } + ) + + body, content_type, ext = serialise(doc, fmt) + return body, content_type, ext, report + + +def export_options() -> Dict[str, Any]: + """What the UI can offer: standards, and the formats valid for each.""" + cw = Crosswalk.load() + return { + sid: {"name": cw.standard(sid).get("name", sid), "formats": allowed_formats(sid)} + for sid in cw.standard_ids() + } diff --git a/api/services/metadata_export/formats.py b/api/services/metadata_export/formats.py new file mode 100644 index 00000000..b31eac6e --- /dev/null +++ b/api/services/metadata_export/formats.py @@ -0,0 +1,33 @@ +"""Serialise a JSON-LD document into the RDF syntaxes catalogue harvesters ask for.""" + +from __future__ import annotations + +import json +from typing import Dict, Tuple + +# format id -> (rdflib serializer name, content type, file extension) +FORMATS: Dict[str, Tuple[str, str, str]] = { + "jsonld": ("json-ld", "application/ld+json", "jsonld"), + "turtle": ("turtle", "text/turtle", "ttl"), + "rdfxml": ("xml", "application/rdf+xml", "rdf"), + "ntriples": ("nt", "application/n-triples", "nt"), +} + +# Croissant is defined as JSON-LD; its validator reads nothing else. +JSONLD_ONLY = {"croissant"} + + +def allowed_formats(standard_id: str) -> list: + return ["jsonld"] if standard_id in JSONLD_ONLY else list(FORMATS) + + +def serialise(document: dict, fmt: str) -> Tuple[str, str, str]: + """Return (body, content_type, extension) for the requested format.""" + serializer, content_type, ext = FORMATS[fmt] + if fmt == "jsonld": + return json.dumps(document, indent=2, ensure_ascii=False), content_type, ext + from rdflib import Graph # imported lazily: only non-JSON formats need it + + graph = Graph() + graph.parse(data=json.dumps(document), format="json-ld") + return graph.serialize(format=serializer), content_type, ext diff --git a/api/services/metadata_export/vocabularies.py b/api/services/metadata_export/vocabularies.py new file mode 100644 index 00000000..3bada132 --- /dev/null +++ b/api/services/metadata_export/vocabularies.py @@ -0,0 +1,107 @@ +"""Resolve the platform's labels to URIs using the vendored value lists. + +The standards want identifiers, not names: ``dcterms:spatial "Assam"`` is not +a spatial reference. The lists under ``contracts/`` (see contracts/README.md +repo's value superset) give a URI per licence, sector and geography. Anything +they do not know is returned as a plain label so the crosswalk reports it as +unresolved instead of silently emitting a bad value. +""" + +from __future__ import annotations + +import csv +from functools import lru_cache +from pathlib import Path +from typing import Dict, List, Optional + +CONTRACTS = Path(__file__).resolve().parent / "contracts" + + +def _rows(name: str) -> List[Dict[str, str]]: + with open(CONTRACTS / name, encoding="utf-8", newline="") as fh: + return list(csv.DictReader(fh)) + + +def _alt_codes(raw: str) -> Dict[str, str]: + """Parse the CSV's 'k=v|k=v' alt_codes column (';' tolerated too).""" + out: Dict[str, str] = {} + for part in (raw or "").replace(";", "|").split("|"): + if "=" in part: + k, v = part.split("=", 1) + out[k.strip()] = v.strip() + return out + + +@lru_cache(maxsize=1) +def _license_index() -> Dict[str, str]: + """Lower-cased key / label / DataSpace enum / SPDX id -> URI.""" + index: Dict[str, str] = {} + for r in _rows("licenses.csv"): + uri = r.get("uri") or "" + if not uri: + continue + for k in (r.get("key"), r.get("label"), r.get("code")): + if k: + index[k.strip().lower()] = uri + alt = _alt_codes(r.get("alt_codes", "")) + for k in ("dataspace_enum", "spdx", "spdx_id", "authority_label"): + if alt.get(k): + index[alt[k].strip().lower()] = uri + return index + + +@lru_cache(maxsize=1) +def _sector_index() -> Dict[str, str]: + index: Dict[str, str] = {} + for r in _rows("sectors.csv"): + uri = r.get("uri") or "" + if not uri: + continue + for k in (r.get("key"), r.get("label"), r.get("code")): + if k: + index[k.strip().lower()] = uri + key = r.get("key") or "" + if ":" in key: # "cdl:child-rights" -> "child-rights" + index[key.split(":", 1)[1].lower()] = uri + return index + + +@lru_cache(maxsize=1) +def _geography_index() -> Dict[str, Dict[str, str]]: + """label(lower) -> {tier -> uri}; tier disambiguates 'Aurangabad' etc.""" + index: Dict[str, Dict[str, str]] = {} + for r in _rows("geographies.csv"): + uri, label, tier = ( + r.get("uri") or "", + (r.get("label") or "").strip().lower(), + (r.get("tier") or "").upper(), + ) + if uri and label: + index.setdefault(label, {})[tier] = uri + return index + + +def license_uri(value: Optional[str]) -> Optional[str]: + return _license_index().get((value or "").strip().lower()) + + +def sector_uri(name: Optional[str]) -> Optional[str]: + return _sector_index().get((name or "").strip().lower()) + + +def geography_uri(name: Optional[str], geo_type: Optional[str] = None) -> Optional[str]: + tiers = _geography_index().get((name or "").strip().lower()) + if not tiers: + return None + if geo_type and geo_type.upper() in tiers: + return tiers[geo_type.upper()] + # Prefer the coarsest match when the platform did not say which level. + for tier in ("COUNTRY", "REGION", "STATE", "UT", "DISTRICT"): + if tier in tiers: + return tiers[tier] + return next(iter(tiers.values())) + + +def concept(name: str, uri: Optional[str]) -> Dict[str, str]: + """Shape the crosswalk renders: uri when known, else the bare name (reported).""" + return {"name": name, "uri": uri} if uri else {"name": name} diff --git a/api/urls.py b/api/urls.py index 0d9da9fa..fbd64b64 100644 --- a/api/urls.py +++ b/api/urls.py @@ -14,6 +14,7 @@ dataset_data, download, generate_dynamic_chart, + metadata_export, publication_download_view, search_aimodel, search_collaborative, @@ -100,6 +101,16 @@ dataset_data.PromptDatasetDataView.as_view(), name="prompt_dataset_data", ), + path( + "datasets//export/", + metadata_export.metadata_export, + name="dataset_metadata_export", + ), + path( + "metadata/export-options/", + metadata_export.metadata_export_options, + name="metadata_export_options", + ), # Single, simple GraphQL endpoint with no redirects path( "graphql", diff --git a/api/views/metadata_export.py b/api/views/metadata_export.py new file mode 100644 index 00000000..a3df43cb --- /dev/null +++ b/api/views/metadata_export.py @@ -0,0 +1,64 @@ +"""GET /api/datasets//export?standard=dcat|croissant|dublin_core&format=jsonld|turtle|rdfxml|ntriples + +Public metadata for a published dataset, generated on request from the +crosswalk contract. Nothing is stored. Add ``report=1`` to receive the gap +report (unresolved vocabulary values, dropped fields, missing mandatory +properties) alongside the document as JSON instead of a file download. +""" + +from __future__ import annotations + +import json +import uuid + +import structlog +from django.http import HttpRequest, HttpResponse, JsonResponse +from django.utils.text import slugify + +from api.models import Dataset +from api.services.metadata_export.exporter import export_dataset, export_options +from api.utils.enums import DatasetStatus + +logger = structlog.get_logger("dataspace.metadata_export") + + +def metadata_export_options(request: HttpRequest) -> JsonResponse: + return JsonResponse({"standards": export_options()}) + + +def metadata_export(request: HttpRequest, dataset_id: uuid.UUID) -> HttpResponse: + try: + dataset = Dataset.objects.select_related("organization", "user").get(id=dataset_id) + except Dataset.DoesNotExist: + return JsonResponse({"error": "Dataset not found"}, status=404) + + # Public endpoint: published datasets only. Owners preview drafts through + # the same view when logged in. + user = getattr(request, "user", None) + is_owner = bool( + user and user.is_authenticated and (dataset.user_id == user.id or user.is_superuser) + ) + if dataset.status != DatasetStatus.PUBLISHED.value and not is_owner: + return JsonResponse({"error": "Dataset not found"}, status=404) + + standard = (request.GET.get("standard") or "dcat").strip().lower() + fmt = (request.GET.get("format") or "jsonld").strip().lower() + try: + body, content_type, ext, report = export_dataset(dataset, standard, fmt) + except ValueError as exc: + return JsonResponse({"error": str(exc), "options": export_options()}, status=400) + except Exception as exc: # pragma: no cover - defensive; never 500 on a public page + logger.error( + "metadata_export_failed", dataset_id=str(dataset_id), standard=standard, error=str(exc) + ) + return JsonResponse({"error": "Could not generate the export"}, status=500) + + if request.GET.get("report") in ("1", "true", "yes"): + payload = {"standard": standard, "format": fmt, "report": report} + payload["document"] = json.loads(body) if fmt == "jsonld" else body + return JsonResponse(payload, json_dumps_params={"ensure_ascii": False}) + + response = HttpResponse(body, content_type=f"{content_type}; charset=utf-8") + filename = f"{slugify(dataset.slug or dataset.title) or 'dataset'}.{standard}.{ext}" + response["Content-Disposition"] = f'attachment; filename="{filename}"' + return response diff --git a/requirements.txt b/requirements.txt index aad4a741..ca9b8bb5 100644 --- a/requirements.txt +++ b/requirements.txt @@ -122,3 +122,4 @@ torch==2.9.0 transformers==4.57.1 sentencepiece==0.2.1 accelerate==1.11.0 +rdflib==7.6.0 From 18ea28686d5d6a520064f26d0b01241832e36dee Mon Sep 17 00:00:00 2001 From: Anant Jain Date: Mon, 28 Sep 2026 10:52:06 +0530 Subject: [PATCH 6/8] feat(api): map admin-defined metadata fields through import and export Metadata definitions (label, URN, type) created in Django admin were opaque strings: the platform import could not fill them and the export could not place them. metadata_mapping.py gives each definition a crosswalk concept, matched by URN (ds:createdOn -> created), then by a standard's own property name (dcterms:issued -> issued), then by label. Import: each enabled dataset definition is prefilled with the platform value for its concept (source page, creator, dates, licence, version, homepage, citation, languages); the definition's validators still apply. Export: definition values are read by concept. Core columns always win; a definition only supplies what the model has no column for. ds:createdOn becomes dcterms:created (the data's origin) while our created column stays dcterms:issued (when the record appeared). Definitions the mapping cannot place are listed under `dropped` in the report instead of vanishing. --- api/services/metadata_mapping.py | 236 ++++++++++++++++++++++++ api/services/platform_import_service.py | 36 ++-- 2 files changed, 248 insertions(+), 24 deletions(-) create mode 100644 api/services/metadata_mapping.py diff --git a/api/services/metadata_mapping.py b/api/services/metadata_mapping.py new file mode 100644 index 00000000..284f10ab --- /dev/null +++ b/api/services/metadata_mapping.py @@ -0,0 +1,236 @@ +"""Bridge between the deployment's *metadata definitions* and the crosswalk. + +Administrators define extra dataset fields in Django admin (``Metadata``: +label, URN, data type, required or optional). Publishers fill them in the +edit form and the values live in ``DatasetMetadata``. Until now those values +were opaque strings: the import could not prefill them and the export could +not place them in a standard. + +This module gives each definition a *crosswalk concept* so both sides can use +it. A definition is matched, in order, by + +1. its URN, once normalised: prefix dropped, camelCase to snake_case, so + ``ds:source_website``, ``dcterms:source`` and ``schema:sourceWebsite`` all + become ``source_website`` / ``source``; +2. any property name a standard uses for a concept in the contract + (``dcterms:issued`` -> ``issued``, ``dcat:landingPage`` -> ``landing_page``); +3. its label, normalised the same way ("Date of Creation of Dataset" -> + ``date_of_creation_of_dataset`` -> ``created``). + +Anything that resolves to nothing stays an opaque string: the import leaves +it for the publisher and the export lists it under ``dropped`` in the report, +so the gap is visible instead of silently lost. +""" + +from __future__ import annotations + +import re +from datetime import date, datetime +from functools import lru_cache +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple + +if TYPE_CHECKING: # pragma: no cover + from api.models import Dataset, Metadata + from api.services.platform_importers.base import PlatformDatasetInfo + +# House names and standard names that all mean the same crosswalk concept. +# Keys are normalised (see ``normalise``); values are concept keys in +# contracts/crosswalk.json, plus ``citation`` which the contract does not +# carry yet but Croissant (citeAs) does. +ALIASES: Dict[str, str] = { + # where the dataset came from + "source": "source", + "source_url": "source", + "source_website": "source", + "source_link": "source", + "original_source": "source", + "original_url": "source", + "data_source": "source", + "was_derived_from": "source", + "source_identifier": "source_identifier", + "source_id": "source_identifier", + "source_platform": "source_platform", + "imported_from": "source_platform", + # people and organisations + "author": "creator", + "creator": "creator", + "authors": "creator", + "original_author": "creator", + "publisher": "publisher", + # dates + "created": "created", + "created_on": "created", + "created_at": "created", + "creation_date": "created", + "date_created": "created", + "date_of_creation": "created", + "date_of_creation_of_dataset": "created", + "issued": "issued", + "published": "issued", + "published_on": "issued", + "date_published": "issued", + "release_date": "issued", + "modified": "modified", + "last_modified": "modified", + "last_updated": "modified", + "updated": "modified", + "updated_on": "modified", + "date_modified": "modified", + "source_last_updated": "modified", + # rights and identity + "license": "license", + "licence": "license", + "original_license": "license", + "source_license": "license", + "rights": "rights", + "version": "version", + "revision": "version", + # links and text + "homepage": "homepage", + "home_page": "homepage", + "website": "homepage", + "project_url": "homepage", + "landing_page": "landing_page", + "citation": "citation", + "cite_as": "citation", + "how_to_cite": "citation", + "language": "language", + "languages": "language", + "in_language": "language", + # coverage and cadence + "temporal_coverage_start": "temporal_coverage_start", + "temporal_start": "temporal_coverage_start", + "period_start": "temporal_coverage_start", + "start_date": "temporal_coverage_start", + "coverage_start": "temporal_coverage_start", + "temporal_coverage_end": "temporal_coverage_end", + "temporal_end": "temporal_coverage_end", + "period_end": "temporal_coverage_end", + "end_date": "temporal_coverage_end", + "coverage_end": "temporal_coverage_end", + "accrual_periodicity": "accrual_periodicity", + "frequency": "accrual_periodicity", + "update_frequency": "accrual_periodicity", + "periodicity": "accrual_periodicity", +} + +# Concepts whose value is a calendar date on the way in and out. +DATE_CONCEPTS = { + "created", + "issued", + "modified", + "temporal_coverage_start", + "temporal_coverage_end", +} +# Concepts whose value is a list joined with commas in the definition's cell. +LIST_CONCEPTS = {"language"} + +_CAMEL = re.compile(r"(?<=[a-z0-9])(?=[A-Z])") +_NON_WORD = re.compile(r"[^a-z0-9]+") + + +def normalise(name: Optional[str]) -> str: + """``ds:createdOn`` -> ``created_on``; ``Source Website`` -> ``source_website``.""" + if not name: + return "" + local = name.strip().rsplit(":", 1)[-1].rsplit("/", 1)[-1].rsplit("#", 1)[-1] + local = _CAMEL.sub("_", local) + return _NON_WORD.sub("_", local.lower()).strip("_") + + +@lru_cache(maxsize=1) +def _contract_property_index() -> Dict[str, str]: + """Normalised standard property name -> concept, from the vendored contract.""" + from api.services.metadata_export.crosswalk import Crosswalk + + index: Dict[str, str] = {} + for std in Crosswalk.load().standards.values(): + # ``export`` is a flat list of entries; ``import`` is nested + # node -> property -> [entries]. + entries: List[dict] = list(std.get("export") or []) + for node, by_property in (std.get("import") or {}).items(): + for prop, found in by_property.items(): + for entry in found if isinstance(found, list) else [found]: + entries.append(dict(entry, property=entry.get("property") or prop, node=node)) + for entry in entries: + prop, concept = entry.get("property"), entry.get("concept") + if prop and concept and entry.get("node", "dataset") == "dataset": + index.setdefault(normalise(prop), concept) + return index + + +def concept_for(urn: Optional[str], label: Optional[str] = None) -> Optional[str]: + """The crosswalk concept a definition stands for, or None if unrecognised.""" + for candidate in (normalise(urn), normalise(label)): + if not candidate: + continue + if candidate in ALIASES: + return ALIASES[candidate] + concept = _contract_property_index().get(candidate) + if concept: + return concept + return None + + +# --------------------------------------------------------------------- import +def platform_values(info: "PlatformDatasetInfo", platform_label: str) -> Dict[str, str]: + """What an import can offer each concept, as the string a definition cell holds.""" + + def day(value: Optional[datetime]) -> str: + return value.date().isoformat() if value else "" + + values = { + "source": info.source_url, + "source_identifier": info.identifier, + "source_platform": platform_label, + "creator": info.author, + "publisher": info.author, + "created": day(info.created_at), + "issued": day(info.created_at), + "modified": day(info.last_updated), + "license": info.license, + "version": info.revision, + "homepage": info.homepage, + "citation": info.citation, + "language": ", ".join(info.languages or []), + } + return {k: v for k, v in values.items() if v} + + +# --------------------------------------------------------------------- export +def _coerce(concept: str, raw: str) -> Any: + value = (raw or "").strip() + if not value: + return None + if concept in LIST_CONCEPTS: + return [v.strip() for v in value.split(",") if v.strip()] + if concept in DATE_CONCEPTS: + try: + return date.fromisoformat(value[:10]).isoformat() + except ValueError: + return value + return value + + +def definition_values(dataset: "Dataset") -> Tuple[Dict[str, Any], List[dict]]: + """Definition-backed values of a dataset, keyed by concept. + + Returns ``(values, unmapped)``. ``unmapped`` lists definitions the mapping + could not place, for the export report. + """ + values: Dict[str, Any] = {} + unmapped: List[dict] = [] + rows = dataset.metadata.select_related("metadata_item").all() + for row in rows: + item: "Metadata" = row.metadata_item + if not item.enabled: + continue + concept = concept_for(item.urn, item.label) + if concept is None: + unmapped.append({"label": item.label, "urn": item.urn, "value": row.value}) + continue + coerced = _coerce(concept, row.value) + if coerced in (None, "", []): + continue + values.setdefault(concept, coerced) + return values, unmapped diff --git a/api/services/platform_import_service.py b/api/services/platform_import_service.py index ffcaf5d8..0889cc9e 100644 --- a/api/services/platform_import_service.py +++ b/api/services/platform_import_service.py @@ -28,6 +28,7 @@ Sector, Tag, ) +from api.services import metadata_mapping from api.services.platform_importers import ( PlatformDatasetInfo, PlatformImportError, @@ -118,33 +119,20 @@ def _prefill_taxonomies(dataset: Dataset, tags: Iterable[str]) -> None: # Optional EAV prefill: if the deployment defines dataset metadata fields whose # label matches one of these (case-insensitive), fill it from the platform. # Deployments without such fields are simply skipped. -METADATA_LABEL_SOURCES = { - "source": "source_url", - "source url": "source_url", - "source platform": "platform_label", - "original source": "source_url", - "author": "author", - "creator": "author", - "publisher": "author", - "license": "license", - "original license": "license", - "last updated": "last_updated", - "source last updated": "last_updated", -} - - def _prefill_metadata(dataset: Dataset, info: PlatformDatasetInfo) -> None: - values = { - "source_url": info.source_url, - "platform_label": PLATFORM_LABELS.get(str(info.platform), str(info.platform).title()), - "author": info.author, - "license": info.license, - "last_updated": info.last_updated.date().isoformat() if info.last_updated else "", - } + """Fill the deployment's dataset metadata definitions from the platform. + + Each enabled definition is matched to a crosswalk concept by URN, then by + label (``api.services.metadata_mapping``); the platform value for that + concept, if any, becomes the definition's value. A definition whose + validators reject the value is skipped and left for the publisher. + """ + platform_label = PLATFORM_LABELS.get(str(info.platform), str(info.platform).title()) + values = metadata_mapping.platform_values(info, platform_label) fields = Metadata.objects.filter(enabled=True, model=MetadataModels.DATASET) for field in fields: - source_key = METADATA_LABEL_SOURCES.get((field.label or "").strip().lower()) - value = values.get(source_key or "", "") + concept = metadata_mapping.concept_for(field.urn, field.label) + value = values.get(concept or "", "") if not value: continue try: From 8968832a74b05ba33091b96e1a8d36798d60f96d Mon Sep 17 00:00:00 2001 From: Anant Jain Date: Mon, 28 Sep 2026 20:00:37 +0530 Subject: [PATCH 7/8] feat(api): align the Croissant export with the MLCommons 1.1 reference Use the official Croissant @context (every cr: term, @language, dct) instead of three bare prefixes, so validators recognise the document; type the dataset as sc:Dataset; declare conformsTo 1.1; and emit the source platform identifier as alternateName for imported datasets, as Hugging Face's own Croissant does. No change to the properties' meaning. --- api/services/metadata_export/adapter.py | 3 + .../metadata_export/contracts/crosswalk.json | 69 +++++++++++++++++-- api/services/metadata_export/exporter.py | 2 +- 3 files changed, 66 insertions(+), 8 deletions(-) diff --git a/api/services/metadata_export/adapter.py b/api/services/metadata_export/adapter.py index 2b7018ad..2726ab7f 100644 --- a/api/services/metadata_export/adapter.py +++ b/api/services/metadata_export/adapter.py @@ -178,6 +178,9 @@ def dataset_to_record(dataset: Dataset) -> Dict[str, Any]: else None ), # provenance and extras, filled for imports; definitions may fill the rest + "alternative_title": ( + [source.source_identifier] if source and source.source_identifier else [] + ), "source": source.source_url if source and source.source_url else None, "homepage": source.source_homepage if source and source.source_homepage else None, "version": source.revision if source and source.revision else None, diff --git a/api/services/metadata_export/contracts/crosswalk.json b/api/services/metadata_export/contracts/crosswalk.json index f27369b7..70707200 100644 --- a/api/services/metadata_export/contracts/crosswalk.json +++ b/api/services/metadata_export/contracts/crosswalk.json @@ -117,11 +117,11 @@ "value_type": "langstring", "repeatable": true, "obligation": "optional", - "dataspace_field": null, + "dataspace_field": "alternative_title", "dataspace_input_type": null, "visible_on_dataspace": false, "controlled_vocabulary": null, - "direction": "import_only", + "direction": "both", "notes": null }, "description": { @@ -1226,22 +1226,64 @@ "sha256": "checksum", "citation": "citation", "temporal_coverage_start": "temporal_coverage_start", - "temporal_coverage_end": "temporal_coverage_end" + "temporal_coverage_end": "temporal_coverage_end", + "alternative_title": "alternative_title" }, "standards": { "croissant": { "name": "Croissant (MLCommons ML-dataset metadata format)", - "version": "1.0", - "url": "https://mlcommons.org/croissant/", + "version": "1.1", + "url": "https://docs.mlcommons.org/croissant/docs/croissant-spec-1.1.html", "serialisation": "json-ld", "uri_style": "string", "context": { + "@language": "en", "@vocab": "https://schema.org/", + "arrayShape": "cr:arrayShape", + "citeAs": "cr:citeAs", + "column": "cr:column", + "conformsTo": "dct:conformsTo", + "containedIn": "cr:containedIn", "cr": "http://mlcommons.org/croissant/", - "sc": "https://schema.org/" + "data": { + "@id": "cr:data", + "@type": "@json" + }, + "dataBiases": "cr:dataBiases", + "dataCollection": "cr:dataCollection", + "dataType": { + "@id": "cr:dataType", + "@type": "@vocab" + }, + "dct": "http://purl.org/dc/terms/", + "extract": "cr:extract", + "field": "cr:field", + "fileProperty": "cr:fileProperty", + "fileObject": "cr:fileObject", + "fileSet": "cr:fileSet", + "format": "cr:format", + "includes": "cr:includes", + "isArray": "cr:isArray", + "isLiveDataset": "cr:isLiveDataset", + "jsonPath": "cr:jsonPath", + "key": "cr:key", + "md5": "cr:md5", + "parentField": "cr:parentField", + "path": "cr:path", + "personalSensitiveInformation": "cr:personalSensitiveInformation", + "recordSet": "cr:recordSet", + "references": "cr:references", + "regex": "cr:regex", + "repeated": "cr:repeated", + "replace": "cr:replace", + "sc": "https://schema.org/", + "separator": "cr:separator", + "source": "cr:source", + "subField": "cr:subField", + "transform": "cr:transform" }, "node_types": { - "dataset": "Dataset", + "dataset": "sc:Dataset", "distribution": "cr:FileObject", "record_set": "cr:RecordSet" }, @@ -1272,6 +1314,19 @@ "controlled_vocabulary": null, "notes": null }, + { + "concept": "alternative_title", + "property": "alternateName", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "alternative_title", + "controlled_vocabulary": null, + "notes": "For imported datasets: the identifier on the source platform." + }, { "concept": "description", "property": "description", diff --git a/api/services/metadata_export/exporter.py b/api/services/metadata_export/exporter.py index 45352600..185e22bd 100644 --- a/api/services/metadata_export/exporter.py +++ b/api/services/metadata_export/exporter.py @@ -15,7 +15,7 @@ from api.services.metadata_export.crosswalk import Crosswalk from api.services.metadata_export.formats import allowed_formats, serialise -CROISSANT_CONFORMS_TO = "http://mlcommons.org/croissant/1.0" +CROISSANT_CONFORMS_TO = "http://mlcommons.org/croissant/1.1" XSD_DATE = "http://www.w3.org/2001/XMLSchema#date" DATE_PROPERTIES = ("dcterms:issued", "dcterms:modified", "dcterms:created") PERIOD_PROPERTIES = ("dcat:startDate", "dcat:endDate") From c87bc5281f57b82dbc76139ceb5fd0fa451f47e9 Mon Sep 17 00:00:00 2001 From: Anant Jain Date: Tue, 29 Sep 2026 11:52:48 +0530 Subject: [PATCH 8/8] feat(api): make the URLs inside exported metadata configurable Exported documents carry absolute links: the dataset landing page and the download URL of each file. Read them from PUBLIC_SITE_URL and PUBLIC_API_URL (defaults: civicdataspace.in) so dev and staging emit their own domains. --- .env.example | 4 ++++ DataSpace/settings.py | 5 +++++ 2 files changed, 9 insertions(+) diff --git a/.env.example b/.env.example index f1a77c70..c72fbf5c 100644 --- a/.env.example +++ b/.env.example @@ -18,3 +18,7 @@ KAGGLE_USERNAME= KAGGLE_KEY= HF_TOKEN= GITHUB_TOKEN= + +# URLs written into exported metadata (DCAT / Croissant); set per environment +PUBLIC_SITE_URL=https://civicdataspace.in +PUBLIC_API_URL=https://api.civicdataspace.in diff --git a/DataSpace/settings.py b/DataSpace/settings.py index f3fcd32c..5220d244 100644 --- a/DataSpace/settings.py +++ b/DataSpace/settings.py @@ -299,6 +299,11 @@ GITHUB_TOKEN = os.getenv("GITHUB_TOKEN", None) # optional, lifts the 60 req/hour anonymous limit PLATFORM_IMPORT_TIMEOUT = float(os.getenv("PLATFORM_IMPORT_TIMEOUT", "15")) +# Absolute URLs written into exported metadata documents (dataset landing page, +# download links). Override on any environment that is not production. +PUBLIC_SITE_URL = os.getenv("PUBLIC_SITE_URL", "https://civicdataspace.in") +PUBLIC_API_URL = os.getenv("PUBLIC_API_URL", "https://api.civicdataspace.in") + # DVC settings DVC_REPO_PATH = os.path.join(BASE_DIR, "dvc") DVC_REMOTE_NAME = os.getenv("DVC_REMOTE_NAME", None)