diff --git a/.env.example b/.env.example index 58fefe7..c72fbf5 100644 --- a/.env.example +++ b/.env.example @@ -12,3 +12,13 @@ URL_WHITELIST=http://localhost:8000,http://localhost,http://localhost:3000 DEBUG=True SECRET_KEY=your-secret-key REDIS_URL=redis://redis:6379/1 + +# Third-party platform imports (optional) +KAGGLE_USERNAME= +KAGGLE_KEY= +HF_TOKEN= +GITHUB_TOKEN= + +# URLs written into exported metadata (DCAT / Croissant); set per environment +PUBLIC_SITE_URL=https://civicdataspace.in +PUBLIC_API_URL=https://api.civicdataspace.in diff --git a/DataSpace/settings.py b/DataSpace/settings.py index 4d25acc..5220d24 100644 --- a/DataSpace/settings.py +++ b/DataSpace/settings.py @@ -289,6 +289,21 @@ } +# Third-party platform imports (link-only). All three platforms work with no +# key for public datasets. KAGGLE_* adds a file count to Kaggle imports, +# HF_TOKEN unlocks gated Hugging Face repos, GITHUB_TOKEN lifts GitHub's +# anonymous rate limit. +KAGGLE_USERNAME = os.getenv("KAGGLE_USERNAME", None) +KAGGLE_KEY = os.getenv("KAGGLE_KEY", None) +HF_TOKEN = os.getenv("HF_TOKEN", None) +GITHUB_TOKEN = os.getenv("GITHUB_TOKEN", None) # optional, lifts the 60 req/hour anonymous limit +PLATFORM_IMPORT_TIMEOUT = float(os.getenv("PLATFORM_IMPORT_TIMEOUT", "15")) + +# Absolute URLs written into exported metadata documents (dataset landing page, +# download links). Override on any environment that is not production. +PUBLIC_SITE_URL = os.getenv("PUBLIC_SITE_URL", "https://civicdataspace.in") +PUBLIC_API_URL = os.getenv("PUBLIC_API_URL", "https://api.civicdataspace.in") + # DVC settings DVC_REPO_PATH = os.path.join(BASE_DIR, "dvc") DVC_REMOTE_NAME = os.getenv("DVC_REMOTE_NAME", None) diff --git a/api/migrations/0048_platform_import.py b/api/migrations/0048_platform_import.py new file mode 100644 index 0000000..9efe2d0 --- /dev/null +++ b/api/migrations/0048_platform_import.py @@ -0,0 +1,81 @@ +# Generated by Django 5.0.4 on 2026-09-23 20:36 + +import uuid + +import django.db.models.deletion +from django.conf import settings +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ("api", "0047_resourcetype_publication_collaborative_publications_and_more"), + migrations.swappable_dependency(settings.AUTH_USER_MODEL), + ] + + operations = [ + migrations.CreateModel( + name="DatasetSource", + fields=[ + ( + "id", + models.UUIDField( + default=uuid.uuid4, editable=False, primary_key=True, serialize=False + ), + ), + ( + "platform", + models.CharField( + choices=[ + ("KAGGLE", "Kaggle"), + ("HUGGINGFACE", "Huggingface"), + ("GITHUB", "Github"), + ], + max_length=50, + ), + ), + ("source_identifier", models.CharField(max_length=300)), + ("source_url", models.URLField(max_length=500)), + ("source_homepage", models.URLField(blank=True, max_length=500)), + ("revision", models.CharField(blank=True, max_length=64)), + ("source_author", models.CharField(blank=True, max_length=300)), + ("source_license", models.CharField(blank=True, max_length=300)), + ("source_readme", models.TextField(blank=True)), + ("citation", models.TextField(blank=True)), + ("languages", models.JSONField(blank=True, default=list)), + ("source_created_at", models.DateTimeField(blank=True, null=True)), + ("source_last_updated", models.DateTimeField(blank=True, null=True)), + ("is_archived", models.BooleanField(default=False)), + ("imported_at", models.DateTimeField(auto_now_add=True)), + ("last_synced_at", models.DateTimeField(auto_now=True)), + ( + "dataset", + models.OneToOneField( + on_delete=django.db.models.deletion.CASCADE, + related_name="source", + to="api.dataset", + ), + ), + ( + "imported_by", + models.ForeignKey( + blank=True, + null=True, + on_delete=django.db.models.deletion.SET_NULL, + related_name="imported_dataset_sources", + to=settings.AUTH_USER_MODEL, + ), + ), + ], + options={ + "db_table": "dataset_source", + "indexes": [ + models.Index( + fields=["platform", "source_identifier"], + name="dataset_sou_platfor_38ca23_idx", + ) + ], + }, + ), + ] diff --git a/api/models/Dataset.py b/api/models/Dataset.py index a2b99a1..296f509 100644 --- a/api/models/Dataset.py +++ b/api/models/Dataset.py @@ -1,5 +1,5 @@ import uuid -from typing import TYPE_CHECKING, Any +from typing import TYPE_CHECKING, Any, Optional from django.db import models from django.db.models import Sum @@ -124,14 +124,22 @@ def formats_indexing(self) -> list[str]: Used in Elasticsearch indexing. """ - return list( - set( - [ - resource.resourcefiledetails.format # type: ignore - for resource in self.resources.all() - ] - ).difference({""}) - ) + formats: set[str] = set() + for resource in self.resources.all(): + # Link-only (EXTERNAL) resources have no file details; skip them. + file_details = getattr(resource, "resourcefiledetails", None) + if file_details is not None and file_details.format: + formats.add(file_details.format) + return list(formats) + + @property + def source_platform_indexing(self) -> Optional[str]: + """Platform this dataset was imported from, or None for native datasets. + + Used in Elasticsearch indexing. + """ + source = getattr(self, "source", None) + return source.platform if source is not None else None @property def catalogs_indexing(self) -> list[str]: diff --git a/api/models/DatasetSource.py b/api/models/DatasetSource.py new file mode 100644 index 0000000..ccf0b2a --- /dev/null +++ b/api/models/DatasetSource.py @@ -0,0 +1,69 @@ +import uuid + +from django.db import models + +from api.utils.enums import ImportPlatform + + +class DatasetSource(models.Model): + """Provenance record for a dataset imported from a third-party platform. + + Imports are link-only: DataSpace never copies the platform's files. This + row holds what the platform told us about the dataset, in typed columns. + Every column here has a reader: either a metadata standard on export + (DCAT / Croissant / Dublin Core) or the platform itself (attribution, + duplicate detection, licence review). No raw payload is kept. + """ + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + dataset = models.OneToOneField("api.Dataset", on_delete=models.CASCADE, related_name="source") + platform = models.CharField(max_length=50, choices=ImportPlatform.choices) + + # --- identity on the platform ------------------------------------------ + # Platform-native identifier, e.g. "owner/dataset-slug" (Kaggle) or + # "namespace/name" (Hugging Face). Normalised by the importer. + source_identifier = models.CharField(max_length=300) + # Human-facing page on the platform. Export: schema:sameAs / prov:wasDerivedFrom. + source_url = models.URLField(max_length=500) + # Home page declared by the source, if any (GitHub `homepage`). Export: dcat:landingPage. + source_homepage = models.URLField(max_length=500, blank=True) + # Commit hash (Hugging Face / GitHub) or version number (Kaggle) at import time. + # Export: Croissant `version`. Later: what a sync compares against. + revision = models.CharField(max_length=64, blank=True) + + # --- descriptive metadata the standards read ---------------------------- + # Who made the data on the platform. Export: dcterms:creator / Croissant creator. + source_author = models.CharField(max_length=300, blank=True) + # License string exactly as the platform reported it (may not map onto + # DatasetLicense; the mapped value lives on Dataset.license). + source_license = models.CharField(max_length=300, blank=True) + # Full dataset card / README. Dataset.description keeps a 1,000-char cut. + source_readme = models.TextField(blank=True) + # BibTeX or free-text citation, when the platform provides one. Export: Croissant citeAs. + citation = models.TextField(blank=True) + # Language codes of the data, e.g. ["en", "hi"]. Export: dcterms:language / inLanguage. + languages = models.JSONField(default=list, blank=True) + # When the dataset was first published on the platform. Export: dcterms:issued. + source_created_at = models.DateTimeField(null=True, blank=True) + # When the platform last changed it. Export: dcterms:modified. + source_last_updated = models.DateTimeField(null=True, blank=True) + # Source is frozen / read-only upstream (GitHub `archived`). Shown as a hint. + is_archived = models.BooleanField(default=False) + + # --- our side ------------------------------------------------------------- + imported_by = models.ForeignKey( + "authorization.User", + on_delete=models.SET_NULL, + null=True, + blank=True, + related_name="imported_dataset_sources", + ) + imported_at = models.DateTimeField(auto_now_add=True) + last_synced_at = models.DateTimeField(auto_now=True) + + class Meta: + db_table = "dataset_source" + indexes = [models.Index(fields=["platform", "source_identifier"])] + + def __str__(self) -> str: + return f"{self.platform}:{self.source_identifier}" diff --git a/api/models/__init__.py b/api/models/__init__.py index 6c54b34..19f7c51 100644 --- a/api/models/__init__.py +++ b/api/models/__init__.py @@ -9,6 +9,7 @@ ) from api.models.Dataset import Dataset, Tag from api.models.DatasetMetadata import DatasetMetadata +from api.models.DatasetSource import DatasetSource from api.models.DataSpace import DataSpace from api.models.Geography import Geography from api.models.Metadata import Metadata diff --git a/api/schema/platform_import_schema.py b/api/schema/platform_import_schema.py new file mode 100644 index 0000000..12de66a --- /dev/null +++ b/api/schema/platform_import_schema.py @@ -0,0 +1,99 @@ +"""GraphQL surface for link-only imports from third-party platforms. + +- ``preview_platform_dataset`` fetches normalised metadata, no side effects. +- ``import_platform_dataset`` creates a DRAFT dataset with one EXTERNAL resource + linking to the dataset page on the platform (files are not imported). + +Both take the same organization/dataspace request headers as ``add_dataset``; +the imported dataset is owned the same way a manually created one would be. +""" + +from typing import Optional + +import strawberry +from strawberry.types import Info + +from api.schema.base_mutation import ( + BaseMutation, + GraphQLValidationError, + MutationResponse, +) +from api.services.platform_import_service import ( + import_platform_dataset, + preview_platform_dataset, +) +from api.services.platform_importers import PlatformImportError +from api.types.type_dataset import TypeDataset +from api.types.type_dataset_source import ( + TypePlatformDatasetPreview, + import_platform_enum, +) +from api.utils.graphql_telemetry import trace_resolver +from authorization.graphql_permissions import IsAuthenticated +from authorization.permissions import CreateDatasetPermission + + +@strawberry.input +class ImportPlatformDatasetInput: + platform: import_platform_enum # type: ignore + #: Short id ("owner/name") or a pasted platform URL. + identifier: str + #: Optional display title on DataSpace; defaults to the platform's title. + title: Optional[str] = None + + +@strawberry.type +class Query: + @strawberry.field(permission_classes=[IsAuthenticated]) + @trace_resolver(name="preview_platform_dataset", attributes={"component": "platform_import"}) + def preview_platform_dataset( + self, info: Info, platform: import_platform_enum, identifier: str # type: ignore + ) -> TypePlatformDatasetPreview: + """Look up a Hugging Face / GitHub / Kaggle dataset and show what an import would create.""" + try: + data = preview_platform_dataset(platform.value, identifier) + except PlatformImportError as exc: + # Surface the importer's user-safe message as a GraphQL error. + raise ValueError(exc.message) from exc + return TypePlatformDatasetPreview.from_info(data) + + +@strawberry.type +class Mutation: + @strawberry.mutation + @BaseMutation.mutation( + permission_classes=[IsAuthenticated, CreateDatasetPermission], + trace_name="import_platform_dataset", + trace_attributes={"component": "platform_import"}, + track_activity={ + "verb": "imported", + "get_data": lambda result, import_input=None, **kwargs: { + "dataset_id": str(result.id), + "dataset_title": result.title, + "platform": import_input.platform.value if import_input else None, + "identifier": import_input.identifier if import_input else None, + "organization": (str(result.organization.id) if result.organization else None), + }, + }, + ) + def import_platform_dataset( + self, info: Info, import_input: ImportPlatformDatasetInput + ) -> MutationResponse[TypeDataset]: + """Create a DRAFT dataset that links to the dataset on the platform.""" + organization = info.context.context.get("organization") + dataspace = info.context.context.get("dataspace") + user = info.context.user + + try: + dataset = import_platform_dataset( + platform=import_input.platform.value, + identifier=import_input.identifier, + user=user, + organization=organization, + dataspace=dataspace, + title=import_input.title, + ) + except PlatformImportError as exc: + return MutationResponse.error_response(GraphQLValidationError.from_message(exc.message)) + + return MutationResponse.success_response(TypeDataset.from_django(dataset)) diff --git a/api/schema/schema.py b/api/schema/schema.py index 4678b63..5f919cf 100644 --- a/api/schema/schema.py +++ b/api/schema/schema.py @@ -16,6 +16,7 @@ import api.schema.metadata_schema import api.schema.organization_data_schema import api.schema.organization_schema +import api.schema.platform_import_schema import api.schema.publication_schema import api.schema.resource_chart_schema import api.schema.resource_schema @@ -77,6 +78,7 @@ def tags(self, info: Info) -> List[TypeTag]: api.schema.user_schema.Query, api.schema.collaborative_schema.Query, api.schema.publication_schema.Query, + api.schema.platform_import_schema.Query, AuthQuery, ), ) @@ -100,6 +102,7 @@ def tags(self, info: Info) -> List[TypeTag]: api.schema.tags_schema.Mutation, api.schema.collaborative_schema.Mutation, api.schema.publication_schema.Mutation, + api.schema.platform_import_schema.Mutation, AuthMutation, ), ) diff --git a/api/services/metadata_export/__init__.py b/api/services/metadata_export/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/api/services/metadata_export/adapter.py b/api/services/metadata_export/adapter.py new file mode 100644 index 0000000..2726ab7 --- /dev/null +++ b/api/services/metadata_export/adapter.py @@ -0,0 +1,205 @@ +"""Turn a Dataset (and its resources / source) into the flat record the +crosswalk engine reads. This is the only module in the export that knows +Django models; the engine only ever sees plain dicts keyed by the contract's +``dataspace_field`` names. Names that a standard wants as identifiers +(licence, sector, geography, language, access rights) leave here already +resolved to URIs. +""" + +from __future__ import annotations + +from typing import Any, Dict, List, Optional + +from django.conf import settings + +from api.models import Dataset, Resource +from api.services import metadata_mapping +from api.services.metadata_export import vocabularies as vocab +from api.utils.enums import DataType + +EMPTY = (None, "", [], {}) + + +def _public_base() -> str: + return getattr(settings, "PUBLIC_SITE_URL", "https://civicdataspace.in").rstrip("/") + + +def _api_base() -> str: + return getattr(settings, "PUBLIC_API_URL", "https://api.civicdataspace.in").rstrip("/") + + +def _iso(dt: Any) -> Optional[str]: + return dt.date().isoformat() if dt else None + + +def _is_url(value: Any) -> bool: + return isinstance(value, str) and value.startswith(("http://", "https://")) + + +# DCAT-AP mandates the EU Access Right authority list for dcterms:accessRights. +ACCESS_RIGHTS_URI = { + "PUBLIC": "http://publications.europa.eu/resource/authority/access-right/PUBLIC", + "RESTRICTED": "http://publications.europa.eu/resource/authority/access-right/RESTRICTED", + "PRIVATE": "http://publications.europa.eu/resource/authority/access-right/NON_PUBLIC", +} + +# Languages are ISO 639-1 codes; the Library of Congress scheme gives each a URI. +LANGUAGE_SCHEME = "http://id.loc.gov/vocabulary/iso639-1/" + +# Our format labels -> IANA media types. DCAT-AP wants dcat:mediaType to be an +# IANA IRI and Croissant wants encodingFormat to be a MIME string; both are +# derived from this. Unknown labels pass through lower-cased so nothing is lost. +MEDIA_TYPES = { + "CSV": "text/csv", + "TSV": "text/tab-separated-values", + "TXT": "text/plain", + "JSON": "application/json", + "GEOJSON": "application/geo+json", + "XML": "application/xml", + "PDF": "application/pdf", + "XLSX": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", + "XLS": "application/vnd.ms-excel", + "ODS": "application/vnd.oasis.opendocument.spreadsheet", + "PARQUET": "application/vnd.apache.parquet", + "ZIP": "application/zip", + "HTML": "text/html", +} +IANA_BASE = "https://www.iana.org/assignments/media-types/" + + +def media_type(label: Optional[str]) -> Optional[str]: + if not label: + return None + key = str(label).strip().upper().lstrip(".") + return MEDIA_TYPES.get(key) or (label if "/" in str(label) else str(label).lower()) + + +def _dataset_type_concept(value: Optional[str]) -> Optional[dict]: + """Our own type vocabulary (DATA / PROMPT), identified under the platform namespace.""" + if not value: + return None + return vocab.concept(value, f"{_public_base()}/id/dataset-type/{str(value).lower()}") + + +def _agent( + name: Optional[str], uri: Optional[str] = None, kind: str = "foaf:Organization" +) -> Optional[dict]: + return {"name": name, "uri": uri, "@type": kind} if name else None + + +def _languages(codes: Any) -> List[dict]: + out: List[dict] = [] + for code in codes or []: + code = str(code).strip().lower() + if code: + out.append(vocab.concept(code, LANGUAGE_SCHEME + code)) + return out + + +def resource_to_record(resource: Resource) -> Dict[str, Any]: + file_details = getattr(resource, "resourcefiledetails", None) + rec: Dict[str, Any] = { + "name": resource.name, + "description": resource.description or None, + "format": media_type(file_details.format if file_details else None), + "size": (file_details.size if file_details else None) or None, + "sha256": None, # not stored yet; see the file-hash follow-up + "download_url": None, + "access_url": None, + "columns": [ + {"name": s.field_name, "type": s.format, "description": s.description or None} + for s in resource.resourceschema_set.all() + ], + "_id": str(resource.id), + "_kind": "link" if resource.type == DataType.EXTERNAL else "file", + } + if resource.type == DataType.EXTERNAL: + # Link-only: the platform page gives *access*; there is no direct file. + rec["access_url"] = resource.url or None + rec["format"] = rec["format"] or "text/html" + elif file_details and file_details.file: + rec["download_url"] = f"{_api_base()}/api/download/resource/{resource.id}" + rec["access_url"] = rec["download_url"] + return rec + + +def dataset_to_record(dataset: Dataset) -> Dict[str, Any]: + source = getattr(dataset, "source", None) + org = dataset.organization + user = dataset.user + landing = f"{_public_base()}/datasets/{dataset.slug}" + + sectors = [vocab.concept(s.name, vocab.sector_uri(s.name)) for s in dataset.sectors.all()] + geographies = [ + vocab.concept(g.name, vocab.geography_uri(g.name, getattr(g, "type", None))) + for g in dataset.geographies.all() + ] + license_value = dataset.license or None + license_uri = vocab.license_uri(license_value) or ( + vocab.license_uri(source.source_license) if source and source.source_license else None + ) + + # Who made the data. For an imported dataset that is the platform author, + # not the person who clicked import; our organisation stays the publisher. + creator = _agent((user.get_full_name() or user.username) if user else None, kind="foaf:Person") + if source and source.source_author: + creator = _agent(source.source_author, kind="foaf:Agent") + + record: Dict[str, Any] = { + "id": str(dataset.id), + "slug": dataset.slug, + "title": dataset.title, + "description": dataset.description or None, + "tags": [t.value for t in dataset.tags.all()], + "sectors": sectors, + "geographies": geographies, + "license": license_uri or license_value, + "organization": _agent(org.name if org else None), + "user": creator, + # when the record appeared (here, or on the platform it came from) + "issued": ( + _iso(source.source_created_at) + if source and source.source_created_at + else _iso(dataset.created) + ), + "modified": _iso(dataset.modified), + # when the data itself came into being: only a definition can say + "created": None, + "datasetType": _dataset_type_concept(dataset.dataset_type), + "accessType": vocab.concept( + dataset.access_type, ACCESS_RIGHTS_URI.get(str(dataset.access_type)) + ), + "status": dataset.status, + "resources": [resource_to_record(r) for r in dataset.resources.all()], + "landing_page": landing, + "in_catalog": ( + f"{_public_base()}/dataspaces/{dataset.dataspace.slug}" + if dataset.dataspace_id and getattr(dataset.dataspace, "slug", None) + else None + ), + # provenance and extras, filled for imports; definitions may fill the rest + "alternative_title": ( + [source.source_identifier] if source and source.source_identifier else [] + ), + "source": source.source_url if source and source.source_url else None, + "homepage": source.source_homepage if source and source.source_homepage else None, + "version": source.revision if source and source.revision else None, + "citation": source.citation if source and source.citation else None, + "language": _languages(source.languages) if source and source.languages else [], + } + + # Definition-backed values (admin-defined fields the publisher filled in), + # keyed by crosswalk concept. Core columns always win; a definition only + # supplies what the model has no column for. Unplaceable definitions are + # carried for the report. + defined, unmapped = metadata_mapping.definition_values(dataset) + for concept, value in defined.items(): + if record.get(concept) not in EMPTY: + continue + if concept == "language": + value = _languages(value) + elif concept in ("source", "homepage") and not _is_url(value): + continue + record[concept] = value + record["_unmapped_definitions"] = unmapped + return record diff --git a/api/services/metadata_export/contracts/README.md b/api/services/metadata_export/contracts/README.md new file mode 100644 index 0000000..298ea04 --- /dev/null +++ b/api/services/metadata_export/contracts/README.md @@ -0,0 +1,20 @@ +# Metadata contract + +These files are owned by DataSpaceBackend and are the source of truth for the +metadata export. Edit them here. + +- `crosswalk.json` — the mapping: one entry per concept, and for each standard + the property it becomes, its value type and obligation, plus the concepts the + standard cannot carry (`gaps`). The engine in `../crosswalk.py` reads it and + knows nothing about any standard by name. +- `licenses.csv`, `sectors.csv`, `geographies.csv` — allowed values with URIs. + `alt_codes` carries `dataspace_enum=` for licences, which is + how a platform value finds its row. Geographies cover regions, states, union + territories and districts. + +Changing a mapping decision means editing `crosswalk.json` in a PR, like any +other code. Adding a standard means adding a block under `standards`. + +Origin: the first version was copied from CivicDataLab/DataSpace-data-ecosystem +at commit 83e577752784 (Sep 2026) and has been edited here since. There is no +runtime or build-time link to that repository. diff --git a/api/services/metadata_export/contracts/crosswalk.json b/api/services/metadata_export/contracts/crosswalk.json new file mode 100644 index 0000000..7070720 --- /dev/null +++ b/api/services/metadata_export/contracts/crosswalk.json @@ -0,0 +1,4902 @@ +{ + "version": 2, + "value_types": { + "literal": "A plain string.", + "langstring": "A string that may carry a language tag.", + "date": "ISO 8601 date (xsd:date).", + "datetime": "ISO 8601 datetime (xsd:dateTime).", + "uri": "An absolute URI, serialised as a node reference not a string.", + "number": "A numeric literal.", + "bytes": "A non-negative integer count of bytes.", + "duration": "ISO 8601 duration (xsd:duration).", + "media_type": "An IANA media type, or a format token where none exists.", + "agent": "A person or organisation; a node with at least a name.", + "concept": "A term from a controlled scheme; prefer its URI.", + "location": "A place; prefer a resolvable URI over a name string.", + "period": "A time interval, expressed as start and end.", + "frequency": "A term from the Dublin Core Frequency vocabulary.", + "checksum": "A hash digest, with its algorithm.", + "vcard": "A contact, serialised as a vCard node.", + "structured": "A nested object whose shape the standard defines." + }, + "nodes": [ + "contact_point", + "dataset", + "distribution", + "period", + "record_set" + ], + "directions": { + "both": "Read on import, write on export.", + "export_only": "Platform-derived. Emit it; never let an import overwrite it.", + "import_only": "Accept from a source; the platform does not re-emit it.", + "none": "Platform-internal. Not metadata; ignore in both directions." + }, + "controlled_vocabularies": { + "license": { + "path": "licenses.csv", + "key_field": "key", + "uri_field": "uri" + }, + "sector": { + "path": "sectors.csv", + "key_field": "key", + "uri_field": "uri" + }, + "geography": { + "path": "geographies.csv", + "key_field": "key", + "uri_field": "uri" + }, + "language": { + "scheme": "http://id.loc.gov/vocabulary/iso639-1/", + "notes": "ISO 639-1 code appended to the scheme" + } + }, + "concepts": { + "identifier": { + "label": "Identifier", + "definition": "A unique identifier for the dataset (platform ID, DOI, catalog UUID).", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "id", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "export_only", + "notes": "Croissant uses @id for this. On import the incoming identifier is kept in source_identifier, never written over the platform's own ID." + }, + "source_identifier": { + "label": "Source identifier", + "definition": "The identifier this dataset carried on the platform it was imported from.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Not in the source workbook. Added because round-tripping a federated record is impossible without somewhere to keep its origin ID." + }, + "slug": { + "label": "Slug", + "definition": "URL-safe short name for the dataset on the platform.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "slug", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "export_only", + "notes": "No standard equivalent; it is the path component of landing_page." + }, + "title": { + "label": "Title", + "definition": "A name given to the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "title", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "alternative_title": { + "label": "Alternative title", + "definition": "An alternative name for the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "dataspace_field": "alternative_title", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "description": { + "label": "Description", + "definition": "A free-text account of the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "description", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "abstract": { + "label": "Abstract", + "definition": "A summary of the dataset, shorter than the description.", + "node": "dataset", + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "keyword": { + "label": "Keywords", + "definition": "Free-text tags describing the dataset, used for search.", + "node": "dataset", + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": "tags", + "dataspace_input_type": "semi_controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "theme": { + "label": "Theme / Sector", + "definition": "The category the dataset belongs to, from a controlled scheme.", + "node": "dataset", + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": "sectors", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": "sector", + "direction": "both", + "notes": "Partial match. CDL sectors are the platform's own scheme; export maps them onto EU data themes, which collapses several sectors onto one theme." + }, + "dataset_type": { + "label": "Dataset type", + "definition": "The nature or genre of the dataset.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "datasetType", + "dataspace_input_type": "controlled", + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The platform's set of types is undefined in the source workbook." + }, + "language": { + "label": "Language", + "definition": "Language of the dataset, as an ISO 639 code or URI.", + "node": "dataset", + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": "language", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "Requires normalisation to ISO codes/URIs on import." + }, + "publisher": { + "label": "Publisher", + "definition": "The entity responsible for making the dataset available.", + "node": "dataset", + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "organization", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The workbook maps organization to dct:publisher but Croissant creator, and user to dct:creator but Croissant publisher. That inversion is preserved here as recorded; see the open question in the README." + }, + "creator": { + "label": "Creator", + "definition": "The entity responsible for producing the dataset.", + "node": "dataset", + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "dataspace_field": "user", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "qualified_attribution": { + "label": "Qualified attribution", + "definition": "A role-qualified link to an agent, for multiple contributors.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Needed when a dataset has several publishers with distinct roles." + }, + "contact_point": { + "label": "Contact point", + "definition": "Contact information for the dataset, as a vCard.", + "node": "dataset", + "value_type": "vcard", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Sub-fields are the contact_* concepts on the contact_point node." + }, + "license": { + "label": "License", + "definition": "The legal document under which the dataset is made available.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "license", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": "license", + "direction": "both", + "notes": "Must serialise as a resolvable LicenseDocument URI, not the enum label. Resolve through the license vocabulary before emitting." + }, + "rights": { + "label": "Rights statement", + "definition": "A statement about rights held in and over the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "access_rights": { + "label": "Access rights", + "definition": "Whether the dataset is public, restricted, or non-public.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "accessType", + "dataspace_input_type": "controlled", + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The workbook calls accessType unclear and marks it unmappable. It lines up with dcterms:accessRights if it means public/restricted; confirm against the platform before relying on this binding." + }, + "issued": { + "label": "Date issued", + "definition": "The date the dataset was first published.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "issued", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The platform's \"created\" is mapped to issued, not dcterms:created. The workbook flags that it is unclear whether this is the original creation date or the upload date." + }, + "created": { + "label": "Date created", + "definition": "The date the dataset itself was created, before publication.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "created", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "modified": { + "label": "Date modified", + "definition": "The most recent date the dataset was changed.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "modified", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "Should be driven by the platform's versioning, per the workbook." + }, + "accrual_periodicity": { + "label": "Update frequency", + "definition": "How often the dataset is updated.", + "node": "dataset", + "value_type": "frequency", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Use the Dublin Core Frequency vocabulary." + }, + "accrual_method": { + "label": "Accrual method", + "definition": "How the dataset is added to.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "version": { + "label": "Version", + "definition": "The version designator of the dataset.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "version", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "DCAT has no single version field; the workbook records the combination of issued + modified + dcterms:hasVersion as the partial match." + }, + "spatial_coverage": { + "label": "Spatial coverage", + "definition": "The geographic region the dataset covers.", + "node": "dataset", + "value_type": "location", + "repeatable": true, + "obligation": "recommended", + "dataspace_field": "geographies", + "dataspace_input_type": "controlled", + "visible_on_dataspace": true, + "controlled_vocabulary": "geography", + "direction": "both", + "notes": "Must serialise as a resolvable place URI. The platform stores name strings today; resolve through the geography vocabulary first." + }, + "spatial_resolution": { + "label": "Spatial resolution", + "definition": "Minimum spatial separation resolvable in the dataset.", + "node": "dataset", + "value_type": "number", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT expects metres. The registry sheet uses admin level (district, village), which needs a conversion or a separate literal field." + }, + "temporal_coverage_start": { + "label": "Temporal coverage (start)", + "definition": "Start of the period the dataset covers.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "temporal_coverage_start", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "Serialises inside a dcterms:temporal PeriodOfTime node." + }, + "temporal_coverage_end": { + "label": "Temporal coverage (end)", + "definition": "End of the period the dataset covers.", + "node": "dataset", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "temporal_coverage_end", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "temporal_resolution": { + "label": "Temporal resolution", + "definition": "Minimum time period resolvable in the dataset.", + "node": "dataset", + "value_type": "duration", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "landing_page": { + "label": "Landing page", + "definition": "A web page giving access to the dataset and its publisher.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "landing_page", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "export_only", + "notes": "Constructed from slug at export time." + }, + "homepage": { + "label": "Homepage", + "definition": "The publisher's home page for the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "homepage", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "source": { + "label": "Source", + "definition": "A related resource the dataset is derived from.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": "source", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "depiction": { + "label": "Depiction", + "definition": "An image representing the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "in_catalog": { + "label": "In catalog", + "definition": "The catalog the dataset is listed in.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "in_catalog", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "export_only", + "notes": null + }, + "in_series": { + "label": "In series", + "definition": "The dataset series this dataset belongs to.", + "node": "dataset", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "conforms_to": { + "label": "Conforms to", + "definition": "An established standard or schema the dataset conforms to.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Unbound. Definition values export by concept through metadata_mapping.py." + }, + "relation": { + "label": "Related resource", + "definition": "A resource with an unspecified relationship to the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "has_part": { + "label": "Has part", + "definition": "A resource included either physically or logically in the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "is_part_of": { + "label": "Is part of", + "definition": "A resource the dataset is included in.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "is_referenced_by": { + "label": "Is referenced by", + "definition": "A resource that references the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "replaces": { + "label": "Replaces", + "definition": "A resource the dataset supplants.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "is_replaced_by": { + "label": "Is replaced by", + "definition": "A resource that supplants the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "is_version_of": { + "label": "Is version of", + "definition": "A resource the dataset is a version of.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "has_version": { + "label": "Has version", + "definition": "A version of the dataset.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "provenance": { + "label": "Provenance", + "definition": "A statement of changes in ownership and custody of the dataset.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT needs the external PROV-O ontology for this." + }, + "jurisdiction_level": { + "label": "Jurisdiction level", + "definition": "The administrative level the dataset's authority operates at.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT-IN, not yet a stable published spec." + }, + "applicable_legislation": { + "label": "Applicable legislation", + "definition": "The legislation mandating the dataset's creation or publication.", + "node": "dataset", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT-IN." + }, + "hvd_category": { + "label": "High-value dataset category", + "definition": "The high-value dataset category the dataset falls in.", + "node": "dataset", + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT-IN." + }, + "note": { + "label": "Note", + "definition": "A free-text note about the dataset.", + "node": "dataset", + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT-IN." + }, + "distribution": { + "label": "Distribution", + "definition": "An accessible form of the dataset — a downloadable file or an API.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "dataspace_field": "resources", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The link from dataset to its distribution nodes." + }, + "access_url": { + "label": "Access URL", + "definition": "A URL giving access to the distribution.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "dataspace_field": "access_url", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "DCAT requires this on every Distribution. Where the platform has only a download URL, use it for both." + }, + "download_url": { + "label": "Download URL", + "definition": "A direct link to the downloadable file.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "download_url", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "media_type": { + "label": "Format / media type", + "definition": "The file format of the distribution.", + "node": "distribution", + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": "format", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": "The platform's dataset-level \"formats\" is the set of its resources' formats, derived rather than entered." + }, + "byte_size": { + "label": "Byte size", + "definition": "The size of the distribution in bytes.", + "node": "distribution", + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "size", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "checksum": { + "label": "Checksum", + "definition": "A hash of the distribution's contents, for integrity checking.", + "node": "distribution", + "value_type": "checksum", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "sha256", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "both", + "notes": "Croissant requires sha256 on a FileObject; DCAT has no such property." + }, + "distribution_title": { + "label": "Distribution title", + "definition": "A name given to the distribution.", + "node": "distribution", + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "name", + "dataspace_input_type": "free_text", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "both", + "notes": null + }, + "endpoint_url": { + "label": "Endpoint URL", + "definition": "The root location of an API-based distribution.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "endpoint_description": { + "label": "Endpoint description", + "definition": "A description of the services available at the endpoint.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "file_set": { + "label": "File set", + "definition": "A group of files described by a pattern rather than individually.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT has no grouping semantics; the nearest is many Distributions." + }, + "containment": { + "label": "Containment", + "definition": "Which file or archive a file is contained in.", + "node": "distribution", + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "DCAT cannot model file hierarchy." + }, + "record_set": { + "label": "Record set", + "definition": "A table within the dataset — its columns and their semantics.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "The main DCAT gap. This is the concept that makes a dataset ML-readable, and it is also what schemas/data_dictionary_template.csv captures." + }, + "field": { + "label": "Field", + "definition": "One column of a record set, with a name, description and type.", + "node": "record_set", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "No column-level metadata exists in DCAT." + }, + "data_type": { + "label": "Data type", + "definition": "The semantic type of a field (including ML types like BoundingBox).", + "node": "record_set", + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "transform": { + "label": "Transform", + "definition": "A transformation applied to source data to produce a field.", + "node": "record_set", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Expressible in DCAT only through external PROV-O." + }, + "annotation": { + "label": "Annotation", + "definition": "A structured ML annotation over the data.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_name": { + "label": "Contact name", + "definition": "Formatted name of the contact.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_title": { + "label": "Contact title", + "definition": "Job title of the contact.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_role": { + "label": "Contact role", + "definition": "Role the contact plays for this dataset.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_organization": { + "label": "Contact organisation", + "definition": "Organisation the contact belongs to.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_email": { + "label": "Contact email", + "definition": "Email address of the contact.", + "node": "contact_point", + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Serialises as a mailto: URI." + }, + "contact_address": { + "label": "Contact address", + "definition": "Postal address of the contact.", + "node": "contact_point", + "value_type": "structured", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": "Nests the contact_street_address / locality / postal_code concepts." + }, + "contact_street_address": { + "label": "Contact street address", + "definition": "Street address of the contact.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_locality": { + "label": "Contact locality", + "definition": "Town or city of the contact's address.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "contact_postal_code": { + "label": "Contact postal code", + "definition": "Postal code of the contact's address.", + "node": "contact_point", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": null, + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "import_only", + "notes": null + }, + "status": { + "label": "Status", + "definition": "Publication workflow state of the dataset on the platform.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "status", + "dataspace_input_type": "automated", + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Croissant has a status concept; the workbook records no DCAT equivalent and marks it not user-facing." + }, + "download_count": { + "label": "Download count", + "definition": "Number of times the dataset has been downloaded.", + "node": "dataset", + "value_type": "number", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "downloadCount", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Usage statistic, not metadata. No DCAT equivalent." + }, + "similar_datasets": { + "label": "Similar datasets", + "definition": "Datasets the platform computes as related.", + "node": "dataset", + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "dataspace_field": "similarDatasets", + "dataspace_input_type": "automated", + "visible_on_dataspace": true, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Platform-level and dynamic; the workbook says it need not be metadata." + }, + "is_individual_dataset": { + "label": "Is individual dataset", + "definition": "Whether the record is a standalone dataset or part of a collection.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "isIndividualDataset", + "dataspace_input_type": "automated", + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "The workbook suspects this links to catalogs. If confirmed, it belongs with in_catalog / in_series rather than here." + }, + "dataspace": { + "label": "Dataspace", + "definition": "Unclear platform field.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "dataspace", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Marked unclear and unmappable in the source workbook." + }, + "prompt_metadata": { + "label": "Prompt metadata", + "definition": "Unclear platform field.", + "node": "dataset", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "dataspace_field": "promptMetadata", + "dataspace_input_type": null, + "visible_on_dataspace": false, + "controlled_vocabulary": null, + "direction": "none", + "notes": "Undefined in the source workbook." + }, + "citation": { + "label": "Citation", + "definition": "How to cite the dataset, as free text (BibTeX or a sentence).", + "dataspace_field": "citation", + "dataspace_input_type": "text", + "value_type": "literal", + "obligation": "optional", + "direction": "both", + "repeatable": false, + "node": "dataset", + "controlled_vocabulary": null, + "visible_on_dataspace": true, + "notes": "Croissant citeAs; Dublin Core / DCAT dcterms:bibliographicCitation." + } + }, + "platform_fields": { + "id": "identifier", + "slug": "slug", + "title": "title", + "description": "description", + "tags": "keyword", + "sectors": "theme", + "datasetType": "dataset_type", + "organization": "publisher", + "user": "creator", + "license": "license", + "accessType": "access_rights", + "created": "created", + "modified": "modified", + "geographies": "spatial_coverage", + "resources": "distribution", + "download_url": "download_url", + "format": "media_type", + "size": "byte_size", + "name": "distribution_title", + "status": "status", + "downloadCount": "download_count", + "similarDatasets": "similar_datasets", + "isIndividualDataset": "is_individual_dataset", + "dataspace": "dataspace", + "promptMetadata": "prompt_metadata", + "issued": "issued", + "source": "source", + "homepage": "homepage", + "language": "language", + "version": "version", + "landing_page": "landing_page", + "in_catalog": "in_catalog", + "access_url": "access_url", + "sha256": "checksum", + "citation": "citation", + "temporal_coverage_start": "temporal_coverage_start", + "temporal_coverage_end": "temporal_coverage_end", + "alternative_title": "alternative_title" + }, + "standards": { + "croissant": { + "name": "Croissant (MLCommons ML-dataset metadata format)", + "version": "1.1", + "url": "https://docs.mlcommons.org/croissant/docs/croissant-spec-1.1.html", + "serialisation": "json-ld", + "uri_style": "string", + "context": { + "@language": "en", + "@vocab": "https://schema.org/", + "arrayShape": "cr:arrayShape", + "citeAs": "cr:citeAs", + "column": "cr:column", + "conformsTo": "dct:conformsTo", + "containedIn": "cr:containedIn", + "cr": "http://mlcommons.org/croissant/", + "data": { + "@id": "cr:data", + "@type": "@json" + }, + "dataBiases": "cr:dataBiases", + "dataCollection": "cr:dataCollection", + "dataType": { + "@id": "cr:dataType", + "@type": "@vocab" + }, + "dct": "http://purl.org/dc/terms/", + "extract": "cr:extract", + "field": "cr:field", + "fileProperty": "cr:fileProperty", + "fileObject": "cr:fileObject", + "fileSet": "cr:fileSet", + "format": "cr:format", + "includes": "cr:includes", + "isArray": "cr:isArray", + "isLiveDataset": "cr:isLiveDataset", + "jsonPath": "cr:jsonPath", + "key": "cr:key", + "md5": "cr:md5", + "parentField": "cr:parentField", + "path": "cr:path", + "personalSensitiveInformation": "cr:personalSensitiveInformation", + "recordSet": "cr:recordSet", + "references": "cr:references", + "regex": "cr:regex", + "repeated": "cr:repeated", + "replace": "cr:replace", + "sc": "https://schema.org/", + "separator": "cr:separator", + "source": "cr:source", + "subField": "cr:subField", + "transform": "cr:transform" + }, + "node_types": { + "dataset": "sc:Dataset", + "distribution": "cr:FileObject", + "record_set": "cr:RecordSet" + }, + "export": [ + { + "concept": "identifier", + "property": "identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "partial", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": "Croissant also uses @id as the node identity; emit both." + }, + { + "concept": "title", + "property": "name", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "alternative_title", + "property": "alternateName", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "alternative_title", + "controlled_vocabulary": null, + "notes": "For imported datasets: the identifier on the source platform." + }, + { + "concept": "description", + "property": "description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "keyword", + "property": "keywords", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "Free text only; no controlled-scheme distinction." + }, + { + "concept": "publisher", + "property": "publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": "The source workbook maps the platform's organization to dct:publisher but to Croissant creator, and user the other way round. Preserved as recorded - see the open question in the README before relying on it." + }, + { + "concept": "creator", + "property": "creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": "See publisher; the inversion is as recorded in the workbook." + }, + { + "concept": "license", + "property": "license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": "A license URL, same steward URI as DCAT." + }, + { + "concept": "issued", + "property": "datePublished", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "modified", + "property": "dateModified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "version", + "property": "version", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "version", + "controlled_vocabulary": null, + "notes": "Croissant has a dedicated version field, which is the one place it is stricter than DCAT." + }, + { + "concept": "spatial_coverage", + "property": "spatialCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": "schema.org Place, not a dcterms:Location. Accepts a name string, so an export can succeed while carrying no resolvable URI - check the value." + }, + { + "concept": "temporal_coverage_start", + "property": "temporalCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "temporal_coverage_start", + "controlled_vocabulary": null, + "notes": "One ISO 8601 interval string (\"2019-01-01/2023-12-31\"), not two fields. Both start and end serialise into it." + }, + { + "concept": "temporal_coverage_end", + "property": "temporalCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "temporal_coverage_end", + "controlled_vocabulary": null, + "notes": "See temporal_coverage_start; both concepts share one property." + }, + { + "concept": "landing_page", + "property": "url", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "landing_page", + "controlled_vocabulary": null, + "notes": "Requires interpretation of intent - url is not specifically a landing page." + }, + { + "concept": "distribution", + "property": "distribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "resources", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "download_url", + "property": "contentUrl", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "media_type", + "property": "encodingFormat", + "node": "distribution", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "byte_size", + "property": "contentSize", + "node": "distribution", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "checksum", + "property": "sha256", + "node": "distribution", + "parent_property": null, + "value_type": "checksum", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "sha256", + "controlled_vocabulary": null, + "notes": "Required on a FileObject. DCAT has nothing equivalent." + }, + { + "concept": "distribution_title", + "property": "name", + "node": "distribution", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "name", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "language", + "property": "inLanguage", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "language", + "controlled_vocabulary": "language", + "notes": null + }, + { + "concept": "created", + "property": "dateCreated", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "created", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "source", + "property": "sameAs", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "source", + "controlled_vocabulary": null, + "notes": "For imported datasets: the same dataset on the source platform." + }, + { + "concept": "citation", + "property": "citeAs", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "citation", + "controlled_vocabulary": null, + "notes": null + } + ], + "import": { + "dataset": { + "identifier": [ + { + "concept": "identifier", + "property": "identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "partial", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": "Croissant also uses @id as the node identity; emit both." + } + ], + "name": [ + { + "concept": "title", + "property": "name", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + } + ], + "alternateName": [ + { + "concept": "alternative_title", + "property": "alternateName", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "description": [ + { + "concept": "description", + "property": "description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + } + ], + "keywords": [ + { + "concept": "keyword", + "property": "keywords", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "Free text only; no controlled-scheme distinction." + } + ], + "inLanguage": [ + { + "concept": "language", + "property": "inLanguage", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "creator": [ + { + "concept": "publisher", + "property": "creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": "The source workbook maps the platform's organization to dct:publisher but to Croissant creator, and user the other way round. Preserved as recorded - see the open question in the README before relying on it." + } + ], + "publisher": [ + { + "concept": "creator", + "property": "publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": "See publisher; the inversion is as recorded in the workbook." + } + ], + "license": [ + { + "concept": "license", + "property": "license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": "A license URL, same steward URI as DCAT." + } + ], + "datePublished": [ + { + "concept": "issued", + "property": "datePublished", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + } + ], + "dateCreated": [ + { + "concept": "created", + "property": "dateCreated", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dateModified": [ + { + "concept": "modified", + "property": "dateModified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + } + ], + "version": [ + { + "concept": "version", + "property": "version", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Croissant has a dedicated version field, which is the one place it is stricter than DCAT." + } + ], + "spatialCoverage": [ + { + "concept": "spatial_coverage", + "property": "spatialCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "mandatory", + "match": "partial", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": "schema.org Place, not a dcterms:Location. Accepts a name string, so an export can succeed while carrying no resolvable URI - check the value." + } + ], + "temporalCoverage": [ + { + "concept": "temporal_coverage_start", + "property": "temporalCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "One ISO 8601 interval string (\"2019-01-01/2023-12-31\"), not two fields. Both start and end serialise into it." + }, + { + "concept": "temporal_coverage_end", + "property": "temporalCoverage", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "See temporal_coverage_start; both concepts share one property." + } + ], + "url": [ + { + "concept": "landing_page", + "property": "url", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Requires interpretation of intent - url is not specifically a landing page." + }, + { + "concept": "homepage", + "property": "url", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "isBasedOn": [ + { + "concept": "source", + "property": "isBasedOn", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "conformsTo": [ + { + "concept": "conforms_to", + "property": "conformsTo", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "metadata", + "controlled_vocabulary": null, + "notes": "Croissant uses this to declare the Croissant spec version itself." + } + ], + "hasPart": [ + { + "concept": "has_part", + "property": "hasPart", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "isPartOf": [ + { + "concept": "is_part_of", + "property": "isPartOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "distribution": [ + { + "concept": "distribution", + "property": "distribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "resources", + "controlled_vocabulary": null, + "notes": null + } + ], + "cr:FileSet": [ + { + "concept": "file_set", + "property": "cr:FileSet", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "A group of files given by an includes/excludes pattern." + } + ], + "cr:recordSet": [ + { + "concept": "record_set", + "property": "cr:recordSet", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "cr:annotation": [ + { + "concept": "annotation", + "property": "cr:annotation", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "distribution": { + "contentUrl": [ + { + "concept": "download_url", + "property": "contentUrl", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": null + } + ], + "encodingFormat": [ + { + "concept": "media_type", + "property": "encodingFormat", + "node": "distribution", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": null + } + ], + "contentSize": [ + { + "concept": "byte_size", + "property": "contentSize", + "node": "distribution", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": null + } + ], + "sha256": [ + { + "concept": "checksum", + "property": "sha256", + "node": "distribution", + "parent_property": null, + "value_type": "checksum", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Required on a FileObject. DCAT has nothing equivalent." + } + ], + "name": [ + { + "concept": "distribution_title", + "property": "name", + "node": "distribution", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "name", + "controlled_vocabulary": null, + "notes": null + } + ], + "containedIn": [ + { + "concept": "containment", + "property": "containedIn", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "record_set": { + "cr:field": [ + { + "concept": "field", + "property": "cr:field", + "node": "record_set", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dataType": [ + { + "concept": "data_type", + "property": "dataType", + "node": "record_set", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Includes ML types such as cr:BoundingBox." + } + ], + "cr:transform": [ + { + "concept": "transform", + "property": "cr:transform", + "node": "record_set", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + } + }, + "gaps": [ + { + "concept": "source_identifier", + "reason": "unbound", + "notes": null + }, + { + "concept": "slug", + "reason": "unbound", + "notes": null + }, + { + "concept": "abstract", + "reason": "declared gap", + "notes": null + }, + { + "concept": "theme", + "reason": "declared gap", + "notes": "No controlled theme scheme. Sectors have to go out as keywords, which loses the vocabulary binding." + }, + { + "concept": "dataset_type", + "reason": "declared gap", + "notes": null + }, + { + "concept": "qualified_attribution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_point", + "reason": "declared gap", + "notes": null + }, + { + "concept": "rights", + "reason": "declared gap", + "notes": null + }, + { + "concept": "access_rights", + "reason": "declared gap", + "notes": "Not defined in Croissant. The workbook notes dcterms:accessRights exists only on the DCAT side." + }, + { + "concept": "accrual_periodicity", + "reason": "declared gap", + "notes": null + }, + { + "concept": "accrual_method", + "reason": "declared gap", + "notes": null + }, + { + "concept": "spatial_resolution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "temporal_resolution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "depiction", + "reason": "declared gap", + "notes": null + }, + { + "concept": "in_catalog", + "reason": "declared gap", + "notes": null + }, + { + "concept": "in_series", + "reason": "declared gap", + "notes": null + }, + { + "concept": "relation", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_referenced_by", + "reason": "unbound", + "notes": null + }, + { + "concept": "replaces", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_replaced_by", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_version_of", + "reason": "unbound", + "notes": null + }, + { + "concept": "has_version", + "reason": "unbound", + "notes": null + }, + { + "concept": "provenance", + "reason": "declared gap", + "notes": "cr:transform covers field derivation, not dataset custody history." + }, + { + "concept": "jurisdiction_level", + "reason": "declared gap", + "notes": null + }, + { + "concept": "applicable_legislation", + "reason": "declared gap", + "notes": null + }, + { + "concept": "hvd_category", + "reason": "declared gap", + "notes": null + }, + { + "concept": "note", + "reason": "declared gap", + "notes": null + }, + { + "concept": "endpoint_url", + "reason": "declared gap", + "notes": null + }, + { + "concept": "endpoint_description", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_name", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_title", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_role", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_organization", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_email", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_address", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_street_address", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_locality", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_postal_code", + "reason": "declared gap", + "notes": null + }, + { + "concept": "download_count", + "reason": "unbound", + "notes": null + }, + { + "concept": "similar_datasets", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_individual_dataset", + "reason": "unbound", + "notes": null + }, + { + "concept": "dataspace", + "reason": "unbound", + "notes": null + }, + { + "concept": "prompt_metadata", + "reason": "unbound", + "notes": null + }, + { + "concept": "homepage", + "reason": "declared gap", + "notes": "Croissant `url` is the landing page; there is no separate homepage property." + } + ] + }, + "dcat": { + "name": "DCAT v3 (Data Catalog Vocabulary)", + "version": "3", + "url": "https://www.w3.org/TR/vocab-dcat-3/", + "serialisation": "json-ld", + "uri_style": "node", + "context": { + "dcat": "http://www.w3.org/ns/dcat#", + "dcterms": "http://purl.org/dc/terms/", + "foaf": "http://xmlns.com/foaf/0.1/", + "prov": "http://www.w3.org/ns/prov#", + "xsd": "http://www.w3.org/2001/XMLSchema#", + "owl": "http://www.w3.org/2002/07/owl#" + }, + "node_types": { + "dataset": "dcat:Dataset", + "distribution": "dcat:Distribution", + "contact_point": "vcard:Kind" + }, + "export": [ + { + "concept": "identifier", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "title", + "property": "dcterms:title", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "description", + "property": "dcterms:description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "keyword", + "property": "dcat:keyword", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "dcat:keyword or dcterms:subject, depending on whether the values are free text or come from a controlled scheme. The platform's tags are semi-controlled, so keyword is the safer target." + }, + { + "concept": "theme", + "property": "dcat:theme", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "sectors", + "controlled_vocabulary": "sector", + "notes": "Expects a skos:Concept from a published scheme. CDL sectors collapse onto EU data themes many-to-one; the mapping is lossy in that direction and not reversible." + }, + { + "concept": "dataset_type", + "property": "dcterms:type", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "datasetType", + "controlled_vocabulary": null, + "notes": "Dataset-level only. DCAT has no field-level typing." + }, + { + "concept": "publisher", + "property": "dcterms:publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "creator", + "property": "dcterms:creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "license", + "property": "dcterms:license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": "Mandatory at Distribution in the registry profile. Emit the steward URI from the license vocabulary, never the platform's enum label." + }, + { + "concept": "access_rights", + "property": "dcterms:accessRights", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "accessType", + "controlled_vocabulary": null, + "notes": "Croissant has no equivalent; this is DCAT-only." + }, + { + "concept": "issued", + "property": "dcterms:issued", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "modified", + "property": "dcterms:modified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "version", + "property": "owl:versionInfo", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": "version", + "controlled_vocabulary": null, + "notes": "DCAT has no dedicated version field. The workbook records the combination issued + modified + dcterms:hasVersion / owl:versionInfo. owl:versionInfo is used here because dcterms:hasVersion means \"a version of this exists over there\" - binding both concepts to it would make an incoming hasVersion impossible to route." + }, + { + "concept": "spatial_coverage", + "property": "dcterms:spatial", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": null + }, + { + "concept": "temporal_coverage_start", + "property": "dcat:startDate", + "node": "period", + "parent_property": "dcterms:temporal", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "temporal_coverage_start", + "controlled_vocabulary": null, + "notes": "Nested inside a dcterms:PeriodOfTime node." + }, + { + "concept": "temporal_coverage_end", + "property": "dcat:endDate", + "node": "period", + "parent_property": "dcterms:temporal", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "temporal_coverage_end", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "landing_page", + "property": "dcat:landingPage", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "landing_page", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "in_catalog", + "property": "dcat:inCatalog", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "in_catalog", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "distribution", + "property": "dcat:distribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "resources", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "access_url", + "property": "dcat:accessURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "access_url", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "download_url", + "property": "dcat:downloadURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "media_type", + "property": "dcat:mediaType", + "node": "distribution", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": "dcterms:format is the looser alternative the workbook also records. Use mediaType when the value is a real MIME type, format otherwise." + }, + { + "concept": "byte_size", + "property": "dcat:byteSize", + "node": "distribution", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "distribution_title", + "property": "dcterms:title", + "node": "distribution", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "name", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "language", + "property": "dcterms:language", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "language", + "controlled_vocabulary": "language", + "notes": null + }, + { + "concept": "created", + "property": "dcterms:created", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "created", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "homepage", + "property": "foaf:homepage", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "homepage", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "source", + "property": "dcterms:source", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "source", + "controlled_vocabulary": null, + "notes": "For imported datasets: the dataset page on the source platform." + }, + { + "concept": "source", + "property": "prov:wasDerivedFrom", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "source", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "citation", + "property": "dcterms:bibliographicCitation", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "citation", + "controlled_vocabulary": null, + "notes": null + } + ], + "import": { + "dataset": { + "dcterms:identifier": [ + { + "concept": "identifier", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:title": [ + { + "concept": "title", + "property": "dcterms:title", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:alternative": [ + { + "concept": "alternative_title", + "property": "dcterms:alternative", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:description": [ + { + "concept": "description", + "property": "dcterms:description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:abstract": [ + { + "concept": "abstract", + "property": "dcterms:abstract", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:keyword": [ + { + "concept": "keyword", + "property": "dcat:keyword", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "dcat:keyword or dcterms:subject, depending on whether the values are free text or come from a controlled scheme. The platform's tags are semi-controlled, so keyword is the safer target." + } + ], + "dcat:theme": [ + { + "concept": "theme", + "property": "dcat:theme", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "sectors", + "controlled_vocabulary": "sector", + "notes": "Expects a skos:Concept from a published scheme. CDL sectors collapse onto EU data themes many-to-one; the mapping is lossy in that direction and not reversible." + } + ], + "dcterms:type": [ + { + "concept": "dataset_type", + "property": "dcterms:type", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "datasetType", + "controlled_vocabulary": null, + "notes": "Dataset-level only. DCAT has no field-level typing." + } + ], + "dcterms:language": [ + { + "concept": "language", + "property": "dcterms:language", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:publisher": [ + { + "concept": "publisher", + "property": "dcterms:publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:creator": [ + { + "concept": "creator", + "property": "dcterms:creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": null + } + ], + "prov:qualifiedAttribution": [ + { + "concept": "qualified_attribution", + "property": "prov:qualifiedAttribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Requires PROV-O, which is outside DCAT proper." + } + ], + "dcat:contactPoint": [ + { + "concept": "contact_point", + "property": "dcat:contactPoint", + "node": "dataset", + "parent_property": null, + "value_type": "vcard", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Nests a vcard:Kind node built from the contact_* concepts." + } + ], + "dcterms:license": [ + { + "concept": "license", + "property": "dcterms:license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": "Mandatory at Distribution in the registry profile. Emit the steward URI from the license vocabulary, never the platform's enum label." + } + ], + "dcterms:rights": [ + { + "concept": "rights", + "property": "dcterms:rights", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accessRights": [ + { + "concept": "access_rights", + "property": "dcterms:accessRights", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "accessType", + "controlled_vocabulary": null, + "notes": "Croissant has no equivalent; this is DCAT-only." + } + ], + "dcterms:issued": [ + { + "concept": "issued", + "property": "dcterms:issued", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:created": [ + { + "concept": "created", + "property": "dcterms:created", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:modified": [ + { + "concept": "modified", + "property": "dcterms:modified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accrualPeriodicity": [ + { + "concept": "accrual_periodicity", + "property": "dcterms:accrualPeriodicity", + "node": "dataset", + "parent_property": null, + "value_type": "frequency", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accrualMethod": [ + { + "concept": "accrual_method", + "property": "dcterms:accrualMethod", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "owl:versionInfo": [ + { + "concept": "version", + "property": "owl:versionInfo", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT has no dedicated version field. The workbook records the combination issued + modified + dcterms:hasVersion / owl:versionInfo. owl:versionInfo is used here because dcterms:hasVersion means \"a version of this exists over there\" - binding both concepts to it would make an incoming hasVersion impossible to route." + } + ], + "dcterms:spatial": [ + { + "concept": "spatial_coverage", + "property": "dcterms:spatial", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": null + } + ], + "dcat:spatialResolutionInMeters": [ + { + "concept": "spatial_resolution", + "property": "dcat:spatialResolutionInMeters", + "node": "dataset", + "parent_property": null, + "value_type": "number", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT wants metres. The registry records admin level instead, which does not convert cleanly — carry it as a literal or drop it." + } + ], + "dcat:temporalResolution": [ + { + "concept": "temporal_resolution", + "property": "dcat:temporalResolution", + "node": "dataset", + "parent_property": null, + "value_type": "duration", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:landingPage": [ + { + "concept": "landing_page", + "property": "dcat:landingPage", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "foaf:homepage": [ + { + "concept": "homepage", + "property": "foaf:homepage", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:source": [ + { + "concept": "source", + "property": "dcterms:source", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "foaf:depiction": [ + { + "concept": "depiction", + "property": "foaf:depiction", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:inCatalog": [ + { + "concept": "in_catalog", + "property": "dcat:inCatalog", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:inSeries": [ + { + "concept": "in_series", + "property": "dcat:inSeries", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:conformsTo": [ + { + "concept": "conforms_to", + "property": "dcterms:conformsTo", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": "metadata", + "controlled_vocabulary": null, + "notes": "DCAT's only way to point at an external schema such as CSVW." + } + ], + "dcterms:relation": [ + { + "concept": "relation", + "property": "dcterms:relation", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:hasPart": [ + { + "concept": "has_part", + "property": "dcterms:hasPart", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isPartOf": [ + { + "concept": "is_part_of", + "property": "dcterms:isPartOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isReferencedBy": [ + { + "concept": "is_referenced_by", + "property": "dcterms:isReferencedBy", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:replaces": [ + { + "concept": "replaces", + "property": "dcterms:replaces", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isReplacedBy": [ + { + "concept": "is_replaced_by", + "property": "dcterms:isReplacedBy", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isVersionOf": [ + { + "concept": "is_version_of", + "property": "dcterms:isVersionOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:hasVersion": [ + { + "concept": "has_version", + "property": "dcterms:hasVersion", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "prov:wasGeneratedBy": [ + { + "concept": "provenance", + "property": "prov:wasGeneratedBy", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Requires PROV-O." + } + ], + "dcatin:jurisdictionLevel": [ + { + "concept": "jurisdiction_level", + "property": "dcatin:jurisdictionLevel", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT-IN, proposed. Namespace is a placeholder until it publishes." + } + ], + "dcatin:applicableLegislation": [ + { + "concept": "applicable_legislation", + "property": "dcatin:applicableLegislation", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT-IN, proposed." + } + ], + "dcatin:hvdCategory": [ + { + "concept": "hvd_category", + "property": "dcatin:hvdCategory", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT-IN, proposed." + } + ], + "dcatin:note": [ + { + "concept": "note", + "property": "dcatin:note", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "DCAT-IN, proposed." + } + ], + "dcat:distribution": [ + { + "concept": "distribution", + "property": "dcat:distribution", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "resources", + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "period": { + "dcat:startDate": [ + { + "concept": "temporal_coverage_start", + "property": "dcat:startDate", + "node": "period", + "parent_property": "dcterms:temporal", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Nested inside a dcterms:PeriodOfTime node." + } + ], + "dcat:endDate": [ + { + "concept": "temporal_coverage_end", + "property": "dcat:endDate", + "node": "period", + "parent_property": "dcterms:temporal", + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "distribution": { + "dcat:accessURL": [ + { + "concept": "access_url", + "property": "dcat:accessURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:downloadURL": [ + { + "concept": "download_url", + "property": "dcat:downloadURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:mediaType": [ + { + "concept": "media_type", + "property": "dcat:mediaType", + "node": "distribution", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": "dcterms:format is the looser alternative the workbook also records. Use mediaType when the value is a real MIME type, format otherwise." + } + ], + "dcat:byteSize": [ + { + "concept": "byte_size", + "property": "dcat:byteSize", + "node": "distribution", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:title": [ + { + "concept": "distribution_title", + "property": "dcterms:title", + "node": "distribution", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "name", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:endpointURL": [ + { + "concept": "endpoint_url", + "property": "dcat:endpointURL", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcat:endpointDescription": [ + { + "concept": "endpoint_description", + "property": "dcat:endpointDescription", + "node": "distribution", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + }, + "contact_point": { + "vcard:fn": [ + { + "concept": "contact_name", + "property": "vcard:fn", + "node": "contact_point", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:title": [ + { + "concept": "contact_title", + "property": "vcard:title", + "node": "contact_point", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:role": [ + { + "concept": "contact_role", + "property": "vcard:role", + "node": "contact_point", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:organization-name": [ + { + "concept": "contact_organization", + "property": "vcard:organization-name", + "node": "contact_point", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:hasEmail": [ + { + "concept": "contact_email", + "property": "vcard:hasEmail", + "node": "contact_point", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:hasAddress": [ + { + "concept": "contact_address", + "property": "vcard:hasAddress", + "node": "contact_point", + "parent_property": null, + "value_type": "structured", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:street-address": [ + { + "concept": "contact_street_address", + "property": "vcard:street-address", + "node": "contact_point", + "parent_property": "vcard:hasAddress", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:locality": [ + { + "concept": "contact_locality", + "property": "vcard:locality", + "node": "contact_point", + "parent_property": "vcard:hasAddress", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "vcard:postal-code": [ + { + "concept": "contact_postal_code", + "property": "vcard:postal-code", + "node": "contact_point", + "parent_property": "vcard:hasAddress", + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ] + } + }, + "gaps": [ + { + "concept": "source_identifier", + "reason": "unbound", + "notes": null + }, + { + "concept": "slug", + "reason": "unbound", + "notes": null + }, + { + "concept": "checksum", + "reason": "declared gap", + "notes": "No integrity-verification property in DCAT. Croissant requires sha256." + }, + { + "concept": "file_set", + "reason": "declared gap", + "notes": "No file grouping or pattern support; nearest is many Distributions." + }, + { + "concept": "containment", + "reason": "declared gap", + "notes": "No file hierarchy modelling." + }, + { + "concept": "record_set", + "reason": "declared gap", + "notes": "Cannot represent tables or rows. Use conforms_to to point outward." + }, + { + "concept": "field", + "reason": "declared gap", + "notes": "No column-level metadata." + }, + { + "concept": "data_type", + "reason": "declared gap", + "notes": "No semantic or ML typing." + }, + { + "concept": "transform", + "reason": "declared gap", + "notes": "Possible only via external PROV-O." + }, + { + "concept": "annotation", + "reason": "declared gap", + "notes": "No native annotation model." + }, + { + "concept": "status", + "reason": "declared gap", + "notes": "Publication workflow state is not a DCAT concept." + }, + { + "concept": "download_count", + "reason": "unbound", + "notes": null + }, + { + "concept": "similar_datasets", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_individual_dataset", + "reason": "unbound", + "notes": null + }, + { + "concept": "dataspace", + "reason": "unbound", + "notes": null + }, + { + "concept": "prompt_metadata", + "reason": "unbound", + "notes": null + } + ] + }, + "dublin_core": { + "name": "Dublin Core Metadata Terms (DCMI)", + "version": "DCMI Metadata Terms 2020-01-20", + "url": "https://www.dublincore.org/specifications/dublin-core/dcmi-terms/", + "serialisation": "json-ld", + "uri_style": "node", + "context": { + "dcterms": "http://purl.org/dc/terms/" + }, + "node_types": { + "dataset": "dcterms:Dataset" + }, + "export": [ + { + "concept": "identifier", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "title", + "property": "dcterms:title", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "description", + "property": "dcterms:description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "keyword", + "property": "dcterms:subject", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "Dublin Core has no free-text keyword element. subject is the nearest and is meant to carry controlled terms, so round-tripping tags through it loses the distinction between a tag and a theme." + }, + { + "concept": "theme", + "property": "dcterms:subject", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "sectors", + "controlled_vocabulary": "sector", + "notes": "Shares dcterms:subject with keyword. On import from Dublin Core there is no way to tell which concept a subject value belongs to; the workbook's DCAT sheet maps Subject to both dcat:theme and dcat:keyword for the same reason. Prefer DCAT when the distinction matters." + }, + { + "concept": "dataset_type", + "property": "dcterms:type", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "datasetType", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "publisher", + "property": "dcterms:publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "creator", + "property": "dcterms:creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "license", + "property": "dcterms:license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": null + }, + { + "concept": "access_rights", + "property": "dcterms:accessRights", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "accessType", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "issued", + "property": "dcterms:issued", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "modified", + "property": "dcterms:modified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "spatial_coverage", + "property": "dcterms:spatial", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": null + }, + { + "concept": "temporal_coverage_start", + "property": "dcterms:temporal", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "temporal_coverage_start", + "controlled_vocabulary": null, + "notes": "Dublin Core has no start/end structure - temporal takes a single literal period. Serialise start and end as one DCMI Period string; parsing it back into two dates is best-effort." + }, + { + "concept": "temporal_coverage_end", + "property": "dcterms:temporal", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "temporal_coverage_end", + "controlled_vocabulary": null, + "notes": "See temporal_coverage_start; both concepts share one property." + }, + { + "concept": "download_url", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": "Dublin Core has no download link. The workbook's Identifier row maps to dcat:landingPage and foaf:homePage for this reason. Lossy either way." + }, + { + "concept": "media_type", + "property": "dcterms:format", + "node": "dataset", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": "Flattened onto the dataset, because Dublin Core has no Distribution. For a multi-file dataset this collapses to the set of its formats." + }, + { + "concept": "byte_size", + "property": "dcterms:extent", + "node": "dataset", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": "Flattened onto the dataset; a literal, not a byte count." + }, + { + "concept": "language", + "property": "dcterms:language", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "language", + "controlled_vocabulary": "language", + "notes": null + }, + { + "concept": "created", + "property": "dcterms:created", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "created", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "source", + "property": "dcterms:source", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "source", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "citation", + "property": "dcterms:bibliographicCitation", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": "citation", + "controlled_vocabulary": null, + "notes": null + } + ], + "import": { + "dataset": { + "dcterms:identifier": [ + { + "concept": "identifier", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "id", + "controlled_vocabulary": null, + "notes": null + }, + { + "concept": "download_url", + "property": "dcterms:identifier", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "download_url", + "controlled_vocabulary": null, + "notes": "Dublin Core has no download link. The workbook's Identifier row maps to dcat:landingPage and foaf:homePage for this reason. Lossy either way." + } + ], + "dcterms:title": [ + { + "concept": "title", + "property": "dcterms:title", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "title", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:alternative": [ + { + "concept": "alternative_title", + "property": "dcterms:alternative", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:description": [ + { + "concept": "description", + "property": "dcterms:description", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "description", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:abstract": [ + { + "concept": "abstract", + "property": "dcterms:abstract", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:subject": [ + { + "concept": "keyword", + "property": "dcterms:subject", + "node": "dataset", + "parent_property": null, + "value_type": "literal", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "tags", + "controlled_vocabulary": null, + "notes": "Dublin Core has no free-text keyword element. subject is the nearest and is meant to carry controlled terms, so round-tripping tags through it loses the distinction between a tag and a theme." + }, + { + "concept": "theme", + "property": "dcterms:subject", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "sectors", + "controlled_vocabulary": "sector", + "notes": "Shares dcterms:subject with keyword. On import from Dublin Core there is no way to tell which concept a subject value belongs to; the workbook's DCAT sheet maps Subject to both dcat:theme and dcat:keyword for the same reason. Prefer DCAT when the distinction matters." + } + ], + "dcterms:type": [ + { + "concept": "dataset_type", + "property": "dcterms:type", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "datasetType", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:language": [ + { + "concept": "language", + "property": "dcterms:language", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": true, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:publisher": [ + { + "concept": "publisher", + "property": "dcterms:publisher", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "organization", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:creator": [ + { + "concept": "creator", + "property": "dcterms:creator", + "node": "dataset", + "parent_property": null, + "value_type": "agent", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "user", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:license": [ + { + "concept": "license", + "property": "dcterms:license", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "license", + "controlled_vocabulary": "license", + "notes": null + } + ], + "dcterms:rights": [ + { + "concept": "rights", + "property": "dcterms:rights", + "node": "dataset", + "parent_property": null, + "value_type": "langstring", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accessRights": [ + { + "concept": "access_rights", + "property": "dcterms:accessRights", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": "accessType", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:issued": [ + { + "concept": "issued", + "property": "dcterms:issued", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "issued", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:created": [ + { + "concept": "created", + "property": "dcterms:created", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:modified": [ + { + "concept": "modified", + "property": "dcterms:modified", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "modified", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accrualPeriodicity": [ + { + "concept": "accrual_periodicity", + "property": "dcterms:accrualPeriodicity", + "node": "dataset", + "parent_property": null, + "value_type": "frequency", + "repeatable": false, + "obligation": "recommended", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:accrualMethod": [ + { + "concept": "accrual_method", + "property": "dcterms:accrualMethod", + "node": "dataset", + "parent_property": null, + "value_type": "concept", + "repeatable": false, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:spatial": [ + { + "concept": "spatial_coverage", + "property": "dcterms:spatial", + "node": "dataset", + "parent_property": null, + "value_type": "location", + "repeatable": true, + "obligation": "mandatory", + "match": "exact", + "dataspace_field": "geographies", + "controlled_vocabulary": "geography", + "notes": null + } + ], + "dcterms:temporal": [ + { + "concept": "temporal_coverage_start", + "property": "dcterms:temporal", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "Dublin Core has no start/end structure - temporal takes a single literal period. Serialise start and end as one DCMI Period string; parsing it back into two dates is best-effort." + }, + { + "concept": "temporal_coverage_end", + "property": "dcterms:temporal", + "node": "dataset", + "parent_property": null, + "value_type": "date", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "See temporal_coverage_start; both concepts share one property." + } + ], + "dcterms:source": [ + { + "concept": "source", + "property": "dcterms:source", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:conformsTo": [ + { + "concept": "conforms_to", + "property": "dcterms:conformsTo", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": "metadata", + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:relation": [ + { + "concept": "relation", + "property": "dcterms:relation", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:hasPart": [ + { + "concept": "has_part", + "property": "dcterms:hasPart", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isPartOf": [ + { + "concept": "is_part_of", + "property": "dcterms:isPartOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isReferencedBy": [ + { + "concept": "is_referenced_by", + "property": "dcterms:isReferencedBy", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:replaces": [ + { + "concept": "replaces", + "property": "dcterms:replaces", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isReplacedBy": [ + { + "concept": "is_replaced_by", + "property": "dcterms:isReplacedBy", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:isVersionOf": [ + { + "concept": "is_version_of", + "property": "dcterms:isVersionOf", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:hasVersion": [ + { + "concept": "has_version", + "property": "dcterms:hasVersion", + "node": "dataset", + "parent_property": null, + "value_type": "uri", + "repeatable": true, + "obligation": "optional", + "match": "exact", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": null + } + ], + "dcterms:provenance": [ + { + "concept": "provenance", + "property": "dcterms:provenance", + "node": "dataset", + "parent_property": null, + "value_type": "structured", + "repeatable": true, + "obligation": "optional", + "match": "partial", + "dataspace_field": null, + "controlled_vocabulary": null, + "notes": "A free-text statement, not the structured PROV-O graph." + } + ], + "dcterms:format": [ + { + "concept": "media_type", + "property": "dcterms:format", + "node": "dataset", + "parent_property": null, + "value_type": "media_type", + "repeatable": false, + "obligation": "recommended", + "match": "partial", + "dataspace_field": "format", + "controlled_vocabulary": null, + "notes": "Flattened onto the dataset, because Dublin Core has no Distribution. For a multi-file dataset this collapses to the set of its formats." + } + ], + "dcterms:extent": [ + { + "concept": "byte_size", + "property": "dcterms:extent", + "node": "dataset", + "parent_property": null, + "value_type": "bytes", + "repeatable": false, + "obligation": "optional", + "match": "partial", + "dataspace_field": "size", + "controlled_vocabulary": null, + "notes": "Flattened onto the dataset; a literal, not a byte count." + } + ] + } + }, + "gaps": [ + { + "concept": "source_identifier", + "reason": "unbound", + "notes": null + }, + { + "concept": "slug", + "reason": "unbound", + "notes": null + }, + { + "concept": "qualified_attribution", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_point", + "reason": "declared gap", + "notes": "No contact structure; publisher is the only agent-ish element." + }, + { + "concept": "version", + "reason": "declared gap", + "notes": "No version element. dcterms:hasVersion points at a different resource that is a version of this one, which is has_version, not this concept." + }, + { + "concept": "spatial_resolution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "temporal_resolution", + "reason": "declared gap", + "notes": null + }, + { + "concept": "landing_page", + "reason": "declared gap", + "notes": null + }, + { + "concept": "homepage", + "reason": "unbound", + "notes": null + }, + { + "concept": "depiction", + "reason": "unbound", + "notes": null + }, + { + "concept": "in_catalog", + "reason": "unbound", + "notes": null + }, + { + "concept": "in_series", + "reason": "unbound", + "notes": null + }, + { + "concept": "jurisdiction_level", + "reason": "unbound", + "notes": null + }, + { + "concept": "applicable_legislation", + "reason": "unbound", + "notes": null + }, + { + "concept": "hvd_category", + "reason": "unbound", + "notes": null + }, + { + "concept": "note", + "reason": "unbound", + "notes": null + }, + { + "concept": "distribution", + "reason": "declared gap", + "notes": "No Distribution resource. Every file-level fact flattens or is lost." + }, + { + "concept": "access_url", + "reason": "declared gap", + "notes": null + }, + { + "concept": "checksum", + "reason": "declared gap", + "notes": null + }, + { + "concept": "distribution_title", + "reason": "unbound", + "notes": null + }, + { + "concept": "endpoint_url", + "reason": "unbound", + "notes": null + }, + { + "concept": "endpoint_description", + "reason": "unbound", + "notes": null + }, + { + "concept": "file_set", + "reason": "declared gap", + "notes": null + }, + { + "concept": "containment", + "reason": "declared gap", + "notes": null + }, + { + "concept": "record_set", + "reason": "declared gap", + "notes": null + }, + { + "concept": "field", + "reason": "declared gap", + "notes": null + }, + { + "concept": "data_type", + "reason": "declared gap", + "notes": null + }, + { + "concept": "transform", + "reason": "declared gap", + "notes": null + }, + { + "concept": "annotation", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_name", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_title", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_role", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_organization", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_email", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_address", + "reason": "unbound", + "notes": null + }, + { + "concept": "contact_street_address", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_locality", + "reason": "declared gap", + "notes": null + }, + { + "concept": "contact_postal_code", + "reason": "declared gap", + "notes": null + }, + { + "concept": "status", + "reason": "declared gap", + "notes": null + }, + { + "concept": "download_count", + "reason": "unbound", + "notes": null + }, + { + "concept": "similar_datasets", + "reason": "unbound", + "notes": null + }, + { + "concept": "is_individual_dataset", + "reason": "unbound", + "notes": null + }, + { + "concept": "dataspace", + "reason": "unbound", + "notes": null + }, + { + "concept": "prompt_metadata", + "reason": "unbound", + "notes": null + }, + { + "concept": "homepage", + "reason": "declared gap", + "notes": "No homepage term in DCMI Metadata Terms." + } + ] + } + }, + "owner": "DataSpaceBackend (api/services/metadata_export/contracts)", + "origin": "Started as a copy of CivicDataLab/DataSpace-data-ecosystem @ 83e577752784 (data_model/metadata/metadata-standards/superset/out/crosswalk.json). Edited here since; this file is the source of truth." +} diff --git a/api/services/metadata_export/contracts/geographies.csv b/api/services/metadata_export/contracts/geographies.csv new file mode 100644 index 0000000..0dfe748 --- /dev/null +++ b/api/services/metadata_export/contracts/geographies.csv @@ -0,0 +1,828 @@ +key,label,code,uri,tier,parent_key,visible_on_dataspace,is_active +region:north-east-india,North East India,north-east-india,https://civicdataspace.in/id/geography/region/north-east-india,REGION,,yes,True +region:southern-asia,Southern Asia,southern-asia,https://civicdataspace.in/id/geography/region/southern-asia,REGION,,yes,True +country:IN,India,IN,https://www.wikidata.org/entity/Q668,COUNTRY,,yes,True +state:1,Jammu and Kashmir,1,http://www.wikidata.org/entity/Q66278313,UT,country:IN,yes,True +state:10,Bihar,10,http://www.wikidata.org/entity/Q1165,STATE,country:IN,yes,True +state:11,Sikkim,11,http://www.wikidata.org/entity/Q1505,STATE,country:IN,yes,True +state:12,Arunachal Pradesh,12,http://www.wikidata.org/entity/Q1162,STATE,country:IN,yes,True +state:13,Nagaland,13,http://www.wikidata.org/entity/Q1599,STATE,country:IN,yes,True +state:14,Manipur,14,http://www.wikidata.org/entity/Q1193,STATE,country:IN,yes,True +state:15,Mizoram,15,http://www.wikidata.org/entity/Q1502,STATE,country:IN,yes,True +state:16,Tripura,16,http://www.wikidata.org/entity/Q1363,STATE,country:IN,yes,True +state:17,Meghalaya,17,http://www.wikidata.org/entity/Q1195,STATE,country:IN,yes,True +state:18,Assam,18,http://www.wikidata.org/entity/Q1164,STATE,country:IN,yes,True +state:19,West Bengal,19,http://www.wikidata.org/entity/Q1356,STATE,country:IN,yes,True +state:2,Himachal Pradesh,2,http://www.wikidata.org/entity/Q1177,STATE,country:IN,yes,True +state:20,Jharkhand,20,http://www.wikidata.org/entity/Q1184,STATE,country:IN,yes,True +state:21,Odisha,21,http://www.wikidata.org/entity/Q22048,STATE,country:IN,yes,True +state:22,Chhattisgarh,22,http://www.wikidata.org/entity/Q1168,STATE,country:IN,yes,True +state:23,Madhya Pradesh,23,http://www.wikidata.org/entity/Q1188,STATE,country:IN,yes,True +state:24,Gujarat,24,http://www.wikidata.org/entity/Q1061,STATE,country:IN,yes,True +state:27,Maharashtra,27,http://www.wikidata.org/entity/Q1191,STATE,country:IN,yes,True +state:28,Andhra Pradesh,28,http://www.wikidata.org/entity/Q1159,STATE,country:IN,yes,True +state:29,Karnataka,29,http://www.wikidata.org/entity/Q1185,STATE,country:IN,yes,True +state:3,Punjab,3,http://www.wikidata.org/entity/Q22424,STATE,country:IN,yes,True +state:30,Goa,30,http://www.wikidata.org/entity/Q1171,STATE,country:IN,yes,True +state:31,Lakshadweep,31,http://www.wikidata.org/entity/Q26927,UT,country:IN,yes,True +state:32,Kerala,32,http://www.wikidata.org/entity/Q1186,STATE,country:IN,yes,True +state:33,Tamil Nadu,33,http://www.wikidata.org/entity/Q1445,STATE,country:IN,yes,True +state:34,Puducherry,34,http://www.wikidata.org/entity/Q66743,UT,country:IN,yes,True +state:35,Andaman and Nicobar Islands,35,http://www.wikidata.org/entity/Q40888,UT,country:IN,yes,True +state:36,Telangana,36,http://www.wikidata.org/entity/Q677037,STATE,country:IN,yes,True +state:37,Ladakh,37,http://www.wikidata.org/entity/Q200667,UT,country:IN,yes,True +state:38,Dadra and Nagar Haveli and Daman and Diu,38,http://www.wikidata.org/entity/Q77997266,UT,country:IN,yes,True +state:4,Chandigarh,4,http://www.wikidata.org/entity/Q120971341,UT,country:IN,yes,True +state:5,Uttarakhand,5,http://www.wikidata.org/entity/Q1499,STATE,country:IN,yes,True +state:6,Haryana,6,http://www.wikidata.org/entity/Q1174,STATE,country:IN,yes,True +state:7,National Capital Territory of Delhi,7,http://www.wikidata.org/entity/Q9357528,UT,country:IN,yes,True +state:8,Rajasthan,8,http://www.wikidata.org/entity/Q1437,STATE,country:IN,yes,True +state:9,Uttar Pradesh,9,http://www.wikidata.org/entity/Q1498,STATE,country:IN,yes,True +district:1,Anantnag,1,http://www.wikidata.org/entity/Q2982349,DISTRICT,state:1,no,True +district:10,Poonch,10,http://www.wikidata.org/entity/Q2983134,DISTRICT,state:1,no,True +district:11,Pulwama,11,http://www.wikidata.org/entity/Q2085364,DISTRICT,state:1,no,True +district:12,Rajouri,12,http://www.wikidata.org/entity/Q544279,DISTRICT,state:1,no,True +district:13,Srinagar,13,http://www.wikidata.org/entity/Q1506029,DISTRICT,state:1,no,True +district:14,Udhampur,14,http://www.wikidata.org/entity/Q1947311,DISTRICT,state:1,no,True +district:2,Budgam,2,http://www.wikidata.org/entity/Q2594218,DISTRICT,state:1,no,True +district:3,Baramulla,3,http://www.wikidata.org/entity/Q1912057,DISTRICT,state:1,no,True +district:4,Doda,4,http://www.wikidata.org/entity/Q2298979,DISTRICT,state:1,no,True +district:5,Jammu,5,http://www.wikidata.org/entity/Q1947371,DISTRICT,state:1,no,True +district:620,Kishtwar,620,http://www.wikidata.org/entity/Q2321899,DISTRICT,state:1,no,True +district:621,Ramban,621,http://www.wikidata.org/entity/Q2321939,DISTRICT,state:1,no,True +district:622,Kulgam,622,http://www.wikidata.org/entity/Q2321867,DISTRICT,state:1,no,True +district:623,Bandipore,623,http://www.wikidata.org/entity/Q2983553,DISTRICT,state:1,no,True +district:624,Samba,624,http://www.wikidata.org/entity/Q1117086,DISTRICT,state:1,no,True +district:625,Shopian,625,http://www.wikidata.org/entity/Q2073646,DISTRICT,state:1,no,True +district:626,Ganderbal,626,http://www.wikidata.org/entity/Q2556028,DISTRICT,state:1,no,True +district:627,Reasi,627,http://www.wikidata.org/entity/Q2321956,DISTRICT,state:1,no,True +district:7,Kathua,7,http://www.wikidata.org/entity/Q2375700,DISTRICT,state:1,no,True +district:8,Kupwara,8,http://www.wikidata.org/entity/Q2297306,DISTRICT,state:1,no,True +district:188,Araria,188,http://www.wikidata.org/entity/Q42901,DISTRICT,state:10,no,True +district:189,Aurangabad,189,http://www.wikidata.org/entity/Q43086,DISTRICT,state:10,no,True +district:190,Banka,190,http://www.wikidata.org/entity/Q43097,DISTRICT,state:10,no,True +district:191,Begusarai,191,http://www.wikidata.org/entity/Q49157,DISTRICT,state:10,no,True +district:192,Bhagalpur,192,http://www.wikidata.org/entity/Q49155,DISTRICT,state:10,no,True +district:193,Bhojpur,193,http://www.wikidata.org/entity/Q49153,DISTRICT,state:10,no,True +district:194,Buxar,194,http://www.wikidata.org/entity/Q49161,DISTRICT,state:10,no,True +district:195,Darbhanga,195,http://www.wikidata.org/entity/Q49160,DISTRICT,state:10,no,True +district:196,Gaya,196,http://www.wikidata.org/entity/Q49173,DISTRICT,state:10,no,True +district:197,Gopalganj,197,http://www.wikidata.org/entity/Q49171,DISTRICT,state:10,no,True +district:198,Jamui,198,http://www.wikidata.org/entity/Q49168,DISTRICT,state:10,no,True +district:199,Jehanabad,199,http://www.wikidata.org/entity/Q49176,DISTRICT,state:10,no,True +district:200,Kaimur,200,http://www.wikidata.org/entity/Q77367,DISTRICT,state:10,no,True +district:201,Katihar,201,http://www.wikidata.org/entity/Q77568,DISTRICT,state:10,no,True +district:202,Khagaria,202,http://www.wikidata.org/entity/Q49175,DISTRICT,state:10,no,True +district:203,Kishanganj,203,http://www.wikidata.org/entity/Q77375,DISTRICT,state:10,no,True +district:204,Lakhisarai,204,http://www.wikidata.org/entity/Q77505,DISTRICT,state:10,no,True +district:205,Madhepura,205,http://www.wikidata.org/entity/Q77746,DISTRICT,state:10,no,True +district:206,Madhubani,206,http://www.wikidata.org/entity/Q77474,DISTRICT,state:10,no,True +district:207,Munger,207,http://www.wikidata.org/entity/Q77452,DISTRICT,state:10,no,True +district:208,Muzaffarpur,208,http://www.wikidata.org/entity/Q77731,DISTRICT,state:10,no,True +district:209,Nalanda,209,http://www.wikidata.org/entity/Q77633,DISTRICT,state:10,no,True +district:210,Nawada,210,http://www.wikidata.org/entity/Q100067,DISTRICT,state:10,no,True +district:211,West Champaran,211,http://www.wikidata.org/entity/Q100124,DISTRICT,state:10,no,True +district:212,Patna,212,http://www.wikidata.org/entity/Q100077,DISTRICT,state:10,no,True +district:213,East Champaran,213,http://www.wikidata.org/entity/Q49159,DISTRICT,state:10,no,True +district:214,Purnia,214,http://www.wikidata.org/entity/Q100082,DISTRICT,state:10,no,True +district:215,Rohtas,215,http://www.wikidata.org/entity/Q100085,DISTRICT,state:10,no,True +district:216,Saharsa,216,http://www.wikidata.org/entity/Q100120,DISTRICT,state:10,no,True +district:217,Samastipur,217,http://www.wikidata.org/entity/Q100117,DISTRICT,state:10,no,True +district:218,Saran,218,http://www.wikidata.org/entity/Q100146,DISTRICT,state:10,no,True +district:219,Sheikhpura,219,http://www.wikidata.org/entity/Q100093,DISTRICT,state:10,no,True +district:220,Sheohar,220,http://www.wikidata.org/entity/Q100095,DISTRICT,state:10,no,True +district:221,Sitamarhi,221,http://www.wikidata.org/entity/Q100144,DISTRICT,state:10,no,True +district:222,Siwan,222,http://www.wikidata.org/entity/Q100131,DISTRICT,state:10,no,True +district:223,Supaul,223,http://www.wikidata.org/entity/Q100139,DISTRICT,state:10,no,True +district:224,Vaishali,224,http://www.wikidata.org/entity/Q100130,DISTRICT,state:10,no,True +district:611,Arwal,611,http://www.wikidata.org/entity/Q42917,DISTRICT,state:10,no,True +district:225,Gangtok,225,http://www.wikidata.org/entity/Q113956160,DISTRICT,state:11,no,True +district:226,Mangan,226,http://www.wikidata.org/entity/Q1784149,DISTRICT,state:11,no,True +district:227,Namchi,227,http://www.wikidata.org/entity/Q1805051,DISTRICT,state:11,no,True +district:228,Gyalshing,228,http://www.wikidata.org/entity/Q611357,DISTRICT,state:11,no,True +district:741,Pakyong,741,http://www.wikidata.org/entity/Q108803704,DISTRICT,state:11,no,True +district:742,Soreng,742,http://www.wikidata.org/entity/Q112939132,DISTRICT,state:11,no,True +district:229,Changlang,229,http://www.wikidata.org/entity/Q15427,DISTRICT,state:12,no,True +district:230,Dibang Valley,230,http://www.wikidata.org/entity/Q15446,DISTRICT,state:12,no,True +district:231,East Kameng,231,http://www.wikidata.org/entity/Q15424,DISTRICT,state:12,no,True +district:232,East Siang,232,http://www.wikidata.org/entity/Q15419,DISTRICT,state:12,no,True +district:233,Kurung Kumey,233,http://www.wikidata.org/entity/Q2449506,DISTRICT,state:12,no,True +district:234,Lohit,234,http://www.wikidata.org/entity/Q15438,DISTRICT,state:12,no,True +district:235,Lower Dibang Valley,235,http://www.wikidata.org/entity/Q2373368,DISTRICT,state:12,no,True +district:236,Lower Subansiri,236,http://www.wikidata.org/entity/Q15436,DISTRICT,state:12,no,True +district:237,Papum Pare,237,http://www.wikidata.org/entity/Q15432,DISTRICT,state:12,no,True +district:238,Tawang,238,http://www.wikidata.org/entity/Q15449,DISTRICT,state:12,no,True +district:239,Tirap,239,http://www.wikidata.org/entity/Q15448,DISTRICT,state:12,no,True +district:240,Upper Siang,240,http://www.wikidata.org/entity/Q15465,DISTRICT,state:12,no,True +district:241,Upper Subansiri,241,http://www.wikidata.org/entity/Q15464,DISTRICT,state:12,no,True +district:242,West Kameng,242,http://www.wikidata.org/entity/Q15459,DISTRICT,state:12,no,True +district:243,West Siang,243,http://www.wikidata.org/entity/Q15453,DISTRICT,state:12,no,True +district:628,Anjaw,628,http://www.wikidata.org/entity/Q15413,DISTRICT,state:12,no,True +district:666,Longding,666,http://www.wikidata.org/entity/Q5627568,DISTRICT,state:12,no,True +district:677,Kra Daadi,677,http://www.wikidata.org/entity/Q21018627,DISTRICT,state:12,no,True +district:678,Namsai,678,http://www.wikidata.org/entity/Q21559824,DISTRICT,state:12,no,True +district:679,Siang,679,http://www.wikidata.org/entity/Q18642331,DISTRICT,state:12,no,True +district:718,Kamle,718,http://www.wikidata.org/entity/Q48731073,DISTRICT,state:12,no,True +district:719,Lower Siang,719,http://www.wikidata.org/entity/Q13602925,DISTRICT,state:12,no,True +district:723,Pakke-Kessang,723,http://www.wikidata.org/entity/Q61439260,DISTRICT,state:12,no,True +district:724,Lepa Rada,724,http://www.wikidata.org/entity/Q63563632,DISTRICT,state:12,no,True +district:725,Shi Yomi,725,http://www.wikidata.org/entity/Q63563625,DISTRICT,state:12,no,True +district:786,Keyi Panyor,786,http://www.wikidata.org/entity/Q124818540,DISTRICT,state:12,no,True +district:787,Bichom,787,http://www.wikidata.org/entity/Q124811465,DISTRICT,state:12,no,True +district:244,Dimapur,244,http://www.wikidata.org/entity/Q634262,DISTRICT,state:13,no,True +district:245,Kohima,245,http://www.wikidata.org/entity/Q953530,DISTRICT,state:13,no,True +district:246,Mokokchung,246,http://www.wikidata.org/entity/Q2175311,DISTRICT,state:13,no,True +district:247,Mon,247,http://www.wikidata.org/entity/Q2339648,DISTRICT,state:13,no,True +district:248,Phek,248,http://www.wikidata.org/entity/Q590882,DISTRICT,state:13,no,True +district:249,Tuensang,249,http://www.wikidata.org/entity/Q2571393,DISTRICT,state:13,no,True +district:250,Wokha,250,http://www.wikidata.org/entity/Q681821,DISTRICT,state:13,no,True +district:251,Zunheboto,251,http://www.wikidata.org/entity/Q2091461,DISTRICT,state:13,no,True +district:613,Peren,613,http://www.wikidata.org/entity/Q516294,DISTRICT,state:13,no,True +district:614,Kiphire,614,http://www.wikidata.org/entity/Q2597908,DISTRICT,state:13,no,True +district:615,Longleng,615,http://www.wikidata.org/entity/Q1426783,DISTRICT,state:13,no,True +district:736,Noklak,736,http://www.wikidata.org/entity/Q48731903,DISTRICT,state:13,no,True +district:757,Tseminyü,757,http://www.wikidata.org/entity/Q110223836,DISTRICT,state:13,no,True +district:758,Chümoukedima,758,http://www.wikidata.org/entity/Q110223837,DISTRICT,state:13,no,True +district:764,Niuland,764,http://www.wikidata.org/entity/Q110223839,DISTRICT,state:13,no,True +district:765,Shamator,765,http://www.wikidata.org/entity/Q111529435,DISTRICT,state:13,no,True +district:788,Meluri,788,http://www.wikidata.org/entity/Q131191793,DISTRICT,state:13,no,True +district:252,Bishnupur,252,http://www.wikidata.org/entity/Q938190,DISTRICT,state:14,no,True +district:253,Chandel,253,http://www.wikidata.org/entity/Q2301769,DISTRICT,state:14,no,True +district:254,Churachandpur,254,http://www.wikidata.org/entity/Q2577281,DISTRICT,state:14,no,True +district:255,Imphal East,255,http://www.wikidata.org/entity/Q1916666,DISTRICT,state:14,no,True +district:256,Imphal West,256,http://www.wikidata.org/entity/Q1822188,DISTRICT,state:14,no,True +district:257,Senapati,257,http://www.wikidata.org/entity/Q2301706,DISTRICT,state:14,no,True +district:258,Tamenglong,258,http://www.wikidata.org/entity/Q2301717,DISTRICT,state:14,no,True +district:259,Thoubal,259,http://www.wikidata.org/entity/Q2086198,DISTRICT,state:14,no,True +district:260,Ukhrul,260,http://www.wikidata.org/entity/Q735101,DISTRICT,state:14,no,True +district:711,Kakching,711,http://www.wikidata.org/entity/Q28173825,DISTRICT,state:14,no,True +district:712,Kangpokpi,712,http://www.wikidata.org/entity/Q28419386,DISTRICT,state:14,no,True +district:713,Jiribam,713,http://www.wikidata.org/entity/Q28419387,DISTRICT,state:14,no,True +district:714,Noney,714,http://www.wikidata.org/entity/Q28419389,DISTRICT,state:14,no,True +district:715,Pherzawl,715,http://www.wikidata.org/entity/Q28173809,DISTRICT,state:14,no,True +district:716,Tengnoupal,716,http://www.wikidata.org/entity/Q28419388,DISTRICT,state:14,no,True +district:717,Kamjong,717,http://www.wikidata.org/entity/Q28419390,DISTRICT,state:14,no,True +district:261,Aizawl,261,http://www.wikidata.org/entity/Q1947322,DISTRICT,state:15,no,True +district:262,Champhai,262,http://www.wikidata.org/entity/Q1965256,DISTRICT,state:15,no,True +district:263,Kolasib,263,http://www.wikidata.org/entity/Q1947343,DISTRICT,state:15,no,True +district:264,Lawngtlai,264,http://www.wikidata.org/entity/Q2086209,DISTRICT,state:15,no,True +district:265,Lunglei,265,http://www.wikidata.org/entity/Q1947352,DISTRICT,state:15,no,True +district:266,Mamit,266,http://www.wikidata.org/entity/Q751531,DISTRICT,state:15,no,True +district:267,Saiha,267,http://www.wikidata.org/entity/Q1821714,DISTRICT,state:15,no,True +district:268,Serchhip,268,http://www.wikidata.org/entity/Q2086190,DISTRICT,state:15,no,True +district:726,Hnahthial,726,http://www.wikidata.org/entity/Q86882590,DISTRICT,state:15,no,True +district:727,Saitual,727,http://www.wikidata.org/entity/Q86882593,DISTRICT,state:15,no,True +district:728,Khawzawl,728,http://www.wikidata.org/entity/Q86882591,DISTRICT,state:15,no,True +district:269,Dhalai,269,http://www.wikidata.org/entity/Q2086546,DISTRICT,state:16,no,True +district:270,North Tripura,270,http://www.wikidata.org/entity/Q1920978,DISTRICT,state:16,no,True +district:271,South Tripura,271,http://www.wikidata.org/entity/Q1822159,DISTRICT,state:16,no,True +district:272,West Tripura,272,http://www.wikidata.org/entity/Q1947570,DISTRICT,state:16,no,True +district:652,Khowai,652,http://www.wikidata.org/entity/Q16086680,DISTRICT,state:16,no,True +district:653,Sepahijala,653,http://www.wikidata.org/entity/Q16086076,DISTRICT,state:16,no,True +district:654,Gomati,654,http://www.wikidata.org/entity/Q16086497,DISTRICT,state:16,no,True +district:655,Unakoti,655,http://www.wikidata.org/entity/Q16087996,DISTRICT,state:16,no,True +district:273,East Garo Hills,273,http://www.wikidata.org/entity/Q2085455,DISTRICT,state:17,no,True +district:274,East Khasi Hills,274,http://www.wikidata.org/entity/Q1945304,DISTRICT,state:17,no,True +district:275,West Jaintia Hills,275,http://www.wikidata.org/entity/Q13181190,DISTRICT,state:17,no,True +district:276,Ri-Bhoi,276,http://www.wikidata.org/entity/Q1884672,DISTRICT,state:17,no,True +district:277,South Garo Hills,277,http://www.wikidata.org/entity/Q2329228,DISTRICT,state:17,no,True +district:278,West Garo Hills,278,http://www.wikidata.org/entity/Q2329181,DISTRICT,state:17,no,True +district:279,West Khasi Hills,279,http://www.wikidata.org/entity/Q2064752,DISTRICT,state:17,no,True +district:656,North Garo Hills,656,http://www.wikidata.org/entity/Q7055466,DISTRICT,state:17,no,True +district:657,East Jaintia Hills,657,http://www.wikidata.org/entity/Q15923776,DISTRICT,state:17,no,True +district:658,South West Khasi Hills,658,http://www.wikidata.org/entity/Q15923741,DISTRICT,state:17,no,True +district:663,South West Garo Hills,663,http://www.wikidata.org/entity/Q15961576,DISTRICT,state:17,no,True +district:740,Eastern West Khasi Hills,740,http://www.wikidata.org/entity/Q110442602,DISTRICT,state:17,no,True +district:280,Barpeta,280,http://www.wikidata.org/entity/Q41249,DISTRICT,state:18,no,True +district:281,Bongaigaon,281,http://www.wikidata.org/entity/Q42197,DISTRICT,state:18,no,True +district:282,Cachar,282,http://www.wikidata.org/entity/Q42209,DISTRICT,state:18,no,True +district:283,Darrang,283,http://www.wikidata.org/entity/Q42461,DISTRICT,state:18,no,True +district:284,Dhemaji,284,http://www.wikidata.org/entity/Q42473,DISTRICT,state:18,no,True +district:285,Dhubri,285,http://www.wikidata.org/entity/Q42485,DISTRICT,state:18,no,True +district:286,Dibrugarh,286,http://www.wikidata.org/entity/Q42479,DISTRICT,state:18,no,True +district:287,Goalpara,287,http://www.wikidata.org/entity/Q42522,DISTRICT,state:18,no,True +district:288,Golaghat,288,http://www.wikidata.org/entity/Q42517,DISTRICT,state:18,no,True +district:289,Hailakandi,289,http://www.wikidata.org/entity/Q42505,DISTRICT,state:18,no,True +district:290,Jorhat,290,http://www.wikidata.org/entity/Q42611,DISTRICT,state:18,no,True +district:291,Kamrup,291,http://www.wikidata.org/entity/Q2247441,DISTRICT,state:18,no,True +district:292,Karbi Anglong,292,http://www.wikidata.org/entity/Q29025081,DISTRICT,state:18,no,True +district:293,Sribhumi,293,http://www.wikidata.org/entity/Q42542,DISTRICT,state:18,no,True +district:294,Kokrajhar,294,http://www.wikidata.org/entity/Q42618,DISTRICT,state:18,no,True +district:295,Lakhimpur,295,http://www.wikidata.org/entity/Q42743,DISTRICT,state:18,no,True +district:296,Morigaon,296,http://www.wikidata.org/entity/Q42737,DISTRICT,state:18,no,True +district:297,,297,http://www.wikidata.org/entity/Q42686,DISTRICT,state:18,no,True +district:298,Nalbari,298,http://www.wikidata.org/entity/Q42779,DISTRICT,state:18,no,True +district:299,Dima Hasao,299,http://www.wikidata.org/entity/Q42774,DISTRICT,state:18,no,True +district:300,Sivasagar,300,http://www.wikidata.org/entity/Q42768,DISTRICT,state:18,no,True +district:301,Sonitpur,301,http://www.wikidata.org/entity/Q42765,DISTRICT,state:18,no,True +district:302,Tinsukia,302,http://www.wikidata.org/entity/Q42756,DISTRICT,state:18,no,True +district:612,Chirang,612,http://www.wikidata.org/entity/Q2574898,DISTRICT,state:18,no,True +district:616,Baksa,616,http://www.wikidata.org/entity/Q2360266,DISTRICT,state:18,no,True +district:617,Udalguri,617,http://www.wikidata.org/entity/Q321998,DISTRICT,state:18,no,True +district:618,Kamrup Metropolitan,618,http://www.wikidata.org/entity/Q2464674,DISTRICT,state:18,no,True +district:705,Bishwanath,705,http://www.wikidata.org/entity/Q22079836,DISTRICT,state:18,no,True +district:706,Majuli,706,http://www.wikidata.org/entity/Q28110729,DISTRICT,state:18,no,True +district:707,South Salmara-Mankachar,707,http://www.wikidata.org/entity/Q24907599,DISTRICT,state:18,no,True +district:708,Charaideo,708,http://www.wikidata.org/entity/Q24039029,DISTRICT,state:18,no,True +district:709,Hojai,709,http://www.wikidata.org/entity/Q24699407,DISTRICT,state:18,no,True +district:710,West Karbi Anglong,710,http://www.wikidata.org/entity/Q24949218,DISTRICT,state:18,no,True +district:739,Bajali,739,http://www.wikidata.org/entity/Q101088203,DISTRICT,state:18,no,True +district:756,Tamulpur,756,http://www.wikidata.org/entity/Q110661970,DISTRICT,state:18,no,True +district:303,North 24 Parganas,303,http://www.wikidata.org/entity/Q338425,DISTRICT,state:19,no,True +district:304,South 24 Parganas,304,http://www.wikidata.org/entity/Q2308319,DISTRICT,state:19,no,True +district:305,Bankura,305,http://www.wikidata.org/entity/Q2088458,DISTRICT,state:19,no,True +district:306,Purba Bardhaman,306,http://www.wikidata.org/entity/Q29257278,DISTRICT,state:19,no,True +district:307,Birbhum,307,http://www.wikidata.org/entity/Q2088440,DISTRICT,state:19,no,True +district:308,Cooch Behar,308,http://www.wikidata.org/entity/Q2728658,DISTRICT,state:19,no,True +district:309,Darjeeling,309,http://www.wikidata.org/entity/Q1134759,DISTRICT,state:19,no,True +district:310,Dakshin Dinajpur,310,http://www.wikidata.org/entity/Q533839,DISTRICT,state:19,no,True +district:311,Uttar Dinajpur,311,http://www.wikidata.org/entity/Q2019766,DISTRICT,state:19,no,True +district:312,Hooghly,312,http://www.wikidata.org/entity/Q548518,DISTRICT,state:19,no,True +district:313,Howrah,313,http://www.wikidata.org/entity/Q1478937,DISTRICT,state:19,no,True +district:314,Jalpaiguri,314,http://www.wikidata.org/entity/Q1351487,DISTRICT,state:19,no,True +district:315,Kolkata,315,http://www.wikidata.org/entity/Q2088496,DISTRICT,state:19,no,True +district:316,Malda,316,http://www.wikidata.org/entity/Q2049820,DISTRICT,state:19,no,True +district:317,Purba Medinipur,317,http://www.wikidata.org/entity/Q1431920,DISTRICT,state:19,no,True +district:318,Paschim Medinipur,318,http://www.wikidata.org/entity/Q1855537,DISTRICT,state:19,no,True +district:319,Murshidabad,319,http://www.wikidata.org/entity/Q1546240,DISTRICT,state:19,no,True +district:320,Nadia,320,http://www.wikidata.org/entity/Q1143880,DISTRICT,state:19,no,True +district:321,Purulia,321,http://www.wikidata.org/entity/Q307474,DISTRICT,state:19,no,True +district:664,Alipurduar,664,http://www.wikidata.org/entity/Q4726845,DISTRICT,state:19,no,True +district:702,Kalimpong,702,http://www.wikidata.org/entity/Q28769140,DISTRICT,state:19,no,True +district:703,Jhargram,703,http://www.wikidata.org/entity/Q29168456,DISTRICT,state:19,no,True +district:704,Paschim Bardhaman,704,http://www.wikidata.org/entity/Q29215602,DISTRICT,state:19,no,True +district:15,Bilaspur,15,http://www.wikidata.org/entity/Q1478939,DISTRICT,state:2,no,True +district:16,Chamba,16,http://www.wikidata.org/entity/Q1060614,DISTRICT,state:2,no,True +district:17,Hamirpur,17,http://www.wikidata.org/entity/Q2086180,DISTRICT,state:2,no,True +district:18,Kangra,18,http://www.wikidata.org/entity/Q727232,DISTRICT,state:2,no,True +district:19,Kinnaur,19,http://www.wikidata.org/entity/Q1862950,DISTRICT,state:2,no,True +district:20,Kullu,20,http://www.wikidata.org/entity/Q2980880,DISTRICT,state:2,no,True +district:21,Lahaul and Spiti,21,http://www.wikidata.org/entity/Q837595,DISTRICT,state:2,no,True +district:22,Mandi,22,http://www.wikidata.org/entity/Q1892161,DISTRICT,state:2,no,True +district:23,Shimla,23,http://www.wikidata.org/entity/Q1921404,DISTRICT,state:2,no,True +district:24,Sirmaur,24,http://www.wikidata.org/entity/Q654331,DISTRICT,state:2,no,True +district:25,Solan,25,http://www.wikidata.org/entity/Q2980937,DISTRICT,state:2,no,True +district:26,Una,26,http://www.wikidata.org/entity/Q2301741,DISTRICT,state:2,no,True +district:322,Bokaro,322,http://www.wikidata.org/entity/Q2295925,DISTRICT,state:20,no,True +district:323,Chatra,323,http://www.wikidata.org/entity/Q1979499,DISTRICT,state:20,no,True +district:324,Deoghar,324,http://www.wikidata.org/entity/Q2030017,DISTRICT,state:20,no,True +district:325,Dhanbad,325,http://www.wikidata.org/entity/Q2240791,DISTRICT,state:20,no,True +district:326,Dumka,326,http://www.wikidata.org/entity/Q2577657,DISTRICT,state:20,no,True +district:327,East Singhbhum,327,http://www.wikidata.org/entity/Q2452921,DISTRICT,state:20,no,True +district:328,Garhwa,328,http://www.wikidata.org/entity/Q2302076,DISTRICT,state:20,no,True +district:329,Giridih,329,http://www.wikidata.org/entity/Q2302065,DISTRICT,state:20,no,True +district:330,Godda,330,http://www.wikidata.org/entity/Q638980,DISTRICT,state:20,no,True +district:331,Gumla,331,http://www.wikidata.org/entity/Q2295865,DISTRICT,state:20,no,True +district:332,Hazaribagh,332,http://www.wikidata.org/entity/Q1945416,DISTRICT,state:20,no,True +district:333,Jamtara,333,http://www.wikidata.org/entity/Q2980986,DISTRICT,state:20,no,True +district:334,Koderma,334,http://www.wikidata.org/entity/Q2085480,DISTRICT,state:20,no,True +district:335,Latehar,335,http://www.wikidata.org/entity/Q2244762,DISTRICT,state:20,no,True +district:336,Lohardaga,336,http://www.wikidata.org/entity/Q1948301,DISTRICT,state:20,no,True +district:337,Pakur,337,http://www.wikidata.org/entity/Q2295930,DISTRICT,state:20,no,True +district:338,Palamu,338,http://www.wikidata.org/entity/Q1797254,DISTRICT,state:20,no,True +district:339,Ranchi,339,http://www.wikidata.org/entity/Q1947380,DISTRICT,state:20,no,True +district:340,Sahebganj,340,http://www.wikidata.org/entity/Q767878,DISTRICT,state:20,no,True +district:341,Seraikela Kharsawan,341,http://www.wikidata.org/entity/Q2362658,DISTRICT,state:20,no,True +district:342,Simdega,342,http://www.wikidata.org/entity/Q2597889,DISTRICT,state:20,no,True +district:343,West Singhbhum,343,http://www.wikidata.org/entity/Q1950527,DISTRICT,state:20,no,True +district:606,Khunti,606,http://www.wikidata.org/entity/Q367344,DISTRICT,state:20,no,True +district:607,Ramgarh,607,http://www.wikidata.org/entity/Q2663612,DISTRICT,state:20,no,True +district:344,Angul,344,http://www.wikidata.org/entity/Q1772807,DISTRICT,state:21,no,True +district:345,Balangir,345,http://www.wikidata.org/entity/Q804642,DISTRICT,state:21,no,True +district:346,Balasore,346,http://www.wikidata.org/entity/Q2022279,DISTRICT,state:21,no,True +district:347,Bargarh,347,http://www.wikidata.org/entity/Q808140,DISTRICT,state:21,no,True +district:348,Bhadrak,348,http://www.wikidata.org/entity/Q685638,DISTRICT,state:21,no,True +district:349,Boudh,349,http://www.wikidata.org/entity/Q2363639,DISTRICT,state:21,no,True +district:350,Cuttack,350,http://www.wikidata.org/entity/Q2022256,DISTRICT,state:21,no,True +district:351,Debagarh,351,http://www.wikidata.org/entity/Q2269639,DISTRICT,state:21,no,True +district:352,Dhenkanal,352,http://www.wikidata.org/entity/Q1948389,DISTRICT,state:21,no,True +district:353,Gajapati,353,http://www.wikidata.org/entity/Q1947292,DISTRICT,state:21,no,True +district:354,Ganjam,354,http://www.wikidata.org/entity/Q776213,DISTRICT,state:21,no,True +district:355,Jagatsinghpur,355,http://www.wikidata.org/entity/Q971581,DISTRICT,state:21,no,True +district:356,Jajpur,356,http://www.wikidata.org/entity/Q2087771,DISTRICT,state:21,no,True +district:357,Jharsuguda,357,http://www.wikidata.org/entity/Q569181,DISTRICT,state:21,no,True +district:358,Kalahandi,358,http://www.wikidata.org/entity/Q1876588,DISTRICT,state:21,no,True +district:359,Kandhamal,359,http://www.wikidata.org/entity/Q2085500,DISTRICT,state:21,no,True +district:360,Kendrapara,360,http://www.wikidata.org/entity/Q2299172,DISTRICT,state:21,no,True +district:361,Kendujhar,361,http://www.wikidata.org/entity/Q2085428,DISTRICT,state:21,no,True +district:362,Khordha,362,http://www.wikidata.org/entity/Q662818,DISTRICT,state:21,no,True +district:363,Koraput,363,http://www.wikidata.org/entity/Q1947300,DISTRICT,state:21,no,True +district:364,Malkangiri,364,http://www.wikidata.org/entity/Q5122619,DISTRICT,state:21,no,True +district:365,Mayurbhanj,365,http://www.wikidata.org/entity/Q1914546,DISTRICT,state:21,no,True +district:366,Nabarangpur,366,http://www.wikidata.org/entity/Q2396798,DISTRICT,state:21,no,True +district:367,Nayagarh,367,http://www.wikidata.org/entity/Q2367388,DISTRICT,state:21,no,True +district:368,Nuapada,368,http://www.wikidata.org/entity/Q1810550,DISTRICT,state:21,no,True +district:369,Puri,369,http://www.wikidata.org/entity/Q1817158,DISTRICT,state:21,no,True +district:370,Rayagada,370,http://www.wikidata.org/entity/Q2577997,DISTRICT,state:21,no,True +district:371,Sambalpur,371,http://www.wikidata.org/entity/Q1267306,DISTRICT,state:21,no,True +district:372,Subarnapur,372,http://www.wikidata.org/entity/Q1473957,DISTRICT,state:21,no,True +district:373,Sundargarh,373,http://www.wikidata.org/entity/Q2296047,DISTRICT,state:21,no,True +district:374,Bastar,374,http://www.wikidata.org/entity/Q100152,DISTRICT,state:22,no,True +district:375,Bilaspur,375,http://www.wikidata.org/entity/Q100157,DISTRICT,state:22,no,True +district:376,Dantewada,376,http://www.wikidata.org/entity/Q100211,DISTRICT,state:22,no,True +district:377,Dhamtari,377,http://www.wikidata.org/entity/Q100190,DISTRICT,state:22,no,True +district:378,Durg,378,http://www.wikidata.org/entity/Q100182,DISTRICT,state:22,no,True +district:379,Janjgir–Champa,379,http://www.wikidata.org/entity/Q2575633,DISTRICT,state:22,no,True +district:380,Jashpur,380,http://www.wikidata.org/entity/Q2577551,DISTRICT,state:22,no,True +district:381,Kanker,381,http://www.wikidata.org/entity/Q2310530,DISTRICT,state:22,no,True +district:382,Kabirdham,382,http://www.wikidata.org/entity/Q2450255,DISTRICT,state:22,no,True +district:383,Korba,383,http://www.wikidata.org/entity/Q2299121,DISTRICT,state:22,no,True +district:384,Koriya,384,http://www.wikidata.org/entity/Q2295896,DISTRICT,state:22,no,True +district:385,Mahasamund,385,http://www.wikidata.org/entity/Q2450240,DISTRICT,state:22,no,True +district:386,Raigarh,386,http://www.wikidata.org/entity/Q2286310,DISTRICT,state:22,no,True +district:387,Raipur,387,http://www.wikidata.org/entity/Q2295914,DISTRICT,state:22,no,True +district:388,Rajnandgaon,388,http://www.wikidata.org/entity/Q2341800,DISTRICT,state:22,no,True +district:389,Surguja,389,http://www.wikidata.org/entity/Q1805075,DISTRICT,state:22,no,True +district:636,Bijapur,636,http://www.wikidata.org/entity/Q100164,DISTRICT,state:22,no,True +district:637,Narayanpur,637,http://www.wikidata.org/entity/Q2322000,DISTRICT,state:22,no,True +district:642,Sukma,642,http://www.wikidata.org/entity/Q16933590,DISTRICT,state:22,no,True +district:643,Kondagaon,643,http://www.wikidata.org/entity/Q12420995,DISTRICT,state:22,no,True +district:644,Baloda Bazar,644,http://www.wikidata.org/entity/Q15663455,DISTRICT,state:22,no,True +district:645,Gariaband,645,http://www.wikidata.org/entity/Q16961365,DISTRICT,state:22,no,True +district:646,Balod,646,http://www.wikidata.org/entity/Q16056266,DISTRICT,state:22,no,True +district:647,Mungeli,647,http://www.wikidata.org/entity/Q13476249,DISTRICT,state:22,no,True +district:648,Surajpur,648,http://www.wikidata.org/entity/Q16938031,DISTRICT,state:22,no,True +district:649,Balrampur–Ramanujganj,649,http://www.wikidata.org/entity/Q16056268,DISTRICT,state:22,no,True +district:650,Bemetara,650,http://www.wikidata.org/entity/Q16254159,DISTRICT,state:22,no,True +district:734,Gaurela-Pendra-Marwahi,734,http://www.wikidata.org/entity/Q96584972,DISTRICT,state:22,no,True +district:759,Khairagarh-Chhuikhadan-Gandai,759,http://www.wikidata.org/entity/Q113485010,DISTRICT,state:22,no,True +district:760,Manendragarh-Chirmiri-Bharatpur,760,http://www.wikidata.org/entity/Q108427451,DISTRICT,state:22,no,True +district:761,Mohla-Manpur-Ambagarh Chowki,761,http://www.wikidata.org/entity/Q108569901,DISTRICT,state:22,no,True +district:762,Sakti,762,http://www.wikidata.org/entity/Q108569905,DISTRICT,state:22,no,True +district:763,Sarangarh-Bilaigarh,763,http://www.wikidata.org/entity/Q108569900,DISTRICT,state:22,no,True +district:390,Anuppur,390,http://www.wikidata.org/entity/Q2299093,DISTRICT,state:23,no,True +district:391,Ashoknagar,391,http://www.wikidata.org/entity/Q2246416,DISTRICT,state:23,no,True +district:392,Balaghat,392,http://www.wikidata.org/entity/Q641904,DISTRICT,state:23,no,True +district:393,Barwani,393,http://www.wikidata.org/entity/Q2126754,DISTRICT,state:23,no,True +district:394,Betul,394,http://www.wikidata.org/entity/Q1815279,DISTRICT,state:23,no,True +district:395,Bhind,395,http://www.wikidata.org/entity/Q2341700,DISTRICT,state:23,no,True +district:396,Bhopal,396,http://www.wikidata.org/entity/Q1797245,DISTRICT,state:23,no,True +district:397,Burhanpur,397,http://www.wikidata.org/entity/Q2125592,DISTRICT,state:23,no,True +district:398,Chhatarpur,398,http://www.wikidata.org/entity/Q2449785,DISTRICT,state:23,no,True +district:399,Chhindwara,399,http://www.wikidata.org/entity/Q1986096,DISTRICT,state:23,no,True +district:400,Damoh,400,http://www.wikidata.org/entity/Q2479331,DISTRICT,state:23,no,True +district:401,Datia,401,http://www.wikidata.org/entity/Q2206266,DISTRICT,state:23,no,True +district:402,Dewas,402,http://www.wikidata.org/entity/Q2025998,DISTRICT,state:23,no,True +district:403,Dhar,403,http://www.wikidata.org/entity/Q2299069,DISTRICT,state:23,no,True +district:404,Dindori,404,http://www.wikidata.org/entity/Q2398551,DISTRICT,state:23,no,True +district:405,Khandwa,405,http://www.wikidata.org/entity/Q2085436,DISTRICT,state:23,no,True +district:406,Guna,406,http://www.wikidata.org/entity/Q930027,DISTRICT,state:23,no,True +district:407,Gwalior,407,http://www.wikidata.org/entity/Q2085310,DISTRICT,state:23,no,True +district:408,Harda,408,http://www.wikidata.org/entity/Q2173003,DISTRICT,state:23,no,True +district:409,Narmadapuram,409,http://www.wikidata.org/entity/Q620801,DISTRICT,state:23,no,True +district:410,Indore,410,http://www.wikidata.org/entity/Q742938,DISTRICT,state:23,no,True +district:411,Jabalpur,411,http://www.wikidata.org/entity/Q632093,DISTRICT,state:23,no,True +district:412,Jhabua,412,http://www.wikidata.org/entity/Q2085336,DISTRICT,state:23,no,True +district:413,Katni,413,http://www.wikidata.org/entity/Q746441,DISTRICT,state:23,no,True +district:414,Khargone,414,http://www.wikidata.org/entity/Q2273900,DISTRICT,state:23,no,True +district:415,Mandla,415,http://www.wikidata.org/entity/Q2341670,DISTRICT,state:23,no,True +district:416,Mandsaur,416,http://www.wikidata.org/entity/Q1870014,DISTRICT,state:23,no,True +district:417,Morena,417,http://www.wikidata.org/entity/Q2341467,DISTRICT,state:23,no,True +district:418,Narsinghpur,418,http://www.wikidata.org/entity/Q2341616,DISTRICT,state:23,no,True +district:419,Neemuch,419,http://www.wikidata.org/entity/Q2341713,DISTRICT,state:23,no,True +district:420,Panna,420,http://www.wikidata.org/entity/Q2341630,DISTRICT,state:23,no,True +district:421,Raisen,421,http://www.wikidata.org/entity/Q1815223,DISTRICT,state:23,no,True +district:422,Rajgarh,422,http://www.wikidata.org/entity/Q1833306,DISTRICT,state:23,no,True +district:423,Ratlam,423,http://www.wikidata.org/entity/Q2299164,DISTRICT,state:23,no,True +district:424,Rewa,424,http://www.wikidata.org/entity/Q526862,DISTRICT,state:23,no,True +district:425,Sagar,425,http://www.wikidata.org/entity/Q2085421,DISTRICT,state:23,no,True +district:426,Satna,426,http://www.wikidata.org/entity/Q2577924,DISTRICT,state:23,no,True +district:427,Sehore,427,http://www.wikidata.org/entity/Q2299029,DISTRICT,state:23,no,True +district:428,Seoni,428,http://www.wikidata.org/entity/Q2221184,DISTRICT,state:23,no,True +district:429,Shahdol,429,http://www.wikidata.org/entity/Q2085464,DISTRICT,state:23,no,True +district:430,Shajapur,430,http://www.wikidata.org/entity/Q2449803,DISTRICT,state:23,no,True +district:431,Sheopur,431,http://www.wikidata.org/entity/Q620105,DISTRICT,state:23,no,True +district:432,Shivpuri,432,http://www.wikidata.org/entity/Q2299042,DISTRICT,state:23,no,True +district:433,Sidhi,433,http://www.wikidata.org/entity/Q2449793,DISTRICT,state:23,no,True +district:434,Tikamgarh,434,http://www.wikidata.org/entity/Q2449760,DISTRICT,state:23,no,True +district:435,Ujjain,435,http://www.wikidata.org/entity/Q892641,DISTRICT,state:23,no,True +district:436,Umaria,436,http://www.wikidata.org/entity/Q620297,DISTRICT,state:23,no,True +district:437,Vidisha,437,http://www.wikidata.org/entity/Q1815253,DISTRICT,state:23,no,True +district:638,Singrauli,638,http://www.wikidata.org/entity/Q2668638,DISTRICT,state:23,no,True +district:639,Alirajpur,639,http://www.wikidata.org/entity/Q2667586,DISTRICT,state:23,no,True +district:667,Agar Malwa,667,http://www.wikidata.org/entity/Q15732396,DISTRICT,state:23,no,True +district:722,Niwari,722,http://www.wikidata.org/entity/Q63563797,DISTRICT,state:23,no,True +district:766,Mauganj,766,http://www.wikidata.org/entity/Q122417864,DISTRICT,state:23,no,True +district:784,Maihar,784,http://www.wikidata.org/entity/Q111675213,DISTRICT,state:23,no,True +district:785,Pandhurna,785,http://www.wikidata.org/entity/Q123286184,DISTRICT,state:23,no,True +district:438,Ahmedabad,438,http://www.wikidata.org/entity/Q401686,DISTRICT,state:24,no,True +district:439,Amreli,439,http://www.wikidata.org/entity/Q257946,DISTRICT,state:24,no,True +district:440,Anand,440,http://www.wikidata.org/entity/Q485683,DISTRICT,state:24,no,True +district:441,Banaskantha,441,http://www.wikidata.org/entity/Q806125,DISTRICT,state:24,no,True +district:442,Bharuch,442,http://www.wikidata.org/entity/Q854900,DISTRICT,state:24,no,True +district:443,Bhavnagar,443,http://www.wikidata.org/entity/Q854963,DISTRICT,state:24,no,True +district:444,Dang,444,http://www.wikidata.org/entity/Q1135616,DISTRICT,state:24,no,True +district:445,Dahod,445,http://www.wikidata.org/entity/Q186518,DISTRICT,state:24,no,True +district:446,Gandhinagar,446,http://www.wikidata.org/entity/Q1772860,DISTRICT,state:24,no,True +district:447,Jamnagar,447,http://www.wikidata.org/entity/Q2982118,DISTRICT,state:24,no,True +district:448,Junagadh,448,http://www.wikidata.org/entity/Q1797344,DISTRICT,state:24,no,True +district:449,Kutch,449,http://www.wikidata.org/entity/Q1063417,DISTRICT,state:24,no,True +district:450,Kheda,450,http://www.wikidata.org/entity/Q1755463,DISTRICT,state:24,no,True +district:451,Mehsana,451,http://www.wikidata.org/entity/Q2019694,DISTRICT,state:24,no,True +district:452,Narmada,452,http://www.wikidata.org/entity/Q1797230,DISTRICT,state:24,no,True +district:453,Navsari,453,http://www.wikidata.org/entity/Q1797349,DISTRICT,state:24,no,True +district:454,Panchmahal,454,http://www.wikidata.org/entity/Q1781463,DISTRICT,state:24,no,True +district:455,Patan,455,http://www.wikidata.org/entity/Q1815269,DISTRICT,state:24,no,True +district:456,Porbandar,456,http://www.wikidata.org/entity/Q1772815,DISTRICT,state:24,no,True +district:457,Rajkot,457,http://www.wikidata.org/entity/Q1815245,DISTRICT,state:24,no,True +district:458,Sabarkantha,458,http://www.wikidata.org/entity/Q1772856,DISTRICT,state:24,no,True +district:459,Surat,459,http://www.wikidata.org/entity/Q1797317,DISTRICT,state:24,no,True +district:460,Surendranagar,460,http://www.wikidata.org/entity/Q237535,DISTRICT,state:24,no,True +district:461,Vadodara,461,http://www.wikidata.org/entity/Q578285,DISTRICT,state:24,no,True +district:462,Valsad,462,http://www.wikidata.org/entity/Q1946743,DISTRICT,state:24,no,True +district:641,Tapi,641,http://www.wikidata.org/entity/Q670165,DISTRICT,state:24,no,True +district:668,Chhota Udaipur,668,http://www.wikidata.org/entity/Q5979243,DISTRICT,state:24,no,True +district:669,Mahisagar,669,http://www.wikidata.org/entity/Q5706885,DISTRICT,state:24,no,True +district:672,Aravalli,672,http://www.wikidata.org/entity/Q12175285,DISTRICT,state:24,no,True +district:673,Morbi,673,http://www.wikidata.org/entity/Q5979727,DISTRICT,state:24,no,True +district:674,Devbhumi Dwarka,674,http://www.wikidata.org/entity/Q14594717,DISTRICT,state:24,no,True +district:675,Gir Somnath,675,http://www.wikidata.org/entity/Q15244465,DISTRICT,state:24,no,True +district:676,Botad,676,http://www.wikidata.org/entity/Q14505072,DISTRICT,state:24,no,True +district:789,Vav-Tharad,789,http://www.wikidata.org/entity/Q131621560,DISTRICT,state:24,no,True +district:466,Ahilyanagar,466,http://www.wikidata.org/entity/Q401744,DISTRICT,state:27,no,True +district:467,Akola,467,http://www.wikidata.org/entity/Q520510,DISTRICT,state:27,no,True +district:468,Amravati,468,http://www.wikidata.org/entity/Q1771774,DISTRICT,state:27,no,True +district:469,Aurangabad,469,http://www.wikidata.org/entity/Q592942,DISTRICT,state:27,no,True +district:470,Beed,470,http://www.wikidata.org/entity/Q814037,DISTRICT,state:27,no,True +district:471,Bhandara,471,http://www.wikidata.org/entity/Q1813857,DISTRICT,state:27,no,True +district:472,Buldhana,472,http://www.wikidata.org/entity/Q47929,DISTRICT,state:27,no,True +district:473,Chandrapur,473,http://www.wikidata.org/entity/Q1797274,DISTRICT,state:27,no,True +district:474,Dhule,474,http://www.wikidata.org/entity/Q1797383,DISTRICT,state:27,no,True +district:475,Gadchiroli,475,http://www.wikidata.org/entity/Q1804847,DISTRICT,state:27,no,True +district:476,Gondia,476,http://www.wikidata.org/entity/Q1917227,DISTRICT,state:27,no,True +district:477,Hingoli,477,http://www.wikidata.org/entity/Q2087615,DISTRICT,state:27,no,True +district:478,Jalgaon,478,http://www.wikidata.org/entity/Q1797291,DISTRICT,state:27,no,True +district:479,Jalna,479,http://www.wikidata.org/entity/Q1804863,DISTRICT,state:27,no,True +district:480,Kolhapur,480,http://www.wikidata.org/entity/Q1797312,DISTRICT,state:27,no,True +district:481,Latur,481,http://www.wikidata.org/entity/Q1948713,DISTRICT,state:27,no,True +district:482,Mumbai City,482,http://www.wikidata.org/entity/Q2341660,DISTRICT,state:27,no,True +district:483,Mumbai Suburban,483,http://www.wikidata.org/entity/Q2085374,DISTRICT,state:27,no,True +district:484,Nagpur,484,http://www.wikidata.org/entity/Q1797367,DISTRICT,state:27,no,True +district:485,Nanded,485,http://www.wikidata.org/entity/Q692389,DISTRICT,state:27,no,True +district:486,Nandurbar,486,http://www.wikidata.org/entity/Q1623525,DISTRICT,state:27,no,True +district:487,Nashik,487,http://www.wikidata.org/entity/Q1797269,DISTRICT,state:27,no,True +district:488,Dharashiv,488,http://www.wikidata.org/entity/Q1647186,DISTRICT,state:27,no,True +district:489,Parbhani,489,http://www.wikidata.org/entity/Q1797389,DISTRICT,state:27,no,True +district:490,Pune,490,http://www.wikidata.org/entity/Q1797336,DISTRICT,state:27,no,True +district:491,Raigad,491,http://www.wikidata.org/entity/Q2019683,DISTRICT,state:27,no,True +district:492,Ratnagiri,492,http://www.wikidata.org/entity/Q1771768,DISTRICT,state:27,no,True +district:493,Sangli,493,http://www.wikidata.org/entity/Q1425060,DISTRICT,state:27,no,True +district:494,Satara,494,http://www.wikidata.org/entity/Q1135612,DISTRICT,state:27,no,True +district:495,Sindhudurg,495,http://www.wikidata.org/entity/Q768332,DISTRICT,state:27,no,True +district:496,Solapur,496,http://www.wikidata.org/entity/Q1797263,DISTRICT,state:27,no,True +district:497,Thane,497,http://www.wikidata.org/entity/Q943099,DISTRICT,state:27,no,True +district:498,Wardha,498,http://www.wikidata.org/entity/Q980608,DISTRICT,state:27,no,True +district:499,Washim,499,http://www.wikidata.org/entity/Q1804858,DISTRICT,state:27,no,True +district:500,Yavatmal,500,http://www.wikidata.org/entity/Q1804852,DISTRICT,state:27,no,True +district:665,Palghar,665,http://www.wikidata.org/entity/Q18003119,DISTRICT,state:27,no,True +district:502,Anantapuramu,502,http://www.wikidata.org/entity/Q15212,DISTRICT,state:28,no,True +district:503,Chittoor,503,http://www.wikidata.org/entity/Q15213,DISTRICT,state:28,no,True +district:504,YSR Kadapa,504,http://www.wikidata.org/entity/Q15342,DISTRICT,state:28,no,True +district:505,East Godavari,505,http://www.wikidata.org/entity/Q15338,DISTRICT,state:28,no,True +district:506,Guntur,506,http://www.wikidata.org/entity/Q15341,DISTRICT,state:28,no,True +district:510,Krishna,510,http://www.wikidata.org/entity/Q15382,DISTRICT,state:28,no,True +district:511,Kurnool,511,http://www.wikidata.org/entity/Q15381,DISTRICT,state:28,no,True +district:515,Sri Potti Sri Ramulu Nellore,515,http://www.wikidata.org/entity/Q15383,DISTRICT,state:28,no,True +district:517,Prakasam,517,http://www.wikidata.org/entity/Q15390,DISTRICT,state:28,no,True +district:519,Srikakulam,519,http://www.wikidata.org/entity/Q15395,DISTRICT,state:28,no,True +district:520,Visakhapatnam,520,http://www.wikidata.org/entity/Q15394,DISTRICT,state:28,no,True +district:521,Vizianagaram,521,http://www.wikidata.org/entity/Q15392,DISTRICT,state:28,no,True +district:523,West Godavari,523,http://www.wikidata.org/entity/Q15404,DISTRICT,state:28,no,True +district:743,Parvathipuram Manyam,743,http://www.wikidata.org/entity/Q110714856,DISTRICT,state:28,no,True +district:744,Anakapalli,744,http://www.wikidata.org/entity/Q110714857,DISTRICT,state:28,no,True +district:745,Alluri Sitharama Raju,745,http://www.wikidata.org/entity/Q110714850,DISTRICT,state:28,no,True +district:746,Kakinada,746,http://www.wikidata.org/entity/Q110714860,DISTRICT,state:28,no,True +district:747,Dr. B.R. Ambedkar Konaseema,747,http://www.wikidata.org/entity/Q110714859,DISTRICT,state:28,no,True +district:748,Eluru,748,http://www.wikidata.org/entity/Q110714851,DISTRICT,state:28,no,True +district:749,NTR,749,http://www.wikidata.org/entity/Q110876763,DISTRICT,state:28,no,True +district:750,Bapatla,750,http://www.wikidata.org/entity/Q110876712,DISTRICT,state:28,no,True +district:751,Palnadu,751,http://www.wikidata.org/entity/Q110714862,DISTRICT,state:28,no,True +district:752,Tirupati,752,http://www.wikidata.org/entity/Q110714853,DISTRICT,state:28,no,True +district:753,Annamayya,753,http://www.wikidata.org/entity/Q110714854,DISTRICT,state:28,no,True +district:754,Sri Sathya Sai,754,http://www.wikidata.org/entity/Q110714863,DISTRICT,state:28,no,True +district:755,Nandyal,755,http://www.wikidata.org/entity/Q110714861,DISTRICT,state:28,no,True +district:524,Bagalkot,524,http://www.wikidata.org/entity/Q1910231,DISTRICT,state:29,no,True +district:525,Bengaluru Urban,525,http://www.wikidata.org/entity/Q806463,DISTRICT,state:29,no,True +district:526,Bengaluru North,526,http://www.wikidata.org/entity/Q806464,DISTRICT,state:29,no,True +district:527,Belagavi,527,http://www.wikidata.org/entity/Q815464,DISTRICT,state:29,no,True +district:528,Ballari,528,http://www.wikidata.org/entity/Q1791926,DISTRICT,state:29,no,True +district:529,Bidar,529,http://www.wikidata.org/entity/Q1790568,DISTRICT,state:29,no,True +district:530,Vijaypura,530,http://www.wikidata.org/entity/Q83108,DISTRICT,state:29,no,True +district:531,Chamarajanagar,531,http://www.wikidata.org/entity/Q862912,DISTRICT,state:29,no,True +district:532,Chikmagalur,532,http://www.wikidata.org/entity/Q743077,DISTRICT,state:29,no,True +district:533,Chitradurga,533,http://www.wikidata.org/entity/Q165264,DISTRICT,state:29,no,True +district:534,Dakshina Kannada,534,http://www.wikidata.org/entity/Q950571,DISTRICT,state:29,no,True +district:535,Davanagere,535,http://www.wikidata.org/entity/Q1863214,DISTRICT,state:29,no,True +district:536,Dharwad,536,http://www.wikidata.org/entity/Q1790904,DISTRICT,state:29,no,True +district:537,Gadag,537,http://www.wikidata.org/entity/Q2353931,DISTRICT,state:29,no,True +district:538,Kalaburgi,538,http://www.wikidata.org/entity/Q2641873,DISTRICT,state:29,no,True +district:539,Hassan,539,http://www.wikidata.org/entity/Q956732,DISTRICT,state:29,no,True +district:540,Haveri,540,http://www.wikidata.org/entity/Q765481,DISTRICT,state:29,no,True +district:541,Kodagu,541,http://www.wikidata.org/entity/Q1553185,DISTRICT,state:29,no,True +district:542,Kolar,542,http://www.wikidata.org/entity/Q2509866,DISTRICT,state:29,no,True +district:543,Koppal,543,http://www.wikidata.org/entity/Q956387,DISTRICT,state:29,no,True +district:544,Mandya,544,http://www.wikidata.org/entity/Q2768290,DISTRICT,state:29,no,True +district:545,Mysuru,545,http://www.wikidata.org/entity/Q591781,DISTRICT,state:29,no,True +district:546,Raichur,546,http://www.wikidata.org/entity/Q1430830,DISTRICT,state:29,no,True +district:547,Shimoga,547,http://www.wikidata.org/entity/Q2981389,DISTRICT,state:29,no,True +district:548,Tumkur,548,http://www.wikidata.org/entity/Q1301635,DISTRICT,state:29,no,True +district:549,Udupi,549,http://www.wikidata.org/entity/Q1483337,DISTRICT,state:29,no,True +district:550,Uttara Kannada,550,http://www.wikidata.org/entity/Q579205,DISTRICT,state:29,no,True +district:630,Chikkaballapura,630,http://www.wikidata.org/entity/Q1072629,DISTRICT,state:29,no,True +district:631,Bengaluru South,631,http://www.wikidata.org/entity/Q427679,DISTRICT,state:29,no,True +district:635,Yadgir,635,http://www.wikidata.org/entity/Q1786949,DISTRICT,state:29,no,True +district:738,Vijayanagara,738,http://www.wikidata.org/entity/Q104876850,DISTRICT,state:29,no,True +district:27,Amritsar,27,http://www.wikidata.org/entity/Q202822,DISTRICT,state:3,no,True +district:28,Bathinda,28,http://www.wikidata.org/entity/Q172488,DISTRICT,state:3,no,True +district:29,Faridkot,29,http://www.wikidata.org/entity/Q172494,DISTRICT,state:3,no,True +district:30,Fatehgarh Sahib,30,http://www.wikidata.org/entity/Q172485,DISTRICT,state:3,no,True +district:31,Firozpur,31,http://www.wikidata.org/entity/Q172385,DISTRICT,state:3,no,True +district:32,Gurdaspur,32,http://www.wikidata.org/entity/Q146708,DISTRICT,state:3,no,True +district:33,Hoshiarpur,33,http://www.wikidata.org/entity/Q304800,DISTRICT,state:3,no,True +district:34,Jalandhar,34,http://www.wikidata.org/entity/Q1817425,DISTRICT,state:3,no,True +district:35,Kapurthala,35,http://www.wikidata.org/entity/Q172363,DISTRICT,state:3,no,True +district:36,Ludhiana,36,http://www.wikidata.org/entity/Q172482,DISTRICT,state:3,no,True +district:37,Mansa,37,http://www.wikidata.org/entity/Q172387,DISTRICT,state:3,no,True +district:38,Moga,38,http://www.wikidata.org/entity/Q1946896,DISTRICT,state:3,no,True +district:39,Sri Muktsar Sahib,39,http://www.wikidata.org/entity/Q1947359,DISTRICT,state:3,no,True +district:40,Shaheed Bhagat Singh Nagar,40,http://www.wikidata.org/entity/Q202710,DISTRICT,state:3,no,True +district:41,Patiala,41,http://www.wikidata.org/entity/Q172391,DISTRICT,state:3,no,True +district:42,Rupnagar,42,http://www.wikidata.org/entity/Q196508,DISTRICT,state:3,no,True +district:43,Sangrur,43,http://www.wikidata.org/entity/Q1945515,DISTRICT,state:3,no,True +district:605,Barnala,605,http://www.wikidata.org/entity/Q2353293,DISTRICT,state:3,no,True +district:608,Sahibzada Ajit Singh Nagar,608,http://www.wikidata.org/entity/Q2037672,DISTRICT,state:3,no,True +district:609,Tarn Taran,609,http://www.wikidata.org/entity/Q2298993,DISTRICT,state:3,no,True +district:651,Fazilka,651,http://www.wikidata.org/entity/Q188702,DISTRICT,state:3,no,True +district:662,Pathankot,662,http://www.wikidata.org/entity/Q172269,DISTRICT,state:3,no,True +district:737,Malerkotla,737,http://www.wikidata.org/entity/Q107016021,DISTRICT,state:3,no,True +district:551,North Goa,551,http://www.wikidata.org/entity/Q108234,DISTRICT,state:30,no,True +district:552,South Goa,552,http://www.wikidata.org/entity/Q108244,DISTRICT,state:30,no,True +district:553,Lakshadweep,553,http://www.wikidata.org/entity/Q10784153,DISTRICT,state:31,no,True +district:554,Alappuzha,554,http://www.wikidata.org/entity/Q928959,DISTRICT,state:32,no,True +district:555,Ernakulam,555,http://www.wikidata.org/entity/Q1356097,DISTRICT,state:32,no,True +district:556,Idukki,556,http://www.wikidata.org/entity/Q301821,DISTRICT,state:32,no,True +district:557,Kannur,557,http://www.wikidata.org/entity/Q2980652,DISTRICT,state:32,no,True +district:558,Kasaragod,558,http://www.wikidata.org/entity/Q1419703,DISTRICT,state:32,no,True +district:559,Kollam,559,http://www.wikidata.org/entity/Q1356124,DISTRICT,state:32,no,True +district:560,Kottayam,560,http://www.wikidata.org/entity/Q1353354,DISTRICT,state:32,no,True +district:561,Kozhikode,561,http://www.wikidata.org/entity/Q1142979,DISTRICT,state:32,no,True +district:562,Malappuram,562,http://www.wikidata.org/entity/Q1030918,DISTRICT,state:32,no,True +district:563,Palakkad,563,http://www.wikidata.org/entity/Q1535742,DISTRICT,state:32,no,True +district:564,Pathanamthitta,564,http://www.wikidata.org/entity/Q634935,DISTRICT,state:32,no,True +district:565,Thiruvananthapuram,565,http://www.wikidata.org/entity/Q162612,DISTRICT,state:32,no,True +district:566,Thrissur,566,http://www.wikidata.org/entity/Q2429655,DISTRICT,state:32,no,True +district:567,Wayanad,567,http://www.wikidata.org/entity/Q1364427,DISTRICT,state:32,no,True +district:568,Chennai,568,http://www.wikidata.org/entity/Q15116,DISTRICT,state:33,no,True +district:569,Coimbatore,569,http://www.wikidata.org/entity/Q15136,DISTRICT,state:33,no,True +district:570,Cuddalore,570,http://www.wikidata.org/entity/Q15150,DISTRICT,state:33,no,True +district:571,Dharmapuri,571,http://www.wikidata.org/entity/Q15152,DISTRICT,state:33,no,True +district:572,Dindigul,572,http://www.wikidata.org/entity/Q15154,DISTRICT,state:33,no,True +district:573,Erode,573,http://www.wikidata.org/entity/Q15155,DISTRICT,state:33,no,True +district:574,Kanchipuram,574,http://www.wikidata.org/entity/Q15157,DISTRICT,state:33,no,True +district:575,Kanniyakumari,575,http://www.wikidata.org/entity/Q15158,DISTRICT,state:33,no,True +district:576,Karur,576,http://www.wikidata.org/entity/Q15182,DISTRICT,state:33,no,True +district:577,Krishnagiri,577,http://www.wikidata.org/entity/Q15183,DISTRICT,state:33,no,True +district:578,Madurai,578,http://www.wikidata.org/entity/Q15184,DISTRICT,state:33,no,True +district:579,Nagapattinam,579,http://www.wikidata.org/entity/Q15185,DISTRICT,state:33,no,True +district:580,Namakkal,580,http://www.wikidata.org/entity/Q15187,DISTRICT,state:33,no,True +district:581,Perambalur,581,http://www.wikidata.org/entity/Q15186,DISTRICT,state:33,no,True +district:582,Pudukkottai,582,http://www.wikidata.org/entity/Q15190,DISTRICT,state:33,no,True +district:583,Ramanathapuram,583,http://www.wikidata.org/entity/Q15191,DISTRICT,state:33,no,True +district:584,Salem,584,http://www.wikidata.org/entity/Q15192,DISTRICT,state:33,no,True +district:585,Sivaganga,585,http://www.wikidata.org/entity/Q15195,DISTRICT,state:33,no,True +district:586,Thanjavur,586,http://www.wikidata.org/entity/Q15194,DISTRICT,state:33,no,True +district:587,Nilgiris,587,http://www.wikidata.org/entity/Q15188,DISTRICT,state:33,no,True +district:588,Theni,588,http://www.wikidata.org/entity/Q15196,DISTRICT,state:33,no,True +district:589,Tiruvallur,589,http://www.wikidata.org/entity/Q15204,DISTRICT,state:33,no,True +district:590,Tiruvarur,590,http://www.wikidata.org/entity/Q15197,DISTRICT,state:33,no,True +district:591,Tiruchirappalli,591,http://www.wikidata.org/entity/Q15201,DISTRICT,state:33,no,True +district:592,Tirunelveli,592,http://www.wikidata.org/entity/Q15200,DISTRICT,state:33,no,True +district:593,Tiruvannamalai,593,http://www.wikidata.org/entity/Q15207,DISTRICT,state:33,no,True +district:594,Thoothukudi,594,http://www.wikidata.org/entity/Q15198,DISTRICT,state:33,no,True +district:595,Vellore,595,http://www.wikidata.org/entity/Q15206,DISTRICT,state:33,no,True +district:596,Viluppuram,596,http://www.wikidata.org/entity/Q15205,DISTRICT,state:33,no,True +district:597,Virudhunagar,597,http://www.wikidata.org/entity/Q15209,DISTRICT,state:33,no,True +district:610,Ariyalur,610,http://www.wikidata.org/entity/Q15112,DISTRICT,state:33,no,True +district:634,Tiruppur,634,http://www.wikidata.org/entity/Q15202,DISTRICT,state:33,no,True +district:729,Kallakurichi,729,http://www.wikidata.org/entity/Q60493360,DISTRICT,state:33,no,True +district:730,Chengalpattu,730,http://www.wikidata.org/entity/Q65976177,DISTRICT,state:33,no,True +district:731,Ranipet,731,http://www.wikidata.org/entity/Q66659623,DISTRICT,state:33,no,True +district:732,Tirupattur,732,http://www.wikidata.org/entity/Q66659621,DISTRICT,state:33,no,True +district:733,Tenkasi,733,http://www.wikidata.org/entity/Q75094121,DISTRICT,state:33,no,True +district:735,Mayiladuthurai,735,http://www.wikidata.org/entity/Q89918869,DISTRICT,state:33,no,True +district:598,Karaikal,598,http://www.wikidata.org/entity/Q639264,DISTRICT,state:34,no,True +district:600,Puducherry,600,http://www.wikidata.org/entity/Q984035,DISTRICT,state:34,no,True +district:602,South Andaman,602,http://www.wikidata.org/entity/Q796979,DISTRICT,state:35,no,True +district:603,Nicobar,603,http://www.wikidata.org/entity/Q797295,DISTRICT,state:35,no,True +district:632,North and Middle Andaman,632,http://www.wikidata.org/entity/Q796983,DISTRICT,state:35,no,True +district:501,Adilabad,501,http://www.wikidata.org/entity/Q15211,DISTRICT,state:36,no,True +district:507,Hyderabad,507,http://www.wikidata.org/entity/Q15340,DISTRICT,state:36,no,True +district:508,Karimnagar,508,http://www.wikidata.org/entity/Q15373,DISTRICT,state:36,no,True +district:509,Khammam,509,http://www.wikidata.org/entity/Q15371,DISTRICT,state:36,no,True +district:512,Mahabubnagar,512,http://www.wikidata.org/entity/Q15380,DISTRICT,state:36,no,True +district:513,Medak,513,http://www.wikidata.org/entity/Q15386,DISTRICT,state:36,no,True +district:514,Nalgonda,514,http://www.wikidata.org/entity/Q15384,DISTRICT,state:36,no,True +district:516,Nizamabad,516,http://www.wikidata.org/entity/Q15391,DISTRICT,state:36,no,True +district:518,Ranga Reddy,518,http://www.wikidata.org/entity/Q15388,DISTRICT,state:36,no,True +district:522,Warangal,522,http://www.wikidata.org/entity/Q15399,DISTRICT,state:36,no,True +district:680,Nirmal,680,http://www.wikidata.org/entity/Q28169750,DISTRICT,state:36,no,True +district:681,Jagtial,681,http://www.wikidata.org/entity/Q28169780,DISTRICT,state:36,no,True +district:682,Peddapalli,682,http://www.wikidata.org/entity/Q27614797,DISTRICT,state:36,no,True +district:683,Rajanna Sircilla,683,http://www.wikidata.org/entity/Q28172781,DISTRICT,state:36,no,True +district:684,Mancherial,684,http://www.wikidata.org/entity/Q28169747,DISTRICT,state:36,no,True +district:685,Kamareddy,685,http://www.wikidata.org/entity/Q27956125,DISTRICT,state:36,no,True +district:686,Hanamkonda,686,http://www.wikidata.org/entity/Q213077,DISTRICT,state:36,no,True +district:687,Jayashankar Bhupalpally,687,http://www.wikidata.org/entity/Q28169775,DISTRICT,state:36,no,True +district:688,Mahabubabad,688,http://www.wikidata.org/entity/Q28169761,DISTRICT,state:36,no,True +district:689,Jangaon,689,http://www.wikidata.org/entity/Q28170170,DISTRICT,state:36,no,True +district:690,Bhadradri Kothagudem,690,http://www.wikidata.org/entity/Q28169767,DISTRICT,state:36,no,True +district:691,Sangareddy,691,http://www.wikidata.org/entity/Q28169753,DISTRICT,state:36,no,True +district:692,Siddipet,692,http://www.wikidata.org/entity/Q28169756,DISTRICT,state:36,no,True +district:693,Wanaparthy,693,http://www.wikidata.org/entity/Q28172504,DISTRICT,state:36,no,True +district:694,Nagarkurnool,694,http://www.wikidata.org/entity/Q28169773,DISTRICT,state:36,no,True +district:695,Jogulamba Gadwal,695,http://www.wikidata.org/entity/Q27897618,DISTRICT,state:36,no,True +district:696,Suryapet,696,http://www.wikidata.org/entity/Q28169770,DISTRICT,state:36,no,True +district:697,Yadadri Bhuvanagiri,697,http://www.wikidata.org/entity/Q28169764,DISTRICT,state:36,no,True +district:698,Vikarabad,698,http://www.wikidata.org/entity/Q28170173,DISTRICT,state:36,no,True +district:699,Kumaram Bheem Asifabad,699,http://www.wikidata.org/entity/Q28170184,DISTRICT,state:36,no,True +district:700,Medchal-Malkajgiri,700,http://www.wikidata.org/entity/Q27614841,DISTRICT,state:36,no,True +district:720,Mulugu,720,http://www.wikidata.org/entity/Q61746006,DISTRICT,state:36,no,True +district:721,Narayanpet,721,http://www.wikidata.org/entity/Q61746013,DISTRICT,state:36,no,True +district:6,Kargil,6,http://www.wikidata.org/entity/Q1650798,DISTRICT,state:37,no,True +district:9,Leh,9,http://www.wikidata.org/entity/Q1921210,DISTRICT,state:37,no,True +district:463,Daman,463,http://www.wikidata.org/entity/Q1158197,DISTRICT,state:38,no,True +district:464,Diu,464,http://www.wikidata.org/entity/Q2552347,DISTRICT,state:38,no,True +district:465,Dadra and Nagar Haveli,465,http://www.wikidata.org/entity/Q46107,DISTRICT,state:38,no,True +district:44,Chandigarh,44,http://www.wikidata.org/entity/Q5071071,DISTRICT,state:4,no,True +district:45,Almora,45,http://www.wikidata.org/entity/Q1805066,DISTRICT,state:5,no,True +district:46,Bageshwar,46,http://www.wikidata.org/entity/Q1815313,DISTRICT,state:5,no,True +district:47,Chamoli,47,http://www.wikidata.org/entity/Q1797372,DISTRICT,state:5,no,True +district:48,Champawat,48,http://www.wikidata.org/entity/Q288278,DISTRICT,state:5,no,True +district:49,Dehradun,49,http://www.wikidata.org/entity/Q1815740,DISTRICT,state:5,no,True +district:50,Haridwar,50,http://www.wikidata.org/entity/Q2270438,DISTRICT,state:5,no,True +district:51,Nainital,51,http://www.wikidata.org/entity/Q1797306,DISTRICT,state:5,no,True +district:52,Pauri Garhwal,52,http://www.wikidata.org/entity/Q2085474,DISTRICT,state:5,no,True +district:53,Pithoragarh,53,http://www.wikidata.org/entity/Q1945425,DISTRICT,state:5,no,True +district:54,Rudraprayag,54,http://www.wikidata.org/entity/Q1805059,DISTRICT,state:5,no,True +district:55,Tehri Garhwal,55,http://www.wikidata.org/entity/Q1357107,DISTRICT,state:5,no,True +district:56,Udham Singh Nagar,56,http://www.wikidata.org/entity/Q1805082,DISTRICT,state:5,no,True +district:57,Uttarkashi,57,http://www.wikidata.org/entity/Q1773437,DISTRICT,state:5,no,True +district:58,Ambala,58,http://www.wikidata.org/entity/Q2086226,DISTRICT,state:6,no,True +district:59,Bhiwani,59,http://www.wikidata.org/entity/Q1852857,DISTRICT,state:6,no,True +district:60,Faridabad,60,http://www.wikidata.org/entity/Q2086173,DISTRICT,state:6,no,True +district:604,Nuh,604,http://www.wikidata.org/entity/Q2216696,DISTRICT,state:6,no,True +district:61,Fatehabad,61,http://www.wikidata.org/entity/Q2301753,DISTRICT,state:6,no,True +district:619,Palwal,619,http://www.wikidata.org/entity/Q2724926,DISTRICT,state:6,no,True +district:62,Gurugram,62,http://www.wikidata.org/entity/Q1815766,DISTRICT,state:6,no,True +district:63,Hisar,63,http://www.wikidata.org/entity/Q1815773,DISTRICT,state:6,no,True +district:64,Jhajjar,64,http://www.wikidata.org/entity/Q1948260,DISTRICT,state:6,no,True +district:65,Jind,65,http://www.wikidata.org/entity/Q268605,DISTRICT,state:6,no,True +district:66,Kaithal,66,http://www.wikidata.org/entity/Q614037,DISTRICT,state:6,no,True +district:67,Karnal,67,http://www.wikidata.org/entity/Q607915,DISTRICT,state:6,no,True +district:68,Kurukshetra,68,http://www.wikidata.org/entity/Q980118,DISTRICT,state:6,no,True +district:69,Mahendragarh,69,http://www.wikidata.org/entity/Q684019,DISTRICT,state:6,no,True +district:70,Panchkula,70,http://www.wikidata.org/entity/Q1898143,DISTRICT,state:6,no,True +district:701,Charkhi Dadri,701,http://www.wikidata.org/entity/Q28172110,DISTRICT,state:6,no,True +district:71,Panipat,71,http://www.wikidata.org/entity/Q2086163,DISTRICT,state:6,no,True +district:72,Rewari,72,http://www.wikidata.org/entity/Q2301759,DISTRICT,state:6,no,True +district:73,Rohtak,73,http://www.wikidata.org/entity/Q967388,DISTRICT,state:6,no,True +district:74,Sirsa,74,http://www.wikidata.org/entity/Q526101,DISTRICT,state:6,no,True +district:75,Sonipat,75,http://www.wikidata.org/entity/Q2241746,DISTRICT,state:6,no,True +district:76,Yamunanagar,76,http://www.wikidata.org/entity/Q1873644,DISTRICT,state:6,no,True +district:670,South East Delhi,670,http://www.wikidata.org/entity/Q25553535,DISTRICT,state:7,no,True +district:671,Shahdara,671,http://www.wikidata.org/entity/Q83486,DISTRICT,state:7,no,True +district:77,Central Delhi,77,http://www.wikidata.org/entity/Q107941,DISTRICT,state:7,no,True +district:78,East Delhi,78,http://www.wikidata.org/entity/Q107960,DISTRICT,state:7,no,True +district:79,New Delhi,79,http://www.wikidata.org/entity/Q8560886,DISTRICT,state:7,no,True +district:794,Outer North Delhi,794,http://www.wikidata.org/entity/Q140805546,DISTRICT,state:7,no,True +district:795,Old Delhi,795,http://www.wikidata.org/entity/Q140721912,DISTRICT,state:7,no,True +district:796,Central North Delhi,796,http://www.wikidata.org/entity/Q140804026,DISTRICT,state:7,no,True +district:80,North Delhi,80,http://www.wikidata.org/entity/Q693367,DISTRICT,state:7,no,True +district:81,North East Delhi,81,http://www.wikidata.org/entity/Q429329,DISTRICT,state:7,no,True +district:82,North West Delhi,82,http://www.wikidata.org/entity/Q766125,DISTRICT,state:7,no,True +district:83,South Delhi,83,http://www.wikidata.org/entity/Q2061938,DISTRICT,state:7,no,True +district:84,South West Delhi,84,http://www.wikidata.org/entity/Q2379189,DISTRICT,state:7,no,True +district:85,West Delhi,85,http://www.wikidata.org/entity/Q549807,DISTRICT,state:7,no,True +district:100,Sri Ganganagar,100,http://www.wikidata.org/entity/Q1419696,DISTRICT,state:8,no,True +district:101,Hanumangarh,101,http://www.wikidata.org/entity/Q1356112,DISTRICT,state:8,no,True +district:102,Jaipur,102,http://www.wikidata.org/entity/Q1134781,DISTRICT,state:8,no,True +district:103,Jaisalmer,103,http://www.wikidata.org/entity/Q1419708,DISTRICT,state:8,no,True +district:104,kishangarh sub division.Ajmer,104,http://www.wikidata.org/entity/Q1460832,DISTRICT,state:8,no,True +district:105,Jhalawar,105,http://www.wikidata.org/entity/Q1471417,DISTRICT,state:8,no,True +district:106,Jhunjhunu,106,http://www.wikidata.org/entity/Q1471427,DISTRICT,state:8,no,True +district:107,Jodhpur,107,http://www.wikidata.org/entity/Q1434965,DISTRICT,state:8,no,True +district:108,Karauli,108,http://www.wikidata.org/entity/Q1419668,DISTRICT,state:8,no,True +district:109,Kota,109,http://www.wikidata.org/entity/Q999432,DISTRICT,state:8,no,True +district:110,Nagaur,110,http://www.wikidata.org/entity/Q1507174,DISTRICT,state:8,no,True +district:111,Pali,111,http://www.wikidata.org/entity/Q46925,DISTRICT,state:8,no,True +district:112,Rajsamand,112,http://www.wikidata.org/entity/Q596693,DISTRICT,state:8,no,True +district:113,Sawai Madhopur,113,http://www.wikidata.org/entity/Q1507166,DISTRICT,state:8,no,True +district:114,Sikar,114,http://www.wikidata.org/entity/Q12945777,DISTRICT,state:8,no,True +district:115,Sirohi,115,http://www.wikidata.org/entity/Q205719,DISTRICT,state:8,no,True +district:116,Tonk,116,http://www.wikidata.org/entity/Q915880,DISTRICT,state:8,no,True +district:117,Udaipur,117,http://www.wikidata.org/entity/Q1321577,DISTRICT,state:8,no,True +district:629,Pratapgarh,629,http://www.wikidata.org/entity/Q1585433,DISTRICT,state:8,no,True +district:767,Deeg,767,http://www.wikidata.org/entity/Q122766908,DISTRICT,state:8,no,True +district:768,Didwana-Kuchaman,768,http://www.wikidata.org/entity/Q122971176,DISTRICT,state:8,no,True +district:769,Dudu,769,http://www.wikidata.org/entity/Q122971166,DISTRICT,state:8,no,True +district:770,Khairthal-Tijara,770,http://www.wikidata.org/entity/Q122971175,DISTRICT,state:8,no,True +district:771,Gangapur City,771,http://www.wikidata.org/entity/Q121607665,DISTRICT,state:8,no,True +district:772,Phalodi,772,http://www.wikidata.org/entity/Q122971167,DISTRICT,state:8,no,True +district:773,Neem Ka Thana,773,http://www.wikidata.org/entity/Q122212229,DISTRICT,state:8,no,True +district:774,Beawar,774,http://www.wikidata.org/entity/Q122971174,DISTRICT,state:8,no,True +district:775,Balotra,775,http://www.wikidata.org/entity/Q121335556,DISTRICT,state:8,no,True +district:776,Anupgarh,776,http://www.wikidata.org/entity/Q117230972,DISTRICT,state:8,no,True +district:777,Salumbar,777,http://www.wikidata.org/entity/Q122971173,DISTRICT,state:8,no,True +district:778,Jodhpur Gramin,778,http://www.wikidata.org/entity/Q122971177,DISTRICT,state:8,no,True +district:779,Sanchore,779,http://www.wikidata.org/entity/Q122276059,DISTRICT,state:8,no,True +district:781,Kekri,781,http://www.wikidata.org/entity/Q122971178,DISTRICT,state:8,no,True +district:782,Kotputli-Behror,782,http://www.wikidata.org/entity/Q117313512,DISTRICT,state:8,no,True +district:86,Ajmer,86,http://www.wikidata.org/entity/Q413037,DISTRICT,state:8,no,True +district:87,Alwar,87,http://www.wikidata.org/entity/Q449690,DISTRICT,state:8,no,True +district:88,Banswara,88,http://www.wikidata.org/entity/Q806969,DISTRICT,state:8,no,True +district:89,Baran,89,http://www.wikidata.org/entity/Q2329717,DISTRICT,state:8,no,True +district:90,Barmer,90,http://www.wikidata.org/entity/Q42016,DISTRICT,state:8,no,True +district:91,Bharatpur,91,http://www.wikidata.org/entity/Q854861,DISTRICT,state:8,no,True +district:92,Bhilwara,92,http://www.wikidata.org/entity/Q41991,DISTRICT,state:8,no,True +district:93,Bikaner,93,http://www.wikidata.org/entity/Q778996,DISTRICT,state:8,no,True +district:94,Bundi,94,http://www.wikidata.org/entity/Q670405,DISTRICT,state:8,no,True +district:95,Chittorgarh,95,http://www.wikidata.org/entity/Q1075011,DISTRICT,state:8,no,True +district:96,Churu,96,http://www.wikidata.org/entity/Q1090006,DISTRICT,state:8,no,True +district:97,Dausa,97,http://www.wikidata.org/entity/Q1173042,DISTRICT,state:8,no,True +district:98,Dholpur,98,http://www.wikidata.org/entity/Q1207709,DISTRICT,state:8,no,True +district:99,Dungarpur,99,http://www.wikidata.org/entity/Q1265687,DISTRICT,state:8,no,True +district:118,Agra,118,http://www.wikidata.org/entity/Q606343,DISTRICT,state:9,no,True +district:119,Aligarh,119,http://www.wikidata.org/entity/Q766918,DISTRICT,state:9,no,True +district:120,Prayagraj,120,http://www.wikidata.org/entity/Q1773426,DISTRICT,state:9,no,True +district:121,Ambedkar Nagar,121,http://www.wikidata.org/entity/Q456764,DISTRICT,state:9,no,True +district:122,Auraiya,122,http://www.wikidata.org/entity/Q1812533,DISTRICT,state:9,no,True +district:123,Azamgarh,123,http://www.wikidata.org/entity/Q793553,DISTRICT,state:9,no,True +district:124,Bagpat,124,http://www.wikidata.org/entity/Q1797363,DISTRICT,state:9,no,True +district:125,Bahraich,125,http://www.wikidata.org/entity/Q1812548,DISTRICT,state:9,no,True +district:126,Ballia,126,http://www.wikidata.org/entity/Q584644,DISTRICT,state:9,no,True +district:127,Balrampur,127,http://www.wikidata.org/entity/Q1948380,DISTRICT,state:9,no,True +district:128,Banda,128,http://www.wikidata.org/entity/Q2131759,DISTRICT,state:9,no,True +district:129,Barabanki,129,http://www.wikidata.org/entity/Q633114,DISTRICT,state:9,no,True +district:130,Bareilly,130,http://www.wikidata.org/entity/Q1797378,DISTRICT,state:9,no,True +district:131,Basti,131,http://www.wikidata.org/entity/Q715267,DISTRICT,state:9,no,True +district:132,Bijnor,132,http://www.wikidata.org/entity/Q1937865,DISTRICT,state:9,no,True +district:133,Budaun,133,http://www.wikidata.org/entity/Q1815262,DISTRICT,state:9,no,True +district:134,Bulandshahr,134,http://www.wikidata.org/entity/Q1752328,DISTRICT,state:9,no,True +district:135,Chandauli,135,http://www.wikidata.org/entity/Q2733369,DISTRICT,state:9,no,True +district:136,Chitrakoot,136,http://www.wikidata.org/entity/Q2089141,DISTRICT,state:9,no,True +district:137,Deoria,137,http://www.wikidata.org/entity/Q731746,DISTRICT,state:9,no,True +district:138,Etah,138,http://www.wikidata.org/entity/Q1773429,DISTRICT,state:9,no,True +district:139,Etawah,139,http://www.wikidata.org/entity/Q1815288,DISTRICT,state:9,no,True +district:140,Ayodhya,140,http://www.wikidata.org/entity/Q1814132,DISTRICT,state:9,no,True +district:141,Farrukhabad,141,http://www.wikidata.org/entity/Q1897251,DISTRICT,state:9,no,True +district:142,Fatehpur,142,http://www.wikidata.org/entity/Q1946829,DISTRICT,state:9,no,True +district:143,Firozabad,143,http://www.wikidata.org/entity/Q1946950,DISTRICT,state:9,no,True +district:144,Gautam Buddh Nagar,144,http://www.wikidata.org/entity/Q1785950,DISTRICT,state:9,no,True +district:145,Ghaziabad,145,http://www.wikidata.org/entity/Q1773444,DISTRICT,state:9,no,True +district:146,Ghazipur,146,http://www.wikidata.org/entity/Q1287993,DISTRICT,state:9,no,True +district:147,Gonda,147,http://www.wikidata.org/entity/Q1937857,DISTRICT,state:9,no,True +district:148,Gorakhpur,148,http://www.wikidata.org/entity/Q1144349,DISTRICT,state:9,no,True +district:149,Hamirpur,149,http://www.wikidata.org/entity/Q2019757,DISTRICT,state:9,no,True +district:150,Hardoi,150,http://www.wikidata.org/entity/Q1772822,DISTRICT,state:9,no,True +district:151,Jalaun,151,http://www.wikidata.org/entity/Q2089115,DISTRICT,state:9,no,True +district:152,Jaunpur,152,http://www.wikidata.org/entity/Q1356060,DISTRICT,state:9,no,True +district:153,Jhansi,153,http://www.wikidata.org/entity/Q1937885,DISTRICT,state:9,no,True +district:154,Amroha,154,http://www.wikidata.org/entity/Q1891677,DISTRICT,state:9,no,True +district:155,Kannauj,155,http://www.wikidata.org/entity/Q627979,DISTRICT,state:9,no,True +district:156,Kanpur Dehat,156,http://www.wikidata.org/entity/Q610612,DISTRICT,state:9,no,True +district:157,Kanpur Nagar,157,http://www.wikidata.org/entity/Q2089152,DISTRICT,state:9,no,True +district:158,Kaushambi,158,http://www.wikidata.org/entity/Q1946937,DISTRICT,state:9,no,True +district:159,Lakhimpur Kheri,159,http://www.wikidata.org/entity/Q1755447,DISTRICT,state:9,no,True +district:160,Kushinagar,160,http://www.wikidata.org/entity/Q1840355,DISTRICT,state:9,no,True +district:161,Lalitpur,161,http://www.wikidata.org/entity/Q1947336,DISTRICT,state:9,no,True +district:162,Lucknow,162,http://www.wikidata.org/entity/Q1773416,DISTRICT,state:9,no,True +district:163,Hathras,163,http://www.wikidata.org/entity/Q1814892,DISTRICT,state:9,no,True +district:164,Maharajganj,164,http://www.wikidata.org/entity/Q1356139,DISTRICT,state:9,no,True +district:165,Mahoba,165,http://www.wikidata.org/entity/Q1815322,DISTRICT,state:9,no,True +district:166,Mainpuri,166,http://www.wikidata.org/entity/Q1816657,DISTRICT,state:9,no,True +district:167,Mathura,167,http://www.wikidata.org/entity/Q1773422,DISTRICT,state:9,no,True +district:168,Mau,168,http://www.wikidata.org/entity/Q1518847,DISTRICT,state:9,no,True +district:169,Meerut,169,http://www.wikidata.org/entity/Q1764627,DISTRICT,state:9,no,True +district:170,Mirzapur,170,http://www.wikidata.org/entity/Q1143894,DISTRICT,state:9,no,True +district:171,Moradabad,171,http://www.wikidata.org/entity/Q1345006,DISTRICT,state:9,no,True +district:172,Muzaffarnagar,172,http://www.wikidata.org/entity/Q2365710,DISTRICT,state:9,no,True +district:173,Pilibhit,173,http://www.wikidata.org/entity/Q2980705,DISTRICT,state:9,no,True +district:174,Pratapgarh,174,http://www.wikidata.org/entity/Q1473962,DISTRICT,state:9,no,True +district:175,Raebareli,175,http://www.wikidata.org/entity/Q1321157,DISTRICT,state:9,no,True +district:176,Rampur,176,http://www.wikidata.org/entity/Q1815331,DISTRICT,state:9,no,True +district:177,Saharanpur,177,http://www.wikidata.org/entity/Q1797326,DISTRICT,state:9,no,True +district:178,Sant Kabir Nagar,178,http://www.wikidata.org/entity/Q1945445,DISTRICT,state:9,no,True +district:179,Bhadohi,179,http://www.wikidata.org/entity/Q127533,DISTRICT,state:9,no,True +district:180,Shahjahanpur,180,http://www.wikidata.org/entity/Q1812557,DISTRICT,state:9,no,True +district:181,Shravasti,181,http://www.wikidata.org/entity/Q1945458,DISTRICT,state:9,no,True +district:182,Siddharthnagar,182,http://www.wikidata.org/entity/Q1815339,DISTRICT,state:9,no,True +district:183,Sitapur,183,http://www.wikidata.org/entity/Q1812539,DISTRICT,state:9,no,True +district:184,Sonbhadra,184,http://www.wikidata.org/entity/Q607798,DISTRICT,state:9,no,True +district:185,Sultanpur,185,http://www.wikidata.org/entity/Q1356154,DISTRICT,state:9,no,True +district:186,Unnao,186,http://www.wikidata.org/entity/Q1937875,DISTRICT,state:9,no,True +district:187,Varanasi,187,http://www.wikidata.org/entity/Q1321140,DISTRICT,state:9,no,True +district:633,Kasganj,633,http://www.wikidata.org/entity/Q890800,DISTRICT,state:9,no,True +district:640,Amethi,640,http://www.wikidata.org/entity/Q1071494,DISTRICT,state:9,no,True +district:659,Sambhal,659,http://www.wikidata.org/entity/Q3000436,DISTRICT,state:9,no,True +district:660,Shamli,660,http://www.wikidata.org/entity/Q2999938,DISTRICT,state:9,no,True +district:661,Hapur,661,http://www.wikidata.org/entity/Q5653340,DISTRICT,state:9,no,True diff --git a/api/services/metadata_export/contracts/licenses.csv b/api/services/metadata_export/contracts/licenses.csv new file mode 100644 index 0000000..cba256f --- /dev/null +++ b/api/services/metadata_export/contracts/licenses.csv @@ -0,0 +1,61 @@ +key,label,code,uri,is_open,visible_on_dataspace,alt_codes +GODL-India,Government Open Data License – India,GODL-India,https://civicdataspace.in/id/licence/godl-india,True,yes,dataspace_enum=GOVERNMENT_OPEN_DATA_LICENSE +CC-PDM-1.0,Creative Commons Public Domain Mark 1.0 Universal,CC-PDM-1.0,https://spdx.org/licenses/CC-PDM-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-PDM-1.0.json +CC-BY-1.0,Creative Commons Attribution 1.0 Generic,CC-BY-1.0,https://spdx.org/licenses/CC-BY-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-1.0.json +CC-BY-2.0,Creative Commons Attribution 2.0 Generic,CC-BY-2.0,https://spdx.org/licenses/CC-BY-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-2.0.json +CC-BY-2.5-AU,Creative Commons Attribution 2.5 Australia,CC-BY-2.5-AU,https://spdx.org/licenses/CC-BY-2.5-AU,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-2.5-AU.json +CC-BY-2.5,Creative Commons Attribution 2.5 Generic,CC-BY-2.5,https://spdx.org/licenses/CC-BY-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-2.5.json +CC-BY-3.0-AU,Creative Commons Attribution 3.0 Australia,CC-BY-3.0-AU,https://spdx.org/licenses/CC-BY-3.0-AU,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-AU.json +CC-BY-3.0-AT,Creative Commons Attribution 3.0 Austria,CC-BY-3.0-AT,https://spdx.org/licenses/CC-BY-3.0-AT,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-AT.json +CC-BY-3.0-DE,Creative Commons Attribution 3.0 Germany,CC-BY-3.0-DE,https://spdx.org/licenses/CC-BY-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-DE.json +CC-BY-3.0-IGO,Creative Commons Attribution 3.0 IGO,CC-BY-3.0-IGO,https://spdx.org/licenses/CC-BY-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-IGO.json +CC-BY-3.0-NL,Creative Commons Attribution 3.0 Netherlands,CC-BY-3.0-NL,https://spdx.org/licenses/CC-BY-3.0-NL,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-NL.json +CC-BY-3.0-US,Creative Commons Attribution 3.0 United States,CC-BY-3.0-US,https://spdx.org/licenses/CC-BY-3.0-US,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0-US.json +CC-BY-3.0,Creative Commons Attribution 3.0 Unported,CC-BY-3.0,https://spdx.org/licenses/CC-BY-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-3.0.json +CC-BY-4.0,Creative Commons Attribution 4.0,CC-BY-4.0,http://creativecommons.org/licenses/by/4.0/,True,yes,authority_label=Creative Commons Attribution 4.0 International|dataspace_enum=CC_BY_4_0_ATTRIBUTION|spdx_detail=https://spdx.org/licenses/CC-BY-4.0.json +CC-BY-ND-1.0,Creative Commons Attribution No Derivatives 1.0 Generic,CC-BY-ND-1.0,https://spdx.org/licenses/CC-BY-ND-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-1.0.json +CC-BY-ND-2.0,Creative Commons Attribution No Derivatives 2.0 Generic,CC-BY-ND-2.0,https://spdx.org/licenses/CC-BY-ND-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-2.0.json +CC-BY-ND-2.5,Creative Commons Attribution No Derivatives 2.5 Generic,CC-BY-ND-2.5,https://spdx.org/licenses/CC-BY-ND-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-2.5.json +CC-BY-ND-3.0-DE,Creative Commons Attribution No Derivatives 3.0 Germany,CC-BY-ND-3.0-DE,https://spdx.org/licenses/CC-BY-ND-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-3.0-DE.json +CC-BY-ND-3.0,Creative Commons Attribution No Derivatives 3.0 Unported,CC-BY-ND-3.0,https://spdx.org/licenses/CC-BY-ND-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-3.0.json +CC-BY-ND-4.0,Creative Commons Attribution No Derivatives 4.0 International,CC-BY-ND-4.0,https://spdx.org/licenses/CC-BY-ND-4.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-ND-4.0.json +CC-BY-NC-1.0,Creative Commons Attribution Non Commercial 1.0 Generic,CC-BY-NC-1.0,https://spdx.org/licenses/CC-BY-NC-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-1.0.json +CC-BY-NC-2.0,Creative Commons Attribution Non Commercial 2.0 Generic,CC-BY-NC-2.0,https://spdx.org/licenses/CC-BY-NC-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-2.0.json +CC-BY-NC-2.5,Creative Commons Attribution Non Commercial 2.5 Generic,CC-BY-NC-2.5,https://spdx.org/licenses/CC-BY-NC-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-2.5.json +CC-BY-NC-3.0-DE,Creative Commons Attribution Non Commercial 3.0 Germany,CC-BY-NC-3.0-DE,https://spdx.org/licenses/CC-BY-NC-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-3.0-DE.json +CC-BY-NC-3.0-IGO,Creative Commons Attribution Non Commercial 3.0 IGO,CC-BY-NC-3.0-IGO,https://spdx.org/licenses/CC-BY-NC-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-3.0-IGO.json +CC-BY-NC-3.0,Creative Commons Attribution Non Commercial 3.0 Unported,CC-BY-NC-3.0,https://spdx.org/licenses/CC-BY-NC-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-3.0.json +CC-BY-NC-4.0,Creative Commons Attribution Non Commercial 4.0 International,CC-BY-NC-4.0,http://creativecommons.org/licenses/by-nc/4.0/,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-4.0.json +CC-BY-NC-ND-1.0,Creative Commons Attribution Non Commercial No Derivatives 1.0 Generic,CC-BY-NC-ND-1.0,https://spdx.org/licenses/CC-BY-NC-ND-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-1.0.json +CC-BY-NC-ND-2.0,Creative Commons Attribution Non Commercial No Derivatives 2.0 Generic,CC-BY-NC-ND-2.0,https://spdx.org/licenses/CC-BY-NC-ND-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-2.0.json +CC-BY-NC-ND-2.5,Creative Commons Attribution Non Commercial No Derivatives 2.5 Generic,CC-BY-NC-ND-2.5,https://spdx.org/licenses/CC-BY-NC-ND-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-2.5.json +CC-BY-NC-ND-3.0-DE,Creative Commons Attribution Non Commercial No Derivatives 3.0 Germany,CC-BY-NC-ND-3.0-DE,https://spdx.org/licenses/CC-BY-NC-ND-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-3.0-DE.json +CC-BY-NC-ND-3.0-IGO,Creative Commons Attribution Non Commercial No Derivatives 3.0 IGO,CC-BY-NC-ND-3.0-IGO,https://spdx.org/licenses/CC-BY-NC-ND-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-3.0-IGO.json +CC-BY-NC-ND-3.0,Creative Commons Attribution Non Commercial No Derivatives 3.0 Unported,CC-BY-NC-ND-3.0,https://spdx.org/licenses/CC-BY-NC-ND-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-3.0.json +CC-BY-NC-ND-4.0,Creative Commons Attribution Non Commercial No Derivatives 4.0 International,CC-BY-NC-ND-4.0,https://spdx.org/licenses/CC-BY-NC-ND-4.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-ND-4.0.json +CC-BY-NC-SA-1.0,Creative Commons Attribution Non Commercial Share Alike 1.0 Generic,CC-BY-NC-SA-1.0,https://spdx.org/licenses/CC-BY-NC-SA-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-1.0.json +CC-BY-NC-SA-2.0-UK,Creative Commons Attribution Non Commercial Share Alike 2.0 England and Wales,CC-BY-NC-SA-2.0-UK,https://spdx.org/licenses/CC-BY-NC-SA-2.0-UK,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.0-UK.json +CC-BY-NC-SA-2.0,Creative Commons Attribution Non Commercial Share Alike 2.0 Generic,CC-BY-NC-SA-2.0,https://spdx.org/licenses/CC-BY-NC-SA-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.0.json +CC-BY-NC-SA-2.0-DE,Creative Commons Attribution Non Commercial Share Alike 2.0 Germany,CC-BY-NC-SA-2.0-DE,https://spdx.org/licenses/CC-BY-NC-SA-2.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.0-DE.json +CC-BY-NC-SA-2.5,Creative Commons Attribution Non Commercial Share Alike 2.5 Generic,CC-BY-NC-SA-2.5,https://spdx.org/licenses/CC-BY-NC-SA-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.5.json +CC-BY-NC-SA-3.0-DE,Creative Commons Attribution Non Commercial Share Alike 3.0 Germany,CC-BY-NC-SA-3.0-DE,https://spdx.org/licenses/CC-BY-NC-SA-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-3.0-DE.json +CC-BY-NC-SA-3.0-IGO,Creative Commons Attribution Non Commercial Share Alike 3.0 IGO,CC-BY-NC-SA-3.0-IGO,https://spdx.org/licenses/CC-BY-NC-SA-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-3.0-IGO.json +CC-BY-NC-SA-3.0,Creative Commons Attribution Non Commercial Share Alike 3.0 Unported,CC-BY-NC-SA-3.0,https://spdx.org/licenses/CC-BY-NC-SA-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-3.0.json +CC-BY-NC-SA-4.0,Creative Commons Attribution Non Commercial Share Alike 4.0 International,CC-BY-NC-SA-4.0,https://spdx.org/licenses/CC-BY-NC-SA-4.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-4.0.json +CC-BY-SA-1.0,Creative Commons Attribution Share Alike 1.0 Generic,CC-BY-SA-1.0,https://spdx.org/licenses/CC-BY-SA-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-1.0.json +CC-BY-SA-2.0-UK,Creative Commons Attribution Share Alike 2.0 England and Wales,CC-BY-SA-2.0-UK,https://spdx.org/licenses/CC-BY-SA-2.0-UK,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-2.0-UK.json +CC-BY-SA-2.0,Creative Commons Attribution Share Alike 2.0 Generic,CC-BY-SA-2.0,https://spdx.org/licenses/CC-BY-SA-2.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-2.0.json +CC-BY-SA-2.1-JP,Creative Commons Attribution Share Alike 2.1 Japan,CC-BY-SA-2.1-JP,https://spdx.org/licenses/CC-BY-SA-2.1-JP,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-2.1-JP.json +CC-BY-SA-2.5,Creative Commons Attribution Share Alike 2.5 Generic,CC-BY-SA-2.5,https://spdx.org/licenses/CC-BY-SA-2.5,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-2.5.json +CC-BY-SA-3.0-AT,Creative Commons Attribution Share Alike 3.0 Austria,CC-BY-SA-3.0-AT,https://spdx.org/licenses/CC-BY-SA-3.0-AT,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-3.0-AT.json +CC-BY-SA-3.0-DE,Creative Commons Attribution Share Alike 3.0 Germany,CC-BY-SA-3.0-DE,https://spdx.org/licenses/CC-BY-SA-3.0-DE,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-3.0-DE.json +CC-BY-SA-3.0,Creative Commons Attribution Share Alike 3.0 Unported,CC-BY-SA-3.0,https://spdx.org/licenses/CC-BY-SA-3.0,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-3.0.json +CC-BY-NC-SA-2.0-FR,Creative Commons Attribution-NonCommercial-ShareAlike 2.0 France,CC-BY-NC-SA-2.0-FR,https://spdx.org/licenses/CC-BY-NC-SA-2.0-FR,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-NC-SA-2.0-FR.json +CC-BY-SA-3.0-IGO,Creative Commons Attribution-ShareAlike 3.0 IGO,CC-BY-SA-3.0-IGO,https://spdx.org/licenses/CC-BY-SA-3.0-IGO,False,no,spdx_detail=https://spdx.org/licenses/CC-BY-SA-3.0-IGO.json +CC-BY-SA-4.0,Creative Commons Attribution-ShareAlike 4.0,CC-BY-SA-4.0,http://creativecommons.org/licenses/by-sa/4.0/,True,yes,authority_label=Creative Commons Attribution Share Alike 4.0 International|dataspace_enum=CC_BY_SA_4_0_ATTRIBUTION_SHARE_ALIKE|spdx_detail=https://spdx.org/licenses/CC-BY-SA-4.0.json +CC-PDDC,Creative Commons Public Domain Dedication and Certification,CC-PDDC,https://spdx.org/licenses/CC-PDDC,False,no,spdx_detail=https://spdx.org/licenses/CC-PDDC.json +CC-SA-1.0,Creative Commons Share Alike 1.0 Generic,CC-SA-1.0,https://spdx.org/licenses/CC-SA-1.0,False,no,spdx_detail=https://spdx.org/licenses/CC-SA-1.0.json +CC0-1.0,Creative Commons Zero v1.0 Universal,CC0-1.0,https://creativecommons.org/publicdomain/zero/1.0/,True,yes,spdx_detail=https://spdx.org/licenses/CC0-1.0.json +ODC-By-1.0,Open Data Commons Attribution License 1.0,ODC-By-1.0,https://opendatacommons.org/licenses/by/1-0/,True,yes,authority_label=Open Data Commons Attribution License v1.0|dataspace_enum=OPEN_DATA_COMMONS_BY_ATTRIBUTION|spdx_detail=https://spdx.org/licenses/ODC-By-1.0.json +PDDL-1.0,Open Data Commons Public Domain Dedication & License 1.0,PDDL-1.0,https://opendatacommons.org/licenses/pddl/1-0/,True,yes,spdx_detail=https://spdx.org/licenses/PDDL-1.0.json +ODbL-1.0,Open Database License 1.0,ODbL-1.0,https://opendatacommons.org/licenses/odbl/1-0/,True,yes,authority_label=Open Data Commons Open Database License v1.0|dataspace_enum=OPEN_DATABASE_LICENSE|spdx_detail=https://spdx.org/licenses/ODbL-1.0.json diff --git a/api/services/metadata_export/contracts/sectors.csv b/api/services/metadata_export/contracts/sectors.csv new file mode 100644 index 0000000..747ebb4 --- /dev/null +++ b/api/services/metadata_export/contracts/sectors.csv @@ -0,0 +1,22 @@ +key,label,code,uri,is_open,visible_on_dataspace,alt_codes +cdl:child-rights,Child Rights,child-rights,https://civicdataspace.in/id/sector/child-rights,,yes,dac_crs=16010|eu_data_theme=SOCI|eu_match_type=broadMatch|sdg= +cdl:climate-action,Climate Action,climate-action,https://civicdataspace.in/id/sector/climate-action,,yes,dac_crs=|eu_data_theme=ENVI|eu_match_type=broadMatch|sdg=13 +cdl:coastal,Coastal,coastal,https://civicdataspace.in/id/sector/coastal,,yes,dac_crs=|eu_data_theme=ENVI|eu_match_type=broadMatch|sdg=14 +cdl:disaster-risk-reduction,Disaster Risk Reduction,disaster-risk-reduction,https://civicdataspace.in/id/sector/disaster-risk-reduction,,yes,dac_crs=74020|eu_data_theme=ENVI|eu_match_type=broadMatch|sdg= +cdl:gender,Gender,gender,https://civicdataspace.in/id/sector/gender,,yes,dac_crs=15170|eu_data_theme=SOCI|eu_match_type=broadMatch|sdg=5 +cdl:law-and-justice,Law and Justice,law-and-justice,https://civicdataspace.in/id/sector/law-and-justice,,yes,dac_crs=|eu_data_theme=JUST|eu_match_type=closeMatch|sdg=16 +cdl:public-finance,Public Finance,public-finance,https://civicdataspace.in/id/sector/public-finance,,yes,dac_crs=|eu_data_theme=GOVE|eu_match_type=closeMatch|sdg= +cdl:urban-development,Urban Development,urban-development,https://civicdataspace.in/id/sector/urban-development,,yes,dac_crs=|eu_data_theme=REGI|eu_match_type=closeMatch|sdg=11 +eu-theme:AGRI,"Agriculture, fisheries, forestry and food",AGRI,http://publications.europa.eu/resource/authority/data-theme/AGRI,,no, +eu-theme:ECON,Economy and finance,ECON,http://publications.europa.eu/resource/authority/data-theme/ECON,,no, +eu-theme:EDUC,"Education, culture and sport",EDUC,http://publications.europa.eu/resource/authority/data-theme/EDUC,,no, +eu-theme:ENER,Energy,ENER,http://publications.europa.eu/resource/authority/data-theme/ENER,,no, +eu-theme:ENVI,Environment,ENVI,http://publications.europa.eu/resource/authority/data-theme/ENVI,,no, +eu-theme:GOVE,Government and public sector,GOVE,http://publications.europa.eu/resource/authority/data-theme/GOVE,,no, +eu-theme:HEAL,Health,HEAL,http://publications.europa.eu/resource/authority/data-theme/HEAL,,no, +eu-theme:INTR,International issues,INTR,http://publications.europa.eu/resource/authority/data-theme/INTR,,no, +eu-theme:JUST,"Justice, legal system and public safety",JUST,http://publications.europa.eu/resource/authority/data-theme/JUST,,no, +eu-theme:SOCI,Population and society,SOCI,http://publications.europa.eu/resource/authority/data-theme/SOCI,,no, +eu-theme:REGI,Regions and cities,REGI,http://publications.europa.eu/resource/authority/data-theme/REGI,,no, +eu-theme:TECH,Science and technology,TECH,http://publications.europa.eu/resource/authority/data-theme/TECH,,no, +eu-theme:TRAN,Transport,TRAN,http://publications.europa.eu/resource/authority/data-theme/TRAN,,no, diff --git a/api/services/metadata_export/crosswalk.py b/api/services/metadata_export/crosswalk.py new file mode 100644 index 0000000..7a43eec --- /dev/null +++ b/api/services/metadata_export/crosswalk.py @@ -0,0 +1,206 @@ +"""Standards crosswalk engine. + +Nothing here knows the name of a standard: every property, node type, context +and obligation comes from ``contracts/crosswalk.json`` (owned by this repo; see +``contracts/README.md``). Add a standard to that file and this module exports +it unchanged. The record it reads is a plain dict keyed by the contract's +``dataspace_field`` names, produced by ``adapter.py``. +""" + +from __future__ import annotations + +import json +from functools import lru_cache +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +CONTRACT_PATH = Path(__file__).resolve().parent / "contracts" / "crosswalk.json" + +# Value types that must be a URI on the way out; a bare name is reported. +MUST_RESOLVE = {"uri", "concept", "location"} +# Value types written as {"@id": …} when the standard's uri_style is "node". +REFERENCE_TYPES = {"uri", "concept", "location", "agent"} +# FOAF agent classes -> schema.org classes, for standards whose uri_style is not "node". +SCHEMA_AGENT_TYPES = { + "foaf:Organization": "Organization", + "foaf:Person": "Person", + "foaf:Agent": "Organization", +} + +EMPTY = (None, "", [], {}) + + +def _is_uri(value: Any) -> bool: + return isinstance(value, str) and (value.startswith("http://") or value.startswith("https://")) + + +class Crosswalk: + def __init__(self, data: dict): + self.data = data + concepts = data.get("concepts", []) + # The contract ships concepts as a list (each with a "key") or a dict keyed by concept. + if isinstance(concepts, dict): + self.concepts: Dict[str, dict] = {k: dict(v, key=k) for k, v in concepts.items()} + else: + self.concepts = {c["key"]: c for c in concepts} + self.standards: Dict[str, dict] = data["standards"] + + @classmethod + @lru_cache(maxsize=1) + def load(cls, path: Optional[str] = None) -> "Crosswalk": + with open(path or CONTRACT_PATH, encoding="utf-8") as fh: + return cls(json.load(fh)) + + # ------------------------------------------------------------------ info + def standard_ids(self) -> List[str]: + return list(self.standards.keys()) + + def standard(self, sid: str) -> dict: + try: + return self.standards[sid] + except KeyError as exc: + raise KeyError(f"Unknown standard '{sid}'. Known: {', '.join(self.standards)}") from exc + + def gaps(self, sid: str) -> List[dict]: + """Concepts this standard cannot carry.""" + return list(self.standard(sid).get("gaps", [])) + + # ---------------------------------------------------------------- export + def export(self, record: dict, sid: str) -> dict: + doc, _ = self.export_with_report(record, sid) + return doc + + def export_with_report(self, record: dict, sid: str) -> Tuple[dict, dict]: + return self._export(record, self.standard(sid)) + + def _export(self, record: dict, std: dict) -> Tuple[dict, dict]: + node_types = std["node_types"] + uri_style = std.get("uri_style", "node") + + doc: Dict[str, Any] = {"@context": std["context"]} + if "dataset" in node_types: + doc["@type"] = node_types["dataset"] + + report: Dict[str, list] = { + "dropped": [], # platform has a value, standard has no binding + "unresolved": [], # needs a URI, got a name string + "missing_mandatory": [], # standard wants it, platform had nothing + } + + # Entries nested under another property (a period, a vCard) are + # collected per parent and attached once at the end. + nested: Dict[Tuple[str, str], Dict[str, Any]] = {} + by_node: Dict[str, List[dict]] = {} + for entry in std["export"]: + by_node.setdefault(entry["node"], []).append(entry) + + # --- dataset-level, plus anything nested under it ------------------- + for entry in by_node.get("dataset", []) + by_node.get("period", []): + value = self._platform_value(record, entry) + if value in EMPTY: + if entry["obligation"] == "mandatory": + report["missing_mandatory"].append(entry["concept"]) + continue + rendered = self._render(value, entry, uri_style, report) + if entry.get("parent_property"): + nested.setdefault(("dataset", entry["parent_property"]), {})[ + entry["property"] + ] = rendered + else: + doc[entry["property"]] = rendered + + for (owner, parent_property), payload in nested.items(): + if owner == "dataset": + doc[parent_property] = payload + + # --- distributions (one per resource) ------------------------------- + dist_entries = by_node.get("distribution", []) + link = next((e for e in std["export"] if e["concept"] == "distribution"), None) + if dist_entries and link: + distributions = [] + for resource in record.get("resources") or []: + dist: Dict[str, Any] = {} + if "distribution" in node_types: + dist["@type"] = node_types["distribution"] + for entry in dist_entries: + if entry["concept"] == "distribution": + continue + value = self._platform_value(resource, entry) + # DCAT requires accessURL on every Distribution; fall back to + # the download URL rather than emit an invalid Distribution. + if value in EMPTY and entry["concept"] == "access_url": + value = resource.get("download_url") + if value in EMPTY: + if entry["obligation"] == "mandatory": + report["missing_mandatory"].append(f"{entry['concept']} (distribution)") + continue + dist[entry["property"]] = self._render(value, entry, uri_style, report) + if len(dist) > (1 if "@type" in dist else 0): + distributions.append(dist) + if distributions: + doc[link["property"]] = distributions + + # --- what this standard cannot carry -------------------------------- + for gap in std.get("gaps", []): + concept = self.concepts.get(gap["concept"], {}) + field = concept.get("dataspace_field") + if field and record.get(field) not in EMPTY: + report["dropped"].append( + { + "concept": gap["concept"], + "dataspace_field": field, + "reason": gap.get("notes") or gap.get("reason"), + } + ) + + return doc, report + + def _platform_value(self, record: dict, entry: dict) -> Any: + """Value for an export entry, by the field the contract binds it to.""" + field = entry.get("dataspace_field") + return record.get(field) if field else None + + def _render(self, value: Any, entry: dict, uri_style: str, report: dict) -> Any: + vtype = entry["value_type"] + if isinstance(value, list): + return [self._render_one(v, entry, vtype, uri_style, report) for v in value] + rendered = self._render_one(value, entry, vtype, uri_style, report) + return [rendered] if entry.get("repeatable") else rendered + + def _render_one(self, value: Any, entry: dict, vtype: str, uri_style: str, report: dict) -> Any: + # Vocabulary values arrive as {"name": …, "uri": …} from the adapter. + if isinstance(value, dict) and vtype in ("concept", "location", "uri", "agent"): + if vtype == "agent" and uri_style == "node": + agent = {"@type": value.get("@type", "foaf:Agent"), "foaf:name": value.get("name")} + if value.get("uri"): + agent["@id"] = value["uri"] + return agent + if vtype == "agent": + kind = SCHEMA_AGENT_TYPES.get(value.get("@type", ""), "Organization") + return {"@type": kind, "name": value.get("name")} | ( + {"url": value["uri"]} if value.get("uri") else {} + ) + value = value.get("uri") or value.get("name") or value.get("label") + elif isinstance(value, dict) and vtype in ("literal", "langstring"): + # A vocabulary value written where the standard wants its plain name. + value = value.get("label") or value.get("name") or value.get("uri") + + if vtype in MUST_RESOLVE and not _is_uri(value): + report["unresolved"].append( + { + "concept": entry["concept"], + "value": value, + "vocabulary": entry.get("controlled_vocabulary"), + "expected": f"{vtype} — a resolvable URI", + } + ) + return value # emit what we have; the report says it is not a URI + + if vtype in REFERENCE_TYPES and uri_style == "node" and _is_uri(value): + return {"@id": value} + if vtype == "bytes": + try: + return int(value) + except (TypeError, ValueError): + return value + return value diff --git a/api/services/metadata_export/exporter.py b/api/services/metadata_export/exporter.py new file mode 100644 index 0000000..185e22b --- /dev/null +++ b/api/services/metadata_export/exporter.py @@ -0,0 +1,127 @@ +"""Orchestrate: model -> record -> crosswalk document -> serialisation fixes -> format. + +Which property a value becomes is decided by the contract. What is left here +is serialisation the contract cannot express: typed date literals, IANA +media-type IRIs, and the few structural rules the specs impose on a +Distribution or FileObject. +""" + +from __future__ import annotations + +from typing import Any, Dict, Tuple + +from api.models import Dataset +from api.services.metadata_export.adapter import IANA_BASE, dataset_to_record +from api.services.metadata_export.crosswalk import Crosswalk +from api.services.metadata_export.formats import allowed_formats, serialise + +CROISSANT_CONFORMS_TO = "http://mlcommons.org/croissant/1.1" +XSD_DATE = "http://www.w3.org/2001/XMLSchema#date" +DATE_PROPERTIES = ("dcterms:issued", "dcterms:modified", "dcterms:created") +PERIOD_PROPERTIES = ("dcat:startDate", "dcat:endDate") + + +def _typed_date(value: Any) -> Any: + """DCAT-AP expects xsd:date literals; JSON-LD needs the type spelled out.""" + if isinstance(value, str) and len(value) >= 10 and value[4] == "-" and value[7] == "-": + return {"@value": value[:10], "@type": XSD_DATE} + return value + + +def _type_dates(doc: Dict[str, Any]) -> None: + for prop in DATE_PROPERTIES: + if prop in doc: + doc[prop] = _typed_date(doc[prop]) + period = doc.get("dcterms:temporal") + if isinstance(period, dict): + for prop in PERIOD_PROPERTIES: + if prop in period: + period[prop] = _typed_date(period[prop]) + + +def _by_name(record: Dict[str, Any]) -> Dict[str, dict]: + return {r.get("name"): r for r in record.get("resources") or [] if r.get("name")} + + +def _fix_dcat_distributions(doc: Dict[str, Any]) -> None: + """Media type as an IANA IRI; licence on every distribution (DCAT-AP).""" + licence = doc.get("dcterms:license") + for dist in doc.get("dcat:distribution") or []: + mt = dist.get("dcat:mediaType") + if isinstance(mt, str) and "/" in mt: + dist["dcat:mediaType"] = {"@id": IANA_BASE + mt} + if licence and "dcterms:license" not in dist: + dist["dcterms:license"] = licence + + +def _fix_croissant_file_objects(doc: Dict[str, Any], record: Dict[str, Any]) -> None: + """Croissant 1.0 requires @id, contentUrl and encodingFormat on a FileObject. + + Croissant has no accessURL, so for a link-only resource the platform page + (what a reader can actually fetch) becomes the contentUrl. sha256 is also + required by the spec and is emitted only when the record carries it. + """ + resources = _by_name(record) + landing = record.get("landing_page") or "" + for n, dist in enumerate(doc.get("distribution") or [], start=1): + res = resources.get(dist.get("name"), {}) + if "contentUrl" not in dist and res.get("access_url"): + dist["contentUrl"] = res["access_url"] + if "encodingFormat" not in dist and res.get("format"): + dist["encodingFormat"] = res["format"] + dist.setdefault( + "@id", + f"{landing}#resource-{res['_id']}" if res.get("_id") else f"{landing}#resource-{n}", + ) + + +def _finish(doc: Dict[str, Any], record: Dict[str, Any], standard: str) -> None: + if standard == "croissant": + doc.setdefault("conformsTo", CROISSANT_CONFORMS_TO) # required by the spec + _fix_croissant_file_objects(doc, record) + elif standard == "dcat": + _fix_dcat_distributions(doc) + _type_dates(doc) + elif standard == "dublin_core": + _type_dates(doc) + + +def export_dataset( + dataset: Dataset, standard: str, fmt: str = "jsonld" +) -> Tuple[str, str, str, dict]: + """Return (body, content_type, extension, report).""" + crosswalk = Crosswalk.load() + if standard not in crosswalk.standard_ids(): + raise ValueError( + f"Unknown standard '{standard}'. Known: {', '.join(crosswalk.standard_ids())}" + ) + if fmt not in allowed_formats(standard): + raise ValueError( + f"Format '{fmt}' is not available for {standard}. " + f"Allowed: {', '.join(allowed_formats(standard))}" + ) + + record = dataset_to_record(dataset) + doc, report = crosswalk.export_with_report(record, standard) + doc.setdefault("@id", record["landing_page"]) + _finish(doc, record, standard) + for definition in record.get("_unmapped_definitions") or []: + report["dropped"].append( + { + "concept": None, + "dataspace_field": f"metadata:{definition.get('urn') or definition.get('label')}", + "reason": "definition has no crosswalk concept; give it a recognised URN", + } + ) + + body, content_type, ext = serialise(doc, fmt) + return body, content_type, ext, report + + +def export_options() -> Dict[str, Any]: + """What the UI can offer: standards, and the formats valid for each.""" + cw = Crosswalk.load() + return { + sid: {"name": cw.standard(sid).get("name", sid), "formats": allowed_formats(sid)} + for sid in cw.standard_ids() + } diff --git a/api/services/metadata_export/formats.py b/api/services/metadata_export/formats.py new file mode 100644 index 0000000..b31eac6 --- /dev/null +++ b/api/services/metadata_export/formats.py @@ -0,0 +1,33 @@ +"""Serialise a JSON-LD document into the RDF syntaxes catalogue harvesters ask for.""" + +from __future__ import annotations + +import json +from typing import Dict, Tuple + +# format id -> (rdflib serializer name, content type, file extension) +FORMATS: Dict[str, Tuple[str, str, str]] = { + "jsonld": ("json-ld", "application/ld+json", "jsonld"), + "turtle": ("turtle", "text/turtle", "ttl"), + "rdfxml": ("xml", "application/rdf+xml", "rdf"), + "ntriples": ("nt", "application/n-triples", "nt"), +} + +# Croissant is defined as JSON-LD; its validator reads nothing else. +JSONLD_ONLY = {"croissant"} + + +def allowed_formats(standard_id: str) -> list: + return ["jsonld"] if standard_id in JSONLD_ONLY else list(FORMATS) + + +def serialise(document: dict, fmt: str) -> Tuple[str, str, str]: + """Return (body, content_type, extension) for the requested format.""" + serializer, content_type, ext = FORMATS[fmt] + if fmt == "jsonld": + return json.dumps(document, indent=2, ensure_ascii=False), content_type, ext + from rdflib import Graph # imported lazily: only non-JSON formats need it + + graph = Graph() + graph.parse(data=json.dumps(document), format="json-ld") + return graph.serialize(format=serializer), content_type, ext diff --git a/api/services/metadata_export/vocabularies.py b/api/services/metadata_export/vocabularies.py new file mode 100644 index 0000000..3bada13 --- /dev/null +++ b/api/services/metadata_export/vocabularies.py @@ -0,0 +1,107 @@ +"""Resolve the platform's labels to URIs using the vendored value lists. + +The standards want identifiers, not names: ``dcterms:spatial "Assam"`` is not +a spatial reference. The lists under ``contracts/`` (see contracts/README.md +repo's value superset) give a URI per licence, sector and geography. Anything +they do not know is returned as a plain label so the crosswalk reports it as +unresolved instead of silently emitting a bad value. +""" + +from __future__ import annotations + +import csv +from functools import lru_cache +from pathlib import Path +from typing import Dict, List, Optional + +CONTRACTS = Path(__file__).resolve().parent / "contracts" + + +def _rows(name: str) -> List[Dict[str, str]]: + with open(CONTRACTS / name, encoding="utf-8", newline="") as fh: + return list(csv.DictReader(fh)) + + +def _alt_codes(raw: str) -> Dict[str, str]: + """Parse the CSV's 'k=v|k=v' alt_codes column (';' tolerated too).""" + out: Dict[str, str] = {} + for part in (raw or "").replace(";", "|").split("|"): + if "=" in part: + k, v = part.split("=", 1) + out[k.strip()] = v.strip() + return out + + +@lru_cache(maxsize=1) +def _license_index() -> Dict[str, str]: + """Lower-cased key / label / DataSpace enum / SPDX id -> URI.""" + index: Dict[str, str] = {} + for r in _rows("licenses.csv"): + uri = r.get("uri") or "" + if not uri: + continue + for k in (r.get("key"), r.get("label"), r.get("code")): + if k: + index[k.strip().lower()] = uri + alt = _alt_codes(r.get("alt_codes", "")) + for k in ("dataspace_enum", "spdx", "spdx_id", "authority_label"): + if alt.get(k): + index[alt[k].strip().lower()] = uri + return index + + +@lru_cache(maxsize=1) +def _sector_index() -> Dict[str, str]: + index: Dict[str, str] = {} + for r in _rows("sectors.csv"): + uri = r.get("uri") or "" + if not uri: + continue + for k in (r.get("key"), r.get("label"), r.get("code")): + if k: + index[k.strip().lower()] = uri + key = r.get("key") or "" + if ":" in key: # "cdl:child-rights" -> "child-rights" + index[key.split(":", 1)[1].lower()] = uri + return index + + +@lru_cache(maxsize=1) +def _geography_index() -> Dict[str, Dict[str, str]]: + """label(lower) -> {tier -> uri}; tier disambiguates 'Aurangabad' etc.""" + index: Dict[str, Dict[str, str]] = {} + for r in _rows("geographies.csv"): + uri, label, tier = ( + r.get("uri") or "", + (r.get("label") or "").strip().lower(), + (r.get("tier") or "").upper(), + ) + if uri and label: + index.setdefault(label, {})[tier] = uri + return index + + +def license_uri(value: Optional[str]) -> Optional[str]: + return _license_index().get((value or "").strip().lower()) + + +def sector_uri(name: Optional[str]) -> Optional[str]: + return _sector_index().get((name or "").strip().lower()) + + +def geography_uri(name: Optional[str], geo_type: Optional[str] = None) -> Optional[str]: + tiers = _geography_index().get((name or "").strip().lower()) + if not tiers: + return None + if geo_type and geo_type.upper() in tiers: + return tiers[geo_type.upper()] + # Prefer the coarsest match when the platform did not say which level. + for tier in ("COUNTRY", "REGION", "STATE", "UT", "DISTRICT"): + if tier in tiers: + return tiers[tier] + return next(iter(tiers.values())) + + +def concept(name: str, uri: Optional[str]) -> Dict[str, str]: + """Shape the crosswalk renders: uri when known, else the bare name (reported).""" + return {"name": name, "uri": uri} if uri else {"name": name} diff --git a/api/services/metadata_mapping.py b/api/services/metadata_mapping.py new file mode 100644 index 0000000..284f10a --- /dev/null +++ b/api/services/metadata_mapping.py @@ -0,0 +1,236 @@ +"""Bridge between the deployment's *metadata definitions* and the crosswalk. + +Administrators define extra dataset fields in Django admin (``Metadata``: +label, URN, data type, required or optional). Publishers fill them in the +edit form and the values live in ``DatasetMetadata``. Until now those values +were opaque strings: the import could not prefill them and the export could +not place them in a standard. + +This module gives each definition a *crosswalk concept* so both sides can use +it. A definition is matched, in order, by + +1. its URN, once normalised: prefix dropped, camelCase to snake_case, so + ``ds:source_website``, ``dcterms:source`` and ``schema:sourceWebsite`` all + become ``source_website`` / ``source``; +2. any property name a standard uses for a concept in the contract + (``dcterms:issued`` -> ``issued``, ``dcat:landingPage`` -> ``landing_page``); +3. its label, normalised the same way ("Date of Creation of Dataset" -> + ``date_of_creation_of_dataset`` -> ``created``). + +Anything that resolves to nothing stays an opaque string: the import leaves +it for the publisher and the export lists it under ``dropped`` in the report, +so the gap is visible instead of silently lost. +""" + +from __future__ import annotations + +import re +from datetime import date, datetime +from functools import lru_cache +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple + +if TYPE_CHECKING: # pragma: no cover + from api.models import Dataset, Metadata + from api.services.platform_importers.base import PlatformDatasetInfo + +# House names and standard names that all mean the same crosswalk concept. +# Keys are normalised (see ``normalise``); values are concept keys in +# contracts/crosswalk.json, plus ``citation`` which the contract does not +# carry yet but Croissant (citeAs) does. +ALIASES: Dict[str, str] = { + # where the dataset came from + "source": "source", + "source_url": "source", + "source_website": "source", + "source_link": "source", + "original_source": "source", + "original_url": "source", + "data_source": "source", + "was_derived_from": "source", + "source_identifier": "source_identifier", + "source_id": "source_identifier", + "source_platform": "source_platform", + "imported_from": "source_platform", + # people and organisations + "author": "creator", + "creator": "creator", + "authors": "creator", + "original_author": "creator", + "publisher": "publisher", + # dates + "created": "created", + "created_on": "created", + "created_at": "created", + "creation_date": "created", + "date_created": "created", + "date_of_creation": "created", + "date_of_creation_of_dataset": "created", + "issued": "issued", + "published": "issued", + "published_on": "issued", + "date_published": "issued", + "release_date": "issued", + "modified": "modified", + "last_modified": "modified", + "last_updated": "modified", + "updated": "modified", + "updated_on": "modified", + "date_modified": "modified", + "source_last_updated": "modified", + # rights and identity + "license": "license", + "licence": "license", + "original_license": "license", + "source_license": "license", + "rights": "rights", + "version": "version", + "revision": "version", + # links and text + "homepage": "homepage", + "home_page": "homepage", + "website": "homepage", + "project_url": "homepage", + "landing_page": "landing_page", + "citation": "citation", + "cite_as": "citation", + "how_to_cite": "citation", + "language": "language", + "languages": "language", + "in_language": "language", + # coverage and cadence + "temporal_coverage_start": "temporal_coverage_start", + "temporal_start": "temporal_coverage_start", + "period_start": "temporal_coverage_start", + "start_date": "temporal_coverage_start", + "coverage_start": "temporal_coverage_start", + "temporal_coverage_end": "temporal_coverage_end", + "temporal_end": "temporal_coverage_end", + "period_end": "temporal_coverage_end", + "end_date": "temporal_coverage_end", + "coverage_end": "temporal_coverage_end", + "accrual_periodicity": "accrual_periodicity", + "frequency": "accrual_periodicity", + "update_frequency": "accrual_periodicity", + "periodicity": "accrual_periodicity", +} + +# Concepts whose value is a calendar date on the way in and out. +DATE_CONCEPTS = { + "created", + "issued", + "modified", + "temporal_coverage_start", + "temporal_coverage_end", +} +# Concepts whose value is a list joined with commas in the definition's cell. +LIST_CONCEPTS = {"language"} + +_CAMEL = re.compile(r"(?<=[a-z0-9])(?=[A-Z])") +_NON_WORD = re.compile(r"[^a-z0-9]+") + + +def normalise(name: Optional[str]) -> str: + """``ds:createdOn`` -> ``created_on``; ``Source Website`` -> ``source_website``.""" + if not name: + return "" + local = name.strip().rsplit(":", 1)[-1].rsplit("/", 1)[-1].rsplit("#", 1)[-1] + local = _CAMEL.sub("_", local) + return _NON_WORD.sub("_", local.lower()).strip("_") + + +@lru_cache(maxsize=1) +def _contract_property_index() -> Dict[str, str]: + """Normalised standard property name -> concept, from the vendored contract.""" + from api.services.metadata_export.crosswalk import Crosswalk + + index: Dict[str, str] = {} + for std in Crosswalk.load().standards.values(): + # ``export`` is a flat list of entries; ``import`` is nested + # node -> property -> [entries]. + entries: List[dict] = list(std.get("export") or []) + for node, by_property in (std.get("import") or {}).items(): + for prop, found in by_property.items(): + for entry in found if isinstance(found, list) else [found]: + entries.append(dict(entry, property=entry.get("property") or prop, node=node)) + for entry in entries: + prop, concept = entry.get("property"), entry.get("concept") + if prop and concept and entry.get("node", "dataset") == "dataset": + index.setdefault(normalise(prop), concept) + return index + + +def concept_for(urn: Optional[str], label: Optional[str] = None) -> Optional[str]: + """The crosswalk concept a definition stands for, or None if unrecognised.""" + for candidate in (normalise(urn), normalise(label)): + if not candidate: + continue + if candidate in ALIASES: + return ALIASES[candidate] + concept = _contract_property_index().get(candidate) + if concept: + return concept + return None + + +# --------------------------------------------------------------------- import +def platform_values(info: "PlatformDatasetInfo", platform_label: str) -> Dict[str, str]: + """What an import can offer each concept, as the string a definition cell holds.""" + + def day(value: Optional[datetime]) -> str: + return value.date().isoformat() if value else "" + + values = { + "source": info.source_url, + "source_identifier": info.identifier, + "source_platform": platform_label, + "creator": info.author, + "publisher": info.author, + "created": day(info.created_at), + "issued": day(info.created_at), + "modified": day(info.last_updated), + "license": info.license, + "version": info.revision, + "homepage": info.homepage, + "citation": info.citation, + "language": ", ".join(info.languages or []), + } + return {k: v for k, v in values.items() if v} + + +# --------------------------------------------------------------------- export +def _coerce(concept: str, raw: str) -> Any: + value = (raw or "").strip() + if not value: + return None + if concept in LIST_CONCEPTS: + return [v.strip() for v in value.split(",") if v.strip()] + if concept in DATE_CONCEPTS: + try: + return date.fromisoformat(value[:10]).isoformat() + except ValueError: + return value + return value + + +def definition_values(dataset: "Dataset") -> Tuple[Dict[str, Any], List[dict]]: + """Definition-backed values of a dataset, keyed by concept. + + Returns ``(values, unmapped)``. ``unmapped`` lists definitions the mapping + could not place, for the export report. + """ + values: Dict[str, Any] = {} + unmapped: List[dict] = [] + rows = dataset.metadata.select_related("metadata_item").all() + for row in rows: + item: "Metadata" = row.metadata_item + if not item.enabled: + continue + concept = concept_for(item.urn, item.label) + if concept is None: + unmapped.append({"label": item.label, "urn": item.urn, "value": row.value}) + continue + coerced = _coerce(concept, row.value) + if coerced in (None, "", []): + continue + values.setdefault(concept, coerced) + return values, unmapped diff --git a/api/services/platform_import_service.py b/api/services/platform_import_service.py new file mode 100644 index 0000000..0889cc9 --- /dev/null +++ b/api/services/platform_import_service.py @@ -0,0 +1,270 @@ +"""Turn a third-party platform dataset into a DataSpace Dataset (link-only). + +Creates the Dataset (DRAFT), a single EXTERNAL Resource that links to the +dataset's page on the platform, a DatasetSource provenance row, tags, taxonomy +and metadata prefill, and the creator's owner permission — all inside one +transaction. Files are never listed or copied: people reach them through the +original dataset link. +""" + +from __future__ import annotations + +from typing import Iterable, Optional + +import structlog +from django.db import transaction +from django.utils.text import slugify + +from api.models import ( + Dataset, + DatasetMetadata, + DatasetSource, + DataSpace, + Geography, + Metadata, + Organization, + Resource, + ResourceSchema, + Sector, + Tag, +) +from api.services import metadata_mapping +from api.services.platform_importers import ( + PlatformDatasetInfo, + PlatformImportError, + get_importer, +) +from api.utils.enums import ( + DatasetAccessType, + DatasetStatus, + DatasetType, + DataType, + MetadataModels, +) +from authorization.models import DatasetPermission, Role, User + +logger = structlog.get_logger("dataspace.platform_import") + + +class DuplicateImportError(PlatformImportError): + """The same platform dataset was already imported into this publisher scope.""" + + def __init__(self, existing: Dataset) -> None: + super().__init__( + f"This dataset was already imported as '{existing.title}' ({existing.slug})" + ) + self.existing = existing + + +def preview_platform_dataset(platform: str, identifier: str) -> PlatformDatasetInfo: + """Fetch normalised metadata without creating anything.""" + importer = get_importer(platform) + return importer.fetch_dataset_info(identifier) + + +def find_existing_import( + platform: str, + identifier: str, + organization: Optional[Organization], + user: Optional[User], +) -> Optional[Dataset]: + """Return a dataset already imported from this source in the same scope. + + Scope is the organization when importing on its behalf, otherwise the + individual user — mirroring how datasets are owned elsewhere. + """ + qs = DatasetSource.objects.select_related("dataset").filter( + platform=platform, source_identifier=identifier + ) + if organization is not None: + qs = qs.filter(dataset__organization=organization) + else: + qs = qs.filter(dataset__organization__isnull=True, dataset__user=user) + source = qs.first() + return source.dataset if source else None + + +def _unique_dataset_slug(title: str) -> str: + """Pick a free slug up front. Dataset.save() retries on a slug collision, + but only a handful of times; platform titles ("Iris", "Wine Reviews") + repeat across organisations far more often than timestamped native ones, + so choose a free slug before saving rather than rely on the retry.""" + base = slugify(title)[:240] or "imported-dataset" + slug, counter = base, 2 + while Dataset.objects.filter(slug=slug).exists(): + slug = f"{base}-{counter}" + counter += 1 + return slug + + +PLATFORM_LABELS = {"HUGGINGFACE": "Hugging Face", "KAGGLE": "Kaggle", "GITHUB": "GitHub"} + + +# Platform tags/topics are matched (case-insensitively) against our own +# taxonomies so the publisher lands on the metadata step with sectors and +# geographies already ticked where the names line up. Publishing requires +# sectors, so this is the most valuable prefill we can do without a human. +def _prefill_taxonomies(dataset: Dataset, tags: Iterable[str]) -> None: + names = {t.strip().lower() for t in tags if t and t.strip()} + if not names: + return + sectors = [s for s in Sector.objects.all() if s.name.strip().lower() in names] + if sectors: + dataset.sectors.add(*sectors) + geographies = [g for g in Geography.objects.all() if g.name.strip().lower() in names] + if geographies: + dataset.geographies.add(*geographies) + + +# Optional EAV prefill: if the deployment defines dataset metadata fields whose +# label matches one of these (case-insensitive), fill it from the platform. +# Deployments without such fields are simply skipped. +def _prefill_metadata(dataset: Dataset, info: PlatformDatasetInfo) -> None: + """Fill the deployment's dataset metadata definitions from the platform. + + Each enabled definition is matched to a crosswalk concept by URN, then by + label (``api.services.metadata_mapping``); the platform value for that + concept, if any, becomes the definition's value. A definition whose + validators reject the value is skipped and left for the publisher. + """ + platform_label = PLATFORM_LABELS.get(str(info.platform), str(info.platform).title()) + values = metadata_mapping.platform_values(info, platform_label) + fields = Metadata.objects.filter(enabled=True, model=MetadataModels.DATASET) + for field in fields: + concept = metadata_mapping.concept_for(field.urn, field.label) + value = values.get(concept or "", "") + if not value: + continue + try: + DatasetMetadata(dataset=dataset, metadata_item=field, value=str(value)[:1000]).save() + except Exception as exc: # validators on the field may reject the value; that's fine + logger.info("platform_import_metadata_skipped", label=field.label, error=str(exc)) + + +def import_platform_dataset( + *, + platform: str, + identifier: str, + user: User, + organization: Optional[Organization] = None, + dataspace: Optional[DataSpace] = None, + info: Optional[PlatformDatasetInfo] = None, + title: Optional[str] = None, +) -> Dataset: + """Create a DRAFT dataset from a platform source. Raises PlatformImportError.""" + importer = get_importer(platform) + canonical_id = importer.parse_identifier(identifier) + + existing = find_existing_import(platform, canonical_id, organization, user) + if existing is not None: + raise DuplicateImportError(existing) + + # Network first, database second: the platform calls (up to a few seconds) + # happen before any transaction is opened, so a slow platform never holds a + # database connection. + if info is None: + info = importer.fetch_dataset_info(canonical_id) + + # The platform may have canonicalised the id (Hugging Face "imdb" -> + # "stanfordnlp/imdb"); re-check duplicates under the canonical id too. + if info.identifier != canonical_id: + existing = find_existing_import(platform, info.identifier, organization, user) + if existing is not None: + raise DuplicateImportError(existing) + + with transaction.atomic(): + return _create_import( + info, user=user, organization=organization, dataspace=dataspace, title=title + ) + + +def _create_import( + info: PlatformDatasetInfo, + *, + user: User, + organization: Optional[Organization], + dataspace: Optional[DataSpace], + title: Optional[str], +) -> Dataset: + """All database writes for one import. Runs inside a transaction.""" + # Publisher may choose the name shown on DataSpace; platform title otherwise. + display_title = (title or "").strip()[:300] or info.title + dataset = Dataset.objects.create( + title=display_title, + slug=_unique_dataset_slug(display_title), + description=info.description, + user=user, + organization=organization, + dataspace=dataspace, + status=DatasetStatus.DRAFT, + access_type=DatasetAccessType.PUBLIC, + license=info.mapped_license, + dataset_type=DatasetType.DATA, + ) + + if info.tags: + tags = [ + Tag.objects.get_or_create(defaults={"value": value}, value__iexact=value)[0] + for value in info.tags + ] + dataset.tags.set(tags) + _prefill_taxonomies(dataset, info.tags) + _prefill_metadata(dataset, info) + + # One resource standing for the whole dataset on the platform. Its URL is + # the dataset page, so "download" redirects there and people browse/fetch + # files with the platform's own tooling. + platform_label = PLATFORM_LABELS.get(str(info.platform), str(info.platform).title()) + link_resource = Resource.objects.create( + dataset=dataset, + type=DataType.EXTERNAL, + name=f"Dataset on {platform_label}"[:200], + url=info.source_url[:500], + description=f"Files are hosted on {platform_label}. Open the link to browse and download them.", + ) + + DatasetSource.objects.create( + dataset=dataset, + platform=info.platform, + source_identifier=info.identifier, + source_url=info.source_url[:500], + source_homepage=(info.homepage or "")[:500], + revision=(info.revision or "")[:64], + source_author=info.author[:300], + source_license=info.license[:300], + source_readme=info.readme or "", + citation=info.citation or "", + languages=list(info.languages or []), + source_created_at=info.created_at, + source_last_updated=info.last_updated, + is_archived=bool(info.is_archived), + imported_by=user, + ) + + # Column definitions the platform declared (Hugging Face dataset_info). + # They live on the link resource so "View All Columns" and the Croissant + # recordSet work without any data being fetched. + ResourceSchema.objects.bulk_create( + [ + ResourceSchema(resource=link_resource, field_name=col.name, format=col.field_type) + for col in info.columns + ] + ) + + try: + owner_role = Role.objects.get(name="owner") + except Role.DoesNotExist as exc: + # Same seed the rest of the app relies on (add_dataset does a bare .get()). + raise PlatformImportError( + "Roles are not initialised on this server (run `manage.py init_roles`)" + ) from exc + DatasetPermission.objects.create(user=user, dataset=dataset, role=owner_role) + + logger.info( + "platform_dataset_imported", + platform=info.platform, + identifier=info.identifier, + dataset_id=str(dataset.id), + user_id=str(user.id), + ) + return dataset diff --git a/api/services/platform_importers/__init__.py b/api/services/platform_importers/__init__.py new file mode 100644 index 0000000..be24049 --- /dev/null +++ b/api/services/platform_importers/__init__.py @@ -0,0 +1,66 @@ +"""Registry of third-party platform importers. + +Add a platform by subclassing ``PlatformImporter`` and registering it here; +the GraphQL layer and import service never reference a concrete importer. +""" + +from typing import Dict, Optional, Type +from urllib.parse import urlparse + +from api.services.platform_importers.base import ( + InvalidIdentifierError, + PlatformAuthError, + PlatformColumn, + PlatformDatasetInfo, + PlatformDatasetNotFoundError, + PlatformImporter, + PlatformImportError, + PlatformUnavailableError, +) +from api.services.platform_importers.github import GitHubImporter +from api.services.platform_importers.huggingface import HuggingFaceImporter +from api.services.platform_importers.kaggle import KaggleImporter +from api.utils.enums import ImportPlatform + +IMPORTERS: Dict[str, Type[PlatformImporter]] = { + ImportPlatform.KAGGLE: KaggleImporter, + ImportPlatform.HUGGINGFACE: HuggingFaceImporter, + ImportPlatform.GITHUB: GitHubImporter, +} + + +def get_importer(platform: str) -> PlatformImporter: + try: + return IMPORTERS[str(platform)]() + except KeyError as exc: + raise InvalidIdentifierError(f"Unsupported platform: {platform}") from exc + + +def detect_platform(value: str) -> Optional[str]: + """Guess the platform from a pasted URL (None for bare identifiers).""" + raw = (value or "").strip() + if not raw: + return None + host = urlparse(raw if "://" in raw else f"https://{raw}").netloc.lower() + for platform, importer in IMPORTERS.items(): + if host in importer.hosts: + return platform + return None + + +__all__ = [ + "IMPORTERS", + "get_importer", + "detect_platform", + "PlatformImporter", + "PlatformDatasetInfo", + "PlatformColumn", + "PlatformImportError", + "InvalidIdentifierError", + "PlatformDatasetNotFoundError", + "PlatformAuthError", + "PlatformUnavailableError", + "HuggingFaceImporter", + "GitHubImporter", + "KaggleImporter", +] diff --git a/api/services/platform_importers/base.py b/api/services/platform_importers/base.py new file mode 100644 index 0000000..61518bf --- /dev/null +++ b/api/services/platform_importers/base.py @@ -0,0 +1,269 @@ +"""Shared contract for third-party platform importers. + +An importer turns a user-supplied identifier (short id or pasted URL) into a +normalised ``PlatformDatasetInfo`` by calling the platform's public API. +Importers fetch *metadata only* — title, description, license, tags, author, +last updated and the dataset's page URL. Files are never listed or copied. +""" + +from __future__ import annotations + +import re +from abc import ABC, abstractmethod +from dataclasses import dataclass, field +from datetime import datetime +from typing import Any, Dict, List, Optional + +import requests +import structlog +from django.conf import settings +from django.utils.dateparse import parse_datetime + +from api.utils.enums import DatasetLicense, ImportPlatform + +logger = structlog.get_logger("dataspace.platform_import") + + +# --------------------------------------------------------------------------- # +# Errors +# --------------------------------------------------------------------------- # +class PlatformImportError(Exception): + """Base class for importer failures. ``message`` is safe to show to users.""" + + def __init__(self, message: str) -> None: + super().__init__(message) + self.message = message + + +class InvalidIdentifierError(PlatformImportError): + """The identifier/URL does not look like anything the platform accepts.""" + + +class PlatformDatasetNotFoundError(PlatformImportError): + """The platform reported no dataset for this identifier.""" + + +class PlatformAuthError(PlatformImportError): + """Credentials are missing/invalid, or the dataset is private/gated.""" + + +class PlatformUnavailableError(PlatformImportError): + """Network failure, timeout, rate limit, or a 5xx from the platform.""" + + +# --------------------------------------------------------------------------- # +# Normalised result +# --------------------------------------------------------------------------- # +@dataclass +class PlatformColumn: + """One column of the dataset as the platform describes it.""" + + name: str + field_type: str # a FieldTypes value: STRING / NUMBER / INTEGER / DATE / BOOLEAN + + +@dataclass +class PlatformDatasetInfo: + """Everything an importer returns. Each field lands in a typed column on + DatasetSource (or on Dataset / ResourceSchema); nothing raw is kept.""" + + platform: str + identifier: str + title: str + description: str # short form, fits Dataset.description (1,000 chars) + source_url: str + author: str = "" + license: str = "" + tags: List[str] = field(default_factory=list) + last_updated: Optional[datetime] = None + created_at: Optional[datetime] = None + revision: str = "" + readme: str = "" # full card / README, unbounded + citation: str = "" + languages: List[str] = field(default_factory=list) + homepage: str = "" + is_archived: bool = False + columns: List[PlatformColumn] = field(default_factory=list) + + @property + def mapped_license(self) -> str: + return map_license(self.license) + + +# --------------------------------------------------------------------------- # +# Helpers shared by importers +# --------------------------------------------------------------------------- # +# Platform license strings (lower-cased) -> DatasetLicense. Anything not listed +# falls back to CC-BY 4.0 and the raw string is kept on DatasetSource. +LICENSE_ALIASES: Dict[str, str] = { + "cc-by-4.0": DatasetLicense.CC_BY_4_0_ATTRIBUTION, + "cc-by": DatasetLicense.CC_BY_4_0_ATTRIBUTION, + "cc by 4.0": DatasetLicense.CC_BY_4_0_ATTRIBUTION, + "attribution 4.0 international (cc by 4.0)": DatasetLicense.CC_BY_4_0_ATTRIBUTION, + "cc-by-sa-4.0": DatasetLicense.CC_BY_SA_4_0_ATTRIBUTION_SHARE_ALIKE, + "cc-by-sa": DatasetLicense.CC_BY_SA_4_0_ATTRIBUTION_SHARE_ALIKE, + "attribution-sharealike 4.0 international (cc by-sa 4.0)": ( + DatasetLicense.CC_BY_SA_4_0_ATTRIBUTION_SHARE_ALIKE + ), + "odc-by": DatasetLicense.OPEN_DATA_COMMONS_BY_ATTRIBUTION, + "odc-by-1.0": DatasetLicense.OPEN_DATA_COMMONS_BY_ATTRIBUTION, + "odc attribution license (odc-by)": DatasetLicense.OPEN_DATA_COMMONS_BY_ATTRIBUTION, + "odbl": DatasetLicense.OPEN_DATABASE_LICENSE, + "odbl-1.0": DatasetLicense.OPEN_DATABASE_LICENSE, + "odc-odbl": DatasetLicense.OPEN_DATABASE_LICENSE, + "database: open database, contents: database contents": DatasetLicense.OPEN_DATABASE_LICENSE, + "database: open database, contents: © original authors": DatasetLicense.OPEN_DATABASE_LICENSE, +} + + +def map_license(raw: str) -> str: + """Map a platform license string onto DatasetLicense (default CC-BY-4.0).""" + key = (raw or "").strip().lower() + return LICENSE_ALIASES.get(key, DatasetLicense.CC_BY_4_0_ATTRIBUTION) + + +# Hard limits: the platforms' own maxima are well under these, so anything +# longer is not a real identifier. Keeps hostile input out of URLs and columns. +MAX_IDENTIFIER_LEN = 200 +MAX_README_CHARS = 200_000 # ~200 KB; the longest real card seen is ~32 KB + +_SEGMENT_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$") + + +def check_identifier_length(value: str, platform_label: str) -> None: + if len(value) > MAX_IDENTIFIER_LEN: + raise InvalidIdentifierError(f"That {platform_label} identifier is too long to be real") + + +def check_path_segments(path: str, what: str) -> None: + """A branch name or sub-path must be plain segments: no '..', no whitespace, no odd characters.""" + for seg in (path or "").split("/"): + if seg and (seg in (".", "..") or not _SEGMENT_RE.match(seg)): + raise InvalidIdentifierError(f"'{path}' is not a valid {what}") + + +def cap_readme(text: str) -> str: + return (text or "")[:MAX_README_CHARS] + + +def field_type_for(dtype: Any) -> str: + """Map a platform column type (Hugging Face / Arrow style names) onto FieldTypes.""" + name = str(dtype if isinstance(dtype, str) else "").lower() + if name in ("bool", "boolean"): + return "BOOLEAN" + if name.startswith(("int", "uint")): + return "INTEGER" + if name.startswith(("float", "double", "decimal")): + return "NUMBER" + if name.startswith(("date", "timestamp", "time")): + return "DATE" + return "STRING" # strings, class labels, nested/binary types + + +def shorten(text: str, limit: int = 1000) -> str: + """Cut on a word boundary with an ellipsis; used for Dataset.description.""" + text = (text or "").strip() + if len(text) <= limit: + return text + cut = text[: limit - 1].rsplit(" ", 1)[0] + return cut + "…" + + +def parse_iso_datetime(value: Optional[str]) -> Optional[datetime]: + """Parse a platform timestamp. Django's parser copes with a trailing ``Z`` + and any number of fractional digits (Kaggle sends e.g. ``...:04.7Z``), + which ``datetime.fromisoformat`` on Python 3.10 does not.""" + if not value: + return None + try: + return parse_datetime(value) + except ValueError: + return None + + +# --------------------------------------------------------------------------- # +# Base importer +# --------------------------------------------------------------------------- # +class PlatformImporter(ABC): + """Contract every platform importer implements.""" + + platform: str + #: Human name used in user-facing messages, e.g. "Hugging Face". + label: str = "" + #: Hostnames whose URLs this importer can parse (used by parse_identifier). + hosts: tuple = () + + def __init__(self, session: Optional[requests.Session] = None) -> None: + self.session = session or requests.Session() + self.timeout: float = float(getattr(settings, "PLATFORM_IMPORT_TIMEOUT", 15)) + + @abstractmethod + def parse_identifier(self, value: str) -> str: + """Normalise a short id or pasted URL to the platform's canonical id. + + Raises InvalidIdentifierError when it cannot. + """ + + @abstractmethod + def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: + """Call the platform API and return normalised metadata.""" + + # -- HTTP plumbing ------------------------------------------------------- # + def _get(self, url: str, **kwargs: Any) -> requests.Response: + """GET with uniform error mapping. Subclasses add auth via kwargs.""" + kwargs.setdefault("timeout", self.timeout) + name = self.label or str(self.platform) + try: + response = self.session.get(url, **kwargs) + except requests.Timeout as exc: + raise PlatformUnavailableError(f"{name} timed out") from exc + except requests.RequestException as exc: + logger.warning( + "platform_import_request_failed", platform=self.platform, url=url, error=str(exc) + ) + raise PlatformUnavailableError(f"Could not reach {name}") from exc + + if response.status_code == 404: + raise PlatformDatasetNotFoundError(f"Dataset not found on {name}") + if response.status_code in (401, 403): + raise PlatformAuthError( + f"{name} refused access. The dataset may be private/gated, " + "or the server credentials are missing or invalid." + ) + if response.status_code == 429: + raise PlatformUnavailableError(f"{name} rate limit reached, try again later") + if response.status_code >= 500: + raise PlatformUnavailableError(f"{name} returned an error ({response.status_code})") + if response.status_code >= 400: + raise PlatformImportError(f"{name} rejected the request ({response.status_code})") + return response + + def _get_json(self, url: str, **kwargs: Any) -> Any: + response = self._get(url, **kwargs) + try: + return response.json() + except ValueError as exc: + raise PlatformUnavailableError( + f"{self.label or self.platform} returned an unreadable response" + ) from exc + + +__all__ = [ + "ImportPlatform", + "PlatformImporter", + "PlatformDatasetInfo", + "PlatformColumn", + "PlatformImportError", + "InvalidIdentifierError", + "PlatformDatasetNotFoundError", + "PlatformAuthError", + "PlatformUnavailableError", + "map_license", + "check_identifier_length", + "check_path_segments", + "cap_readme", + "MAX_IDENTIFIER_LEN", + "field_type_for", + "shorten", + "parse_iso_datetime", +] diff --git a/api/services/platform_importers/github.py b/api/services/platform_importers/github.py new file mode 100644 index 0000000..e5792b1 --- /dev/null +++ b/api/services/platform_importers/github.py @@ -0,0 +1,175 @@ +"""GitHub repository importer (metadata only). + +A repo (optionally a branch and sub-folder) is treated as a dataset. Three +small requests per import: the repo metadata, the branch head (for the +revision) and the README. Public repos need no +token; ``GITHUB_TOKEN`` (settings) is sent when present, mainly to lift the +anonymous 60 requests/hour rate limit. +""" + +from __future__ import annotations + +import re +from typing import Any, Dict, Optional, Tuple +from urllib.parse import urlparse + +from django.conf import settings + +from api.services.platform_importers.base import ( + InvalidIdentifierError, + PlatformDatasetInfo, + PlatformImporter, + PlatformImportError, + cap_readme, + check_identifier_length, + check_path_segments, + parse_iso_datetime, + shorten, +) +from api.utils.enums import ImportPlatform + +GITHUB_HOSTS = ("github.com", "www.github.com") +GITHUB_API = "https://api.github.com" +GITHUB_WEB = "https://github.com" + +_NAME_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.-]*$") +MAX_DESCRIPTION = 1000 + + +class GitHubImporter(PlatformImporter): + platform = ImportPlatform.GITHUB + label = "GitHub" + hosts = GITHUB_HOSTS + + def _headers(self) -> Dict[str, str]: + headers = {"Accept": "application/vnd.github+json"} + token = getattr(settings, "GITHUB_TOKEN", None) + if token: + headers["Authorization"] = f"Bearer {token}" + return headers + + # -- identifier ---------------------------------------------------------- # + def parse_identifier(self, value: str) -> str: + """Normalise to ``owner/repo`` or ``owner/repo@branch:sub/path``.""" + owner, repo, branch, path = self._parse(value) + ident = f"{owner}/{repo}" + if branch: + ident += f"@{branch}" + if path: + ident += f":{path}" + return ident + + def _parse(self, value: str) -> Tuple[str, str, Optional[str], str]: + raw = (value or "").strip() + if not raw: + raise InvalidIdentifierError("Enter a GitHub repository (owner/repo) or URL") + + branch: Optional[str] = None + path = "" + if "://" in raw or raw.startswith(GITHUB_HOSTS): + parsed = urlparse(raw if "://" in raw else f"https://{raw}") + if parsed.netloc.lower() not in GITHUB_HOSTS: + raise InvalidIdentifierError("That is not a github.com URL") + parts = [p for p in parsed.path.split("/") if p] + if len(parts) < 2: + raise InvalidIdentifierError( + "Expected a repository URL like https://github.com//" + ) + owner, repo = parts[0], parts[1].removesuffix(".git") + # /tree// or /blob// + if len(parts) >= 4 and parts[2] in ("tree", "blob"): + branch = parts[3] + path = "/".join(parts[4:]) + else: + spec = raw.strip("/") + if ":" in spec: + spec, path = spec.split(":", 1) + if "@" in spec: + spec, branch = spec.split("@", 1) + parts = spec.split("/") + if len(parts) != 2: + raise InvalidIdentifierError( + "GitHub repos look like 'owner/repo' (optionally owner/repo@branch:path)" + ) + owner, repo = parts + + if not (_NAME_RE.match(owner) and _NAME_RE.match(repo)): + raise InvalidIdentifierError( + "GitHub owner and repo names use letters, digits, '-', '_' and '.'" + ) + path = path.strip("/") + if branch: + check_path_segments(branch, "branch name") + if path: + check_path_segments(path, "folder path") + check_identifier_length(f"{owner}/{repo}@{branch or ''}:{path}", "GitHub") + return owner, repo, (branch or None), path + + # -- fetch --------------------------------------------------------------- # + def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: + owner, repo, branch, sub_path = self._parse(identifier) + headers = self._headers() + + meta: Dict[str, Any] = self._get_json(f"{GITHUB_API}/repos/{owner}/{repo}", headers=headers) + if meta.get("disabled"): + raise PlatformImportError("This repository has been disabled on GitHub") + branch = branch or meta.get("default_branch") or "main" + full_name = meta.get("full_name") or f"{owner}/{repo}" + + # Repo names are slugs ("covid-19-data"); make a readable title from them. + title = repo.replace("-", " ").replace("_", " ").strip() + if sub_path: + title = f"{title} / {sub_path}" + + source_url = f"{GITHUB_WEB}/{full_name}" + if sub_path: + source_url += f"/tree/{branch}/{sub_path}" + + readme = self._readme(full_name, headers) or str(meta.get("description") or "") + license_info = meta.get("license") or {} + + return PlatformDatasetInfo( + platform=self.platform, + identifier=self.parse_identifier(identifier), + title=title[:300], + description=shorten(readme, MAX_DESCRIPTION), + source_url=source_url, + author=str((meta.get("owner") or {}).get("login") or owner)[:300], + license=self._license(license_info), + tags=[t[:50] for t in (meta.get("topics") or []) if isinstance(t, str)][:30], + last_updated=parse_iso_datetime(meta.get("pushed_at") or meta.get("updated_at")), + created_at=parse_iso_datetime(meta.get("created_at")), + revision=self._head_sha(full_name, branch, headers), + readme=cap_readme(readme), + homepage=str(meta.get("homepage") or "")[:500], + is_archived=bool(meta.get("archived")), + ) + + # -- pieces -------------------------------------------------------------- # + def _readme(self, full_name: str, headers: Dict[str, str]) -> str: + try: + response = self._get( + f"{GITHUB_API}/repos/{full_name}/readme", + headers={**headers, "Accept": "application/vnd.github.raw"}, + ) + return response.text.strip() + except PlatformImportError: + return "" + + def _head_sha(self, full_name: str, branch: str, headers: Dict[str, str]) -> str: + """Commit SHA at the branch head; empty if the lookup fails.""" + try: + ref: Dict[str, Any] = self._get_json( + f"{GITHUB_API}/repos/{full_name}/git/ref/heads/{branch}", headers=headers + ) + return str((ref.get("object") or {}).get("sha") or "")[:64] + except PlatformImportError: + return "" + + @staticmethod + def _license(license_info: Dict[str, Any]) -> str: + """SPDX id when GitHub could detect one. 'NOASSERTION' / 'other' mean it could not.""" + spdx = str(license_info.get("spdx_id") or "") + if spdx.upper() in ("", "NOASSERTION", "OTHER"): + return "" + return spdx diff --git a/api/services/platform_importers/huggingface.py b/api/services/platform_importers/huggingface.py new file mode 100644 index 0000000..8a8b4c2 --- /dev/null +++ b/api/services/platform_importers/huggingface.py @@ -0,0 +1,233 @@ +"""Hugging Face Hub dataset importer (metadata only). + +Two small requests per import: the repo metadata (without its file list) and +the README (dataset card). +Public datasets need no token; ``HF_TOKEN`` (settings) is sent when present so +gated/private repos the token can see also work. +""" + +from __future__ import annotations + +import re +from typing import Any, Dict, List +from urllib.parse import urlparse + +from django.conf import settings + +from api.services.platform_importers.base import ( + InvalidIdentifierError, + PlatformAuthError, + PlatformColumn, + PlatformDatasetInfo, + PlatformImporter, + PlatformImportError, + cap_readme, + check_identifier_length, + field_type_for, + parse_iso_datetime, + shorten, +) +from api.utils.enums import ImportPlatform + +HF_HOSTS = ("huggingface.co", "www.huggingface.co", "hf.co") +HF_API = "https://huggingface.co/api/datasets" +HF_WEB = "https://huggingface.co/datasets" + +# "namespace/name" or a canonical "name"; HF ids allow letters, digits, - _ . +_ID_RE = re.compile(r"^(?:[A-Za-z0-9][A-Za-z0-9._-]*/)?[A-Za-z0-9][A-Za-z0-9._-]*$") + +MAX_DESCRIPTION = 1000 # Dataset.description column length + +# Everything the Hub returns by default EXCEPT ``siblings`` (the per-file list). +# We never list files, and for large repos that one key is most of the payload: +# ~9.6 MB for an 85k-file repo versus ~4 KB without it. Asking for fields by +# name means the list is never downloaded, and never stored anywhere. +_EXPAND_FIELDS = ( + "author", + "cardData", + "citation", + "createdAt", + "description", + "disabled", + "gated", + "lastModified", + "private", + "sha", + "tags", +) + + +class HuggingFaceImporter(PlatformImporter): + platform = ImportPlatform.HUGGINGFACE + label = "Hugging Face" + hosts = HF_HOSTS + + def _headers(self) -> Dict[str, str]: + token = getattr(settings, "HF_TOKEN", None) + return {"Authorization": f"Bearer {token}"} if token else {} + + # -- identifier ---------------------------------------------------------- # + def parse_identifier(self, value: str) -> str: + raw = (value or "").strip() + if not raw: + raise InvalidIdentifierError("Enter a Hugging Face dataset id or URL") + + if "://" in raw or raw.startswith(tuple(HF_HOSTS)): + parsed = urlparse(raw if "://" in raw else f"https://{raw}") + if parsed.netloc.lower() not in HF_HOSTS: + raise InvalidIdentifierError("That is not a huggingface.co URL") + parts = [p for p in parsed.path.split("/") if p] + if not parts or parts[0] != "datasets": + raise InvalidIdentifierError( + "Expected a dataset URL like https://huggingface.co/datasets//" + ) + parts = parts[1:] + # Trim sub-paths such as /tree/main, /blob/main/..., /viewer. + if len(parts) >= 2 and parts[1] not in ( + "tree", + "blob", + "viewer", + "resolve", + "discussions", + ): + repo_id = f"{parts[0]}/{parts[1]}" + elif parts: + repo_id = parts[0] + else: + raise InvalidIdentifierError("Could not find a dataset id in that URL") + else: + repo_id = raw[len("datasets/") :] if raw.startswith("datasets/") else raw + + repo_id = repo_id.strip("/") + check_identifier_length(repo_id, "Hugging Face") + if not _ID_RE.match(repo_id): + raise InvalidIdentifierError( + "Hugging Face ids look like 'namespace/name' (letters, digits, '-', '_', '.')" + ) + return repo_id + + # -- fetch --------------------------------------------------------------- # + def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: + repo_id = self.parse_identifier(identifier) + headers = self._headers() + + try: + meta: Dict[str, Any] = self._get_json( + f"{HF_API}/{repo_id}", + headers=headers, + params=[("expand[]", name) for name in _EXPAND_FIELDS], + ) + except PlatformAuthError as exc: + # Hugging Face answers 401 for repos that do not exist as well as + # for private/gated ones, so say both. + raise PlatformAuthError( + f"'{repo_id}' was not found on Hugging Face, or it is private/gated. " + "Check the spelling; only public datasets can be imported." + ) from exc + meta.pop("siblings", None) # never keep the file list, whatever the API sends + if meta.get("disabled"): + raise PlatformImportError("This dataset has been disabled on Hugging Face") + + # The Hub redirects legacy short names ("imdb") to their canonical id; keep + # the canonical one so the same dataset cannot be imported twice under two names. + repo_id = str(meta.get("id") or repo_id) + + card: Dict[str, Any] = meta.get("cardData") or {} + tags: List[str] = meta.get("tags") or [] + + title = card.get("pretty_name") or repo_id.split("/")[-1] + author = meta.get("author") or (repo_id.split("/")[0] if "/" in repo_id else "") + readme = self._readme(repo_id, meta, headers) + + return PlatformDatasetInfo( + platform=self.platform, + identifier=repo_id, + title=str(title)[:300], + description=shorten(readme, MAX_DESCRIPTION), + source_url=f"{HF_WEB}/{repo_id}", + author=str(author)[:300], + license=self._license(card, tags), + tags=self._tags(tags), + last_updated=parse_iso_datetime(meta.get("lastModified")), + created_at=parse_iso_datetime(meta.get("createdAt")), + revision=str(meta.get("sha") or "")[:64], + readme=cap_readme(readme), + citation=str(meta.get("citation") or "")[:20_000], + languages=self._languages(card, tags), + columns=self._columns(card), + ) + + # -- pieces -------------------------------------------------------------- # + @staticmethod + def _license(card: Dict[str, Any], tags: List[str]) -> str: + lic = card.get("license") + if isinstance(lic, list): + lic = lic[0] if lic else "" + if lic: + return str(lic) + for tag in tags: + if tag.startswith("license:"): + return tag[len("license:") :] + return "" + + @staticmethod + def _tags(tags: List[str]) -> List[str]: + """Keep human-meaningful tags: plain ones plus task/language values.""" + out: List[str] = [] + for tag in tags: + if ":" not in tag: + value = tag + else: + prefix, _, value = tag.partition(":") + if prefix not in ("task_categories", "language", "task_ids"): + continue + value = value.strip() + if value and value not in out: + out.append(value[:50]) + return out[:30] + + def _readme(self, repo_id: str, meta: Dict[str, Any], headers: Dict[str, str]) -> str: + """Full dataset card body (README) without its YAML front matter; else the API field.""" + try: + response = self._get(f"{HF_WEB}/{repo_id}/resolve/main/README.md", headers=headers) + text = response.text + except PlatformImportError: + text = "" + if text: + # Strip YAML front matter. + if text.startswith("---"): + end = text.find("\n---", 3) + if end != -1: + text = text[end + 4 :] + text = text.strip() + if not text: + text = str(meta.get("description") or "") + return text + + @staticmethod + def _languages(card: Dict[str, Any], tags: List[str]) -> List[str]: + langs = card.get("language") + if isinstance(langs, str): + langs = [langs] + out = [str(v).strip() for v in (langs or []) if str(v).strip()] + if not out: + out = [t[len("language:") :] for t in tags if t.startswith("language:")] + return out[:20] + + @staticmethod + def _columns(card: Dict[str, Any]) -> List[PlatformColumn]: + """Column names/types from the card's dataset_info (first config if several).""" + info = card.get("dataset_info") + if isinstance(info, list): + info = info[0] if info else None + if not isinstance(info, dict): + return [] + cols: List[PlatformColumn] = [] + for feat in info.get("features") or []: + if isinstance(feat, dict) and feat.get("name"): + cols.append( + PlatformColumn( + name=str(feat["name"])[:255], field_type=field_type_for(feat.get("dtype")) + ) + ) + return cols[:200] diff --git a/api/services/platform_importers/kaggle.py b/api/services/platform_importers/kaggle.py new file mode 100644 index 0000000..45bbba4 --- /dev/null +++ b/api/services/platform_importers/kaggle.py @@ -0,0 +1,148 @@ +"""Kaggle dataset importer (metadata only). + +One request per import: the dataset *view* endpoint, which answers anonymously +for public datasets and carries everything we store — title, description, +license, tags, owner and last updated. No key is needed. ``KAGGLE_USERNAME`` / +``KAGGLE_KEY`` (settings) are sent as basic auth when configured, which is only +useful for datasets the account can see but the public cannot. +""" + +from __future__ import annotations + +import re +from typing import Any, Dict, List, Optional, Tuple +from urllib.parse import urlparse + +from django.conf import settings + +from api.services.platform_importers.base import ( + InvalidIdentifierError, + PlatformAuthError, + PlatformDatasetInfo, + PlatformImporter, + cap_readme, + check_identifier_length, + parse_iso_datetime, + shorten, +) +from api.utils.enums import ImportPlatform + +KAGGLE_HOSTS = ("www.kaggle.com", "kaggle.com") +KAGGLE_API = "https://www.kaggle.com/api/v1" +KAGGLE_WEB = "https://www.kaggle.com/datasets" + +_SLUG_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_-]*$") +MAX_DESCRIPTION = 1000 + + +class KaggleImporter(PlatformImporter): + platform = ImportPlatform.KAGGLE + label = "Kaggle" + hosts = KAGGLE_HOSTS + + def _auth(self) -> Optional[Tuple[str, str]]: + """Platform-level credentials if configured; None means anonymous.""" + username = getattr(settings, "KAGGLE_USERNAME", None) + key = getattr(settings, "KAGGLE_KEY", None) + return (username, key) if username and key else None + + # -- identifier ---------------------------------------------------------- # + def parse_identifier(self, value: str) -> str: + raw = (value or "").strip() + if not raw: + raise InvalidIdentifierError("Enter a Kaggle dataset ref (owner/dataset) or URL") + + if "://" in raw or raw.startswith(KAGGLE_HOSTS): + parsed = urlparse(raw if "://" in raw else f"https://{raw}") + if parsed.netloc.lower() not in KAGGLE_HOSTS: + raise InvalidIdentifierError("That is not a kaggle.com URL") + parts = [p for p in parsed.path.split("/") if p] + if parts and parts[0] == "datasets": + parts = parts[1:] + if len(parts) < 2: + raise InvalidIdentifierError( + "Expected a dataset URL like https://www.kaggle.com/datasets//" + ) + owner, slug = parts[0], parts[1] + else: + parts = raw.strip("/").split("/") + if len(parts) != 2: + raise InvalidIdentifierError("Kaggle refs look like 'owner/dataset-name'") + owner, slug = parts + + check_identifier_length(f"{owner}/{slug}", "Kaggle") + if not (_SLUG_RE.match(owner) and _SLUG_RE.match(slug)): + raise InvalidIdentifierError( + "Kaggle owner and dataset names use letters, digits, '-' and '_'" + ) + return f"{owner}/{slug}" + + # -- fetch --------------------------------------------------------------- # + def fetch_dataset_info(self, identifier: str) -> PlatformDatasetInfo: + ref = self.parse_identifier(identifier) + owner, slug = ref.split("/") + + try: + meta: Dict[str, Any] = self._get_json( + f"{KAGGLE_API}/datasets/view/{owner}/{slug}", auth=self._auth() + ) + except PlatformAuthError as exc: + # Kaggle answers 403 for datasets that do not exist as well as private ones. + raise PlatformAuthError( + f"'{ref}' was not found on Kaggle, or it is private. " + "Check the spelling; only public datasets can be imported." + ) from exc + if meta.get("isPrivate"): + raise PlatformAuthError("This Kaggle dataset is private") + + title = meta.get("title") or slug + readme = (meta.get("description") or meta.get("subtitle") or "").strip() + version = meta.get("currentVersionNumber") + + return PlatformDatasetInfo( + platform=self.platform, + identifier=ref, + title=str(title)[:300], + description=shorten(readme, MAX_DESCRIPTION), + source_url=meta.get("url") or f"{KAGGLE_WEB}/{ref}", + author=str(meta.get("ownerName") or owner)[:300], + license=self._license(meta), + tags=self._tags(meta), + last_updated=parse_iso_datetime(meta.get("lastUpdated")), + created_at=self._first_version_date(meta), + revision=str(version) if version is not None else "", + readme=cap_readme(readme), + ) + + # -- pieces -------------------------------------------------------------- # + @staticmethod + def _license(meta: Dict[str, Any]) -> str: + licenses = meta.get("licenses") or [] + if licenses and isinstance(licenses[0], dict): + return str(licenses[0].get("name") or "") + return str(meta.get("licenseName") or "") + + @staticmethod + def _tags(meta: Dict[str, Any]) -> List[str]: + out: List[str] = [] + for entry in meta.get("keywords") or []: + if isinstance(entry, str) and entry.strip(): + out.append(entry.strip()[:50]) + for entry in meta.get("tags") or []: + name = entry.get("name") if isinstance(entry, dict) else entry + if isinstance(name, str) and name.strip() and name.strip() not in out: + out.append(name.strip()[:50]) + return out[:30] + + @staticmethod + def _first_version_date(meta: Dict[str, Any]): + """Kaggle has no created date. The earliest version's date is the same thing, + but the view only lists recent versions for datasets with many, so use it + only when the list is complete.""" + versions = [v for v in (meta.get("versions") or []) if isinstance(v, dict)] + current = meta.get("currentVersionNumber") + if not versions or (isinstance(current, int) and len(versions) < current): + return None + dates = [parse_iso_datetime(v.get("creationDate")) for v in versions] + dates = [d for d in dates if d is not None] + return min(dates) if dates else None diff --git a/api/types/type_dataset.py b/api/types/type_dataset.py index b8e71ec..dcc0ed3 100644 --- a/api/types/type_dataset.py +++ b/api/types/type_dataset.py @@ -11,6 +11,7 @@ from api.models import Dataset, DatasetMetadata, PromptDataset, Resource, Tag from api.types.base_type import BaseType from api.types.type_dataset_metadata import TypeDatasetMetadata +from api.types.type_dataset_source import TypeDatasetSource from api.types.type_geo import TypeGeo from api.types.type_organization import TypeOrganization from api.types.type_resource import TypeResource @@ -65,6 +66,12 @@ class TypeDataset(BaseType): download_count: int user: Optional["TypeUser"] + @strawberry.field + def source(self) -> Optional["TypeDatasetSource"]: + """Provenance for datasets imported from a third-party platform (else null).""" + source = getattr(self, "source", None) + return TypeDatasetSource.from_django(source) if source is not None else None + @strawberry.field def sectors(self, info: Info) -> List["TypeSector"]: """Get sectors for this dataset. diff --git a/api/types/type_dataset_source.py b/api/types/type_dataset_source.py new file mode 100644 index 0000000..00c6226 --- /dev/null +++ b/api/types/type_dataset_source.py @@ -0,0 +1,78 @@ +"""GraphQL types for third-party platform imports (link-only).""" + +from datetime import datetime +from typing import List, Optional + +import strawberry +import strawberry_django +from strawberry import auto +from strawberry.enum import EnumType + +from api.models import DatasetSource +from api.services.platform_importers import PlatformDatasetInfo +from api.types.base_type import BaseType +from api.utils.enums import ImportPlatform + +import_platform_enum: EnumType = strawberry.enum(ImportPlatform) # type: ignore + + +@strawberry_django.type(DatasetSource) +class TypeDatasetSource(BaseType): + """Where an imported dataset came from, and what the platform said about it.""" + + id: auto + platform: import_platform_enum # type: ignore + source_identifier: auto + source_url: auto + source_homepage: auto + revision: auto + source_author: auto + source_license: auto + source_readme: auto + citation: auto + source_created_at: auto + source_last_updated: auto + is_archived: auto + imported_at: auto + last_synced_at: auto + + @strawberry.field + def languages(self) -> List[str]: + return list(getattr(self, "languages", None) or []) + + +@strawberry.type +class TypePlatformDatasetPreview: + """What an import *would* create — returned by the preview query, no side effects.""" + + platform: import_platform_enum # type: ignore + identifier: str + title: str + description: str + source_url: str + author: str + license: str + mapped_license: str + tags: List[str] + last_updated: Optional[datetime] + languages: List[str] + revision: str + column_count: int + + @classmethod + def from_info(cls, info: PlatformDatasetInfo) -> "TypePlatformDatasetPreview": + return cls( + platform=ImportPlatform(info.platform), + identifier=info.identifier, + title=info.title, + description=info.description, + source_url=info.source_url, + author=info.author, + license=info.license, + mapped_license=info.mapped_license, + tags=list(info.tags), + last_updated=info.last_updated, + languages=list(info.languages), + revision=info.revision, + column_count=len(info.columns), + ) diff --git a/api/types/type_resource.py b/api/types/type_resource.py index cd6d09c..e77d351 100644 --- a/api/types/type_resource.py +++ b/api/types/type_resource.py @@ -68,6 +68,7 @@ class TypeResource(BaseType): preview_enabled: auto preview_details: Optional[TypePreviewDetails] download_count: auto + url: auto # @strawberry.field # def model_resources(self) -> List[TypeAccessModelResourceFields]: diff --git a/api/urls.py b/api/urls.py index 0d9da9f..fbd64b6 100644 --- a/api/urls.py +++ b/api/urls.py @@ -14,6 +14,7 @@ dataset_data, download, generate_dynamic_chart, + metadata_export, publication_download_view, search_aimodel, search_collaborative, @@ -100,6 +101,16 @@ dataset_data.PromptDatasetDataView.as_view(), name="prompt_dataset_data", ), + path( + "datasets//export/", + metadata_export.metadata_export, + name="dataset_metadata_export", + ), + path( + "metadata/export-options/", + metadata_export.metadata_export_options, + name="metadata_export_options", + ), # Single, simple GraphQL endpoint with no redirects path( "graphql", diff --git a/api/utils/enums.py b/api/utils/enums.py index 646749f..49e1a61 100644 --- a/api/utils/enums.py +++ b/api/utils/enums.py @@ -324,3 +324,11 @@ class EndpointAuthType(models.TextChoices): OAUTH2 = "OAUTH2" CUSTOM = "CUSTOM" NONE = "NONE" + + +class ImportPlatform(models.TextChoices): + """Third-party platforms a dataset can be imported (link-only) from.""" + + KAGGLE = "KAGGLE" + HUGGINGFACE = "HUGGINGFACE" + GITHUB = "GITHUB" diff --git a/api/views/download_view.py b/api/views/download_view.py index cce801a..5d10648 100644 --- a/api/views/download_view.py +++ b/api/views/download_view.py @@ -6,7 +6,7 @@ from asgiref.sync import sync_to_async from django.core.exceptions import ObjectDoesNotExist from django.core.files.uploadedfile import UploadedFile -from django.http import HttpRequest, HttpResponse, JsonResponse +from django.http import HttpRequest, HttpResponse, HttpResponseRedirect, JsonResponse from pyecharts.charts.chart import Chart from pyecharts.render import make_snapshot from selenium import webdriver @@ -16,6 +16,7 @@ from api.models import Resource, ResourceChartDetails, ResourceChartImage from api.types.type_resource_chart import chart_base +from api.utils.enums import DataType @sync_to_async @@ -47,13 +48,21 @@ def get_resource_response( resource: Resource, request: Optional[HttpRequest] = None ) -> HttpResponse: """Get file response for a resource.""" - file_details = resource.resourcefiledetails + # Link-only (platform-imported) resources: we hold no bytes, send the + # user to the file on the source platform. Still counts as a download. + if resource.type == DataType.EXTERNAL: + if not resource.url: + return JsonResponse({"error": "External resource has no URL"}, status=404) + resource.download_count += 1 + resource.save(update_fields=["download_count"]) + _track_download(resource, request) + return HttpResponseRedirect(resource.url) + + file_details = getattr(resource, "resourcefiledetails", None) if not file_details or not file_details.file: return JsonResponse({"error": "File not found"}, status=404) - response = HttpResponse( - file_details.file.read(), content_type="application/octet-stream" - ) + response = HttpResponse(file_details.file.read(), content_type="application/octet-stream") # Handle filename and basename explicitly default_name = f"resource_{resource.name}.csv" @@ -69,7 +78,14 @@ def get_resource_response( resource.download_count += 1 resource.save() - # Track the download activity if the user is authenticated + _track_download(resource, request) + + response["Content-Disposition"] = f'attachment; filename="{basename}"' + return response + + +def _track_download(resource: Resource, request: Optional[HttpRequest]) -> None: + """Record the download in the activity stream for authenticated users.""" if request and hasattr(request, "user") and request.user.is_authenticated: # Import here to avoid circular imports import asyncio @@ -80,9 +96,6 @@ def get_resource_response( sync_to_async(track_resource_downloaded)(request.user, resource, request) ) - response["Content-Disposition"] = f'attachment; filename="{basename}"' - return response - @sync_to_async def get_chart_image_response(chart_image: ResourceChartImage) -> HttpResponse: @@ -90,9 +103,7 @@ def get_chart_image_response(chart_image: ResourceChartImage) -> HttpResponse: if not chart_image.image: return JsonResponse({"error": "File not found"}, status=404) - response = HttpResponse( - chart_image.image.read(), content_type="application/octet-stream" - ) + response = HttpResponse(chart_image.image.read(), content_type="application/octet-stream") # Handle filename and basename explicitly default_name = f"chart_{chart_image.id}.png" @@ -180,9 +191,7 @@ def get_file_chart_image_response(chart_image: ResourceChartImage) -> HttpRespon file_obj.seek(0) # Reset file pointer response = HttpResponse(file_obj, content_type=mime_type) file_name = str(file_obj.name) - response["Content-Disposition"] = ( - f'attachment; filename="{os.path.basename(file_name)}"' - ) + response["Content-Disposition"] = f'attachment; filename="{os.path.basename(file_name)}"' else: response = HttpResponse("File doesn't exist", content_type="text/plain") return response @@ -192,9 +201,7 @@ def get_custom_webdriver() -> WebDriver: """Configure and return a custom Selenium WebDriver.""" chrome_options = Options() chrome_options.add_argument("--no-sandbox") # Bypass OS security model - chrome_options.add_argument( - "--disable-dev-shm-usage" - ) # Overcome limited resource problems + chrome_options.add_argument("--disable-dev-shm-usage") # Overcome limited resource problems chrome_options.add_argument("--headless") # Run headless browser chrome_options.add_argument("--disable-gpu") # Disable GPU for headless browser diff --git a/api/views/metadata_export.py b/api/views/metadata_export.py new file mode 100644 index 0000000..a3df43c --- /dev/null +++ b/api/views/metadata_export.py @@ -0,0 +1,64 @@ +"""GET /api/datasets//export?standard=dcat|croissant|dublin_core&format=jsonld|turtle|rdfxml|ntriples + +Public metadata for a published dataset, generated on request from the +crosswalk contract. Nothing is stored. Add ``report=1`` to receive the gap +report (unresolved vocabulary values, dropped fields, missing mandatory +properties) alongside the document as JSON instead of a file download. +""" + +from __future__ import annotations + +import json +import uuid + +import structlog +from django.http import HttpRequest, HttpResponse, JsonResponse +from django.utils.text import slugify + +from api.models import Dataset +from api.services.metadata_export.exporter import export_dataset, export_options +from api.utils.enums import DatasetStatus + +logger = structlog.get_logger("dataspace.metadata_export") + + +def metadata_export_options(request: HttpRequest) -> JsonResponse: + return JsonResponse({"standards": export_options()}) + + +def metadata_export(request: HttpRequest, dataset_id: uuid.UUID) -> HttpResponse: + try: + dataset = Dataset.objects.select_related("organization", "user").get(id=dataset_id) + except Dataset.DoesNotExist: + return JsonResponse({"error": "Dataset not found"}, status=404) + + # Public endpoint: published datasets only. Owners preview drafts through + # the same view when logged in. + user = getattr(request, "user", None) + is_owner = bool( + user and user.is_authenticated and (dataset.user_id == user.id or user.is_superuser) + ) + if dataset.status != DatasetStatus.PUBLISHED.value and not is_owner: + return JsonResponse({"error": "Dataset not found"}, status=404) + + standard = (request.GET.get("standard") or "dcat").strip().lower() + fmt = (request.GET.get("format") or "jsonld").strip().lower() + try: + body, content_type, ext, report = export_dataset(dataset, standard, fmt) + except ValueError as exc: + return JsonResponse({"error": str(exc), "options": export_options()}, status=400) + except Exception as exc: # pragma: no cover - defensive; never 500 on a public page + logger.error( + "metadata_export_failed", dataset_id=str(dataset_id), standard=standard, error=str(exc) + ) + return JsonResponse({"error": "Could not generate the export"}, status=500) + + if request.GET.get("report") in ("1", "true", "yes"): + payload = {"standard": standard, "format": fmt, "report": report} + payload["document"] = json.loads(body) if fmt == "jsonld" else body + return JsonResponse(payload, json_dumps_params={"ensure_ascii": False}) + + response = HttpResponse(body, content_type=f"{content_type}; charset=utf-8") + filename = f"{slugify(dataset.slug or dataset.title) or 'dataset'}.{standard}.{ext}" + response["Content-Disposition"] = f'attachment; filename="{filename}"' + return response diff --git a/api/views/search_dataset.py b/api/views/search_dataset.py index 680d509..b58b811 100644 --- a/api/views/search_dataset.py +++ b/api/views/search_dataset.py @@ -81,6 +81,7 @@ class DatasetDocumentSerializer(serializers.ModelSerializer): tags = serializers.ListField() sectors = serializers.ListField() formats = serializers.ListField() + source_platform = serializers.CharField(required=False, allow_null=True) catalogs = serializers.ListField() geographies = serializers.ListField() has_charts = serializers.BooleanField() @@ -118,6 +119,7 @@ class Meta: "tags", "sectors", "formats", + "source_platform", "catalogs", "geographies", "has_charts", @@ -175,6 +177,7 @@ def get_searchable_and_aggregations(self) -> Tuple[List[str], Dict[str, str]]: "catalogs.raw": "terms", "geographies.raw": "terms", "dataset_type": "terms", + "source_platform": "terms", } for metadata in enabled_metadata: # type: Metadata if metadata.filterable: @@ -273,6 +276,13 @@ def add_filters(self, filters: Dict[str, str], search: Search) -> Search: elif filter == "dataset_type": # Filter by dataset type (DATA or PROMPT) search = search.filter("term", dataset_type=filters[filter]) + elif filter == "source_platform": + # Filter by import platform (HUGGINGFACE, GITHUB, KAGGLE); "NATIVE" + # selects datasets that were not imported at all. + if filters[filter] == "NATIVE": + search = search.exclude("exists", field="source_platform") + else: + search = search.filter("terms", source_platform=filters[filter].split(",")) elif filter == "task_type": # Filter by prompt task type (nested in prompt_metadata) search = search.filter( diff --git a/api/views/search_unified.py b/api/views/search_unified.py index e718d58..1323f5e 100644 --- a/api/views/search_unified.py +++ b/api/views/search_unified.py @@ -64,6 +64,7 @@ class UserSerializer(serializers.Serializer): # Type-specific fields # Dataset specific formats = serializers.ListField(required=False) + source_platform = serializers.CharField(required=False, allow_null=True) has_charts = serializers.BooleanField(required=False) download_count = serializers.IntegerField(required=False) is_individual_dataset = serializers.BooleanField(required=False) diff --git a/requirements.txt b/requirements.txt index aad4a74..ca9b8bb 100644 --- a/requirements.txt +++ b/requirements.txt @@ -122,3 +122,4 @@ torch==2.9.0 transformers==4.57.1 sentencepiece==0.2.1 accelerate==1.11.0 +rdflib==7.6.0 diff --git a/search/documents/dataset_document.py b/search/documents/dataset_document.py index 83f8d6e..559d5cf 100644 --- a/search/documents/dataset_document.py +++ b/search/documents/dataset_document.py @@ -6,6 +6,7 @@ Catalog, Dataset, DatasetMetadata, + DatasetSource, Geography, Metadata, Organization, @@ -94,6 +95,10 @@ class DatasetDocument(Document): } ) + # Platform this dataset was imported from (KAGGLE, HUGGINGFACE) or null + # for datasets created natively. Lets listings badge/filter imports. + source_platform = fields.KeywordField(attr="source_platform_indexing") + formats = fields.TextField( attr="formats_indexing", analyzer=ngram_analyser, @@ -238,6 +243,8 @@ def get_instances_from_related( """Get Dataset instances from related models.""" if isinstance(related_instance, Resource): return related_instance.dataset + elif isinstance(related_instance, DatasetSource): + return related_instance.dataset elif isinstance(related_instance, Metadata): ds_metadata_objects = related_instance.datasetmetadata_set.all() return [obj.dataset for obj in ds_metadata_objects] # type: ignore @@ -271,6 +278,7 @@ class Django: related_models = [ Resource, + DatasetSource, Metadata, DatasetMetadata, PromptDataset,