Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -12,3 +12,13 @@ URL_WHITELIST=http://localhost:8000,http://localhost,http://localhost:3000
DEBUG=True
SECRET_KEY=your-secret-key
REDIS_URL=redis://redis:6379/1

# Third-party platform imports (optional)
KAGGLE_USERNAME=
KAGGLE_KEY=
HF_TOKEN=
GITHUB_TOKEN=

# URLs written into exported metadata (DCAT / Croissant); set per environment
PUBLIC_SITE_URL=https://civicdataspace.in
PUBLIC_API_URL=https://api.civicdataspace.in
15 changes: 15 additions & 0 deletions DataSpace/settings.py
Original file line number Diff line number Diff line change
Expand Up @@ -289,6 +289,21 @@
}


# Third-party platform imports (link-only). All three platforms work with no
# key for public datasets. KAGGLE_* adds a file count to Kaggle imports,
# HF_TOKEN unlocks gated Hugging Face repos, GITHUB_TOKEN lifts GitHub's
# anonymous rate limit.
KAGGLE_USERNAME = os.getenv("KAGGLE_USERNAME", None)
KAGGLE_KEY = os.getenv("KAGGLE_KEY", None)
HF_TOKEN = os.getenv("HF_TOKEN", None)
GITHUB_TOKEN = os.getenv("GITHUB_TOKEN", None) # optional, lifts the 60 req/hour anonymous limit
PLATFORM_IMPORT_TIMEOUT = float(os.getenv("PLATFORM_IMPORT_TIMEOUT", "15"))

# Absolute URLs written into exported metadata documents (dataset landing page,
# download links). Override on any environment that is not production.
PUBLIC_SITE_URL = os.getenv("PUBLIC_SITE_URL", "https://civicdataspace.in")
PUBLIC_API_URL = os.getenv("PUBLIC_API_URL", "https://api.civicdataspace.in")

# DVC settings
DVC_REPO_PATH = os.path.join(BASE_DIR, "dvc")
DVC_REMOTE_NAME = os.getenv("DVC_REMOTE_NAME", None)
Expand Down
81 changes: 81 additions & 0 deletions api/migrations/0048_platform_import.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
# Generated by Django 5.0.4 on 2026-09-23 20:36

import uuid

import django.db.models.deletion
from django.conf import settings
from django.db import migrations, models


class Migration(migrations.Migration):

dependencies = [
("api", "0047_resourcetype_publication_collaborative_publications_and_more"),
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
]

operations = [
migrations.CreateModel(
name="DatasetSource",
fields=[
(
"id",
models.UUIDField(
default=uuid.uuid4, editable=False, primary_key=True, serialize=False
),
),
(
"platform",
models.CharField(
choices=[
("KAGGLE", "Kaggle"),
("HUGGINGFACE", "Huggingface"),
("GITHUB", "Github"),
],
max_length=50,
),
),
("source_identifier", models.CharField(max_length=300)),
("source_url", models.URLField(max_length=500)),
("source_homepage", models.URLField(blank=True, max_length=500)),
("revision", models.CharField(blank=True, max_length=64)),
("source_author", models.CharField(blank=True, max_length=300)),
("source_license", models.CharField(blank=True, max_length=300)),
("source_readme", models.TextField(blank=True)),
("citation", models.TextField(blank=True)),
("languages", models.JSONField(blank=True, default=list)),
("source_created_at", models.DateTimeField(blank=True, null=True)),
("source_last_updated", models.DateTimeField(blank=True, null=True)),
("is_archived", models.BooleanField(default=False)),
("imported_at", models.DateTimeField(auto_now_add=True)),
("last_synced_at", models.DateTimeField(auto_now=True)),
(
"dataset",
models.OneToOneField(
on_delete=django.db.models.deletion.CASCADE,
related_name="source",
to="api.dataset",
),
),
(
"imported_by",
models.ForeignKey(
blank=True,
null=True,
on_delete=django.db.models.deletion.SET_NULL,
related_name="imported_dataset_sources",
to=settings.AUTH_USER_MODEL,
),
),
],
options={
"db_table": "dataset_source",
"indexes": [
models.Index(
fields=["platform", "source_identifier"],
name="dataset_sou_platfor_38ca23_idx",
)
],
},
),
]
26 changes: 17 additions & 9 deletions api/models/Dataset.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
import uuid
from typing import TYPE_CHECKING, Any
from typing import TYPE_CHECKING, Any, Optional

from django.db import models
from django.db.models import Sum
Expand Down Expand Up @@ -124,14 +124,22 @@ def formats_indexing(self) -> list[str]:

Used in Elasticsearch indexing.
"""
return list(
set(
[
resource.resourcefiledetails.format # type: ignore
for resource in self.resources.all()
]
).difference({""})
)
formats: set[str] = set()
for resource in self.resources.all():
# Link-only (EXTERNAL) resources have no file details; skip them.
file_details = getattr(resource, "resourcefiledetails", None)
if file_details is not None and file_details.format:
formats.add(file_details.format)
return list(formats)

@property
def source_platform_indexing(self) -> Optional[str]:
"""Platform this dataset was imported from, or None for native datasets.

Used in Elasticsearch indexing.
"""
source = getattr(self, "source", None)
return source.platform if source is not None else None

@property
def catalogs_indexing(self) -> list[str]:
Expand Down
69 changes: 69 additions & 0 deletions api/models/DatasetSource.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
import uuid

from django.db import models

from api.utils.enums import ImportPlatform


class DatasetSource(models.Model):
"""Provenance record for a dataset imported from a third-party platform.

Imports are link-only: DataSpace never copies the platform's files. This
row holds what the platform told us about the dataset, in typed columns.
Every column here has a reader: either a metadata standard on export
(DCAT / Croissant / Dublin Core) or the platform itself (attribution,
duplicate detection, licence review). No raw payload is kept.
"""

id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False)
dataset = models.OneToOneField("api.Dataset", on_delete=models.CASCADE, related_name="source")
platform = models.CharField(max_length=50, choices=ImportPlatform.choices)

# --- identity on the platform ------------------------------------------
# Platform-native identifier, e.g. "owner/dataset-slug" (Kaggle) or
# "namespace/name" (Hugging Face). Normalised by the importer.
source_identifier = models.CharField(max_length=300)
# Human-facing page on the platform. Export: schema:sameAs / prov:wasDerivedFrom.
source_url = models.URLField(max_length=500)
# Home page declared by the source, if any (GitHub `homepage`). Export: dcat:landingPage.
source_homepage = models.URLField(max_length=500, blank=True)
# Commit hash (Hugging Face / GitHub) or version number (Kaggle) at import time.
# Export: Croissant `version`. Later: what a sync compares against.
revision = models.CharField(max_length=64, blank=True)

# --- descriptive metadata the standards read ----------------------------
# Who made the data on the platform. Export: dcterms:creator / Croissant creator.
source_author = models.CharField(max_length=300, blank=True)
# License string exactly as the platform reported it (may not map onto
# DatasetLicense; the mapped value lives on Dataset.license).
source_license = models.CharField(max_length=300, blank=True)
# Full dataset card / README. Dataset.description keeps a 1,000-char cut.
source_readme = models.TextField(blank=True)
# BibTeX or free-text citation, when the platform provides one. Export: Croissant citeAs.
citation = models.TextField(blank=True)
# Language codes of the data, e.g. ["en", "hi"]. Export: dcterms:language / inLanguage.
languages = models.JSONField(default=list, blank=True)
# When the dataset was first published on the platform. Export: dcterms:issued.
source_created_at = models.DateTimeField(null=True, blank=True)
# When the platform last changed it. Export: dcterms:modified.
source_last_updated = models.DateTimeField(null=True, blank=True)
# Source is frozen / read-only upstream (GitHub `archived`). Shown as a hint.
is_archived = models.BooleanField(default=False)

# --- our side -------------------------------------------------------------
imported_by = models.ForeignKey(
"authorization.User",
on_delete=models.SET_NULL,
null=True,
blank=True,
related_name="imported_dataset_sources",
)
imported_at = models.DateTimeField(auto_now_add=True)
last_synced_at = models.DateTimeField(auto_now=True)

class Meta:
db_table = "dataset_source"
indexes = [models.Index(fields=["platform", "source_identifier"])]

def __str__(self) -> str:
return f"{self.platform}:{self.source_identifier}"
1 change: 1 addition & 0 deletions api/models/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
)
from api.models.Dataset import Dataset, Tag
from api.models.DatasetMetadata import DatasetMetadata
from api.models.DatasetSource import DatasetSource
from api.models.DataSpace import DataSpace
from api.models.Geography import Geography
from api.models.Metadata import Metadata
Expand Down
99 changes: 99 additions & 0 deletions api/schema/platform_import_schema.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,99 @@
"""GraphQL surface for link-only imports from third-party platforms.

- ``preview_platform_dataset`` fetches normalised metadata, no side effects.
- ``import_platform_dataset`` creates a DRAFT dataset with one EXTERNAL resource
linking to the dataset page on the platform (files are not imported).

Both take the same organization/dataspace request headers as ``add_dataset``;
the imported dataset is owned the same way a manually created one would be.
"""

from typing import Optional

import strawberry
from strawberry.types import Info

from api.schema.base_mutation import (
BaseMutation,
GraphQLValidationError,
MutationResponse,
)
from api.services.platform_import_service import (
import_platform_dataset,
preview_platform_dataset,
)
from api.services.platform_importers import PlatformImportError
from api.types.type_dataset import TypeDataset
from api.types.type_dataset_source import (
TypePlatformDatasetPreview,
import_platform_enum,
)
from api.utils.graphql_telemetry import trace_resolver
from authorization.graphql_permissions import IsAuthenticated
from authorization.permissions import CreateDatasetPermission


@strawberry.input
class ImportPlatformDatasetInput:
platform: import_platform_enum # type: ignore
#: Short id ("owner/name") or a pasted platform URL.
identifier: str
#: Optional display title on DataSpace; defaults to the platform's title.
title: Optional[str] = None


@strawberry.type
class Query:
@strawberry.field(permission_classes=[IsAuthenticated])
@trace_resolver(name="preview_platform_dataset", attributes={"component": "platform_import"})
def preview_platform_dataset(
self, info: Info, platform: import_platform_enum, identifier: str # type: ignore
) -> TypePlatformDatasetPreview:
"""Look up a Hugging Face / GitHub / Kaggle dataset and show what an import would create."""
try:
data = preview_platform_dataset(platform.value, identifier)
except PlatformImportError as exc:
# Surface the importer's user-safe message as a GraphQL error.
raise ValueError(exc.message) from exc
return TypePlatformDatasetPreview.from_info(data)


@strawberry.type
class Mutation:
@strawberry.mutation
@BaseMutation.mutation(
permission_classes=[IsAuthenticated, CreateDatasetPermission],
trace_name="import_platform_dataset",
trace_attributes={"component": "platform_import"},
track_activity={
"verb": "imported",
"get_data": lambda result, import_input=None, **kwargs: {
"dataset_id": str(result.id),
"dataset_title": result.title,
"platform": import_input.platform.value if import_input else None,
"identifier": import_input.identifier if import_input else None,
"organization": (str(result.organization.id) if result.organization else None),
},
},
)
def import_platform_dataset(
self, info: Info, import_input: ImportPlatformDatasetInput
) -> MutationResponse[TypeDataset]:
"""Create a DRAFT dataset that links to the dataset on the platform."""
organization = info.context.context.get("organization")
dataspace = info.context.context.get("dataspace")
user = info.context.user

try:
dataset = import_platform_dataset(
platform=import_input.platform.value,
identifier=import_input.identifier,
user=user,
organization=organization,
dataspace=dataspace,
title=import_input.title,
)
except PlatformImportError as exc:
return MutationResponse.error_response(GraphQLValidationError.from_message(exc.message))

return MutationResponse.success_response(TypeDataset.from_django(dataset))
3 changes: 3 additions & 0 deletions api/schema/schema.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@
import api.schema.metadata_schema
import api.schema.organization_data_schema
import api.schema.organization_schema
import api.schema.platform_import_schema
import api.schema.publication_schema
import api.schema.resource_chart_schema
import api.schema.resource_schema
Expand Down Expand Up @@ -77,6 +78,7 @@ def tags(self, info: Info) -> List[TypeTag]:
api.schema.user_schema.Query,
api.schema.collaborative_schema.Query,
api.schema.publication_schema.Query,
api.schema.platform_import_schema.Query,
AuthQuery,
),
)
Expand All @@ -100,6 +102,7 @@ def tags(self, info: Info) -> List[TypeTag]:
api.schema.tags_schema.Mutation,
api.schema.collaborative_schema.Mutation,
api.schema.publication_schema.Mutation,
api.schema.platform_import_schema.Mutation,
AuthMutation,
),
)
Expand Down
Empty file.
Loading
Loading