Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -12,3 +12,9 @@ URL_WHITELIST=http://localhost:8000,http://localhost,http://localhost:3000
DEBUG=True
SECRET_KEY=your-secret-key
REDIS_URL=redis://redis:6379/1

# Third-party platform imports (optional)
KAGGLE_USERNAME=
KAGGLE_KEY=
HF_TOKEN=
GITHUB_TOKEN=
10 changes: 10 additions & 0 deletions DataSpace/settings.py
Original file line number Diff line number Diff line change
Expand Up @@ -289,6 +289,16 @@
}


# Third-party platform imports (link-only). All three platforms work with no
# key for public datasets. KAGGLE_* adds a file count to Kaggle imports,
# HF_TOKEN unlocks gated Hugging Face repos, GITHUB_TOKEN lifts GitHub's
# anonymous rate limit.
KAGGLE_USERNAME = os.getenv("KAGGLE_USERNAME", None)
KAGGLE_KEY = os.getenv("KAGGLE_KEY", None)
HF_TOKEN = os.getenv("HF_TOKEN", None)
GITHUB_TOKEN = os.getenv("GITHUB_TOKEN", None) # optional, lifts the 60 req/hour anonymous limit
PLATFORM_IMPORT_TIMEOUT = float(os.getenv("PLATFORM_IMPORT_TIMEOUT", "15"))

# DVC settings
DVC_REPO_PATH = os.path.join(BASE_DIR, "dvc")
DVC_REMOTE_NAME = os.getenv("DVC_REMOTE_NAME", None)
Expand Down
43 changes: 43 additions & 0 deletions api/management/commands/backfill_file_hashes.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
"""Compute the SHA-256 of resource files uploaded before the column existed.

python manage.py backfill_file_hashes # fill every missing hash
python manage.py backfill_file_hashes --dry-run # only count

Writes through a queryset update, so the DVC versioning signal does not run.
"""

from django.core.management.base import BaseCommand

from api.models import ResourceFileDetails
from api.models.Resource import compute_sha256


class Command(BaseCommand):
help = "Fill ResourceFileDetails.sha256 for files that have none."

def add_arguments(self, parser):
parser.add_argument("--dry-run", action="store_true", help="Report only; write nothing.")

def handle(self, *args, **options):
pending = (
ResourceFileDetails.objects.filter(sha256__isnull=True)
.exclude(file="")
.select_related("resource")
.order_by("id")
)
total = pending.count()
self.stdout.write(f"{total} file(s) without a hash")
if options["dry_run"]:
return
done = failed = 0
for details in pending.iterator():
digest = compute_sha256(details.file)
if digest is None:
failed += 1
self.stderr.write(
f" skipped {details.resource_id}: file unreadable ({details.file.name})"
)
continue
ResourceFileDetails.objects.filter(pk=details.pk).update(sha256=digest)
done += 1
self.stdout.write(self.style.SUCCESS(f"hashed {done}, skipped {failed}"))
81 changes: 81 additions & 0 deletions api/migrations/0048_platform_import.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
# Generated by Django 5.0.4 on 2026-09-23 20:36

import uuid

import django.db.models.deletion
from django.conf import settings
from django.db import migrations, models


class Migration(migrations.Migration):

dependencies = [
("api", "0047_resourcetype_publication_collaborative_publications_and_more"),
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
]

operations = [
migrations.CreateModel(
name="DatasetSource",
fields=[
(
"id",
models.UUIDField(
default=uuid.uuid4, editable=False, primary_key=True, serialize=False
),
),
(
"platform",
models.CharField(
choices=[
("KAGGLE", "Kaggle"),
("HUGGINGFACE", "Huggingface"),
("GITHUB", "Github"),
],
max_length=50,
),
),
("source_identifier", models.CharField(max_length=300)),
("source_url", models.URLField(max_length=500)),
("source_homepage", models.URLField(blank=True, max_length=500)),
("revision", models.CharField(blank=True, max_length=64)),
("source_author", models.CharField(blank=True, max_length=300)),
("source_license", models.CharField(blank=True, max_length=300)),
("source_readme", models.TextField(blank=True)),
("citation", models.TextField(blank=True)),
("languages", models.JSONField(blank=True, default=list)),
("source_created_at", models.DateTimeField(blank=True, null=True)),
("source_last_updated", models.DateTimeField(blank=True, null=True)),
("is_archived", models.BooleanField(default=False)),
("imported_at", models.DateTimeField(auto_now_add=True)),
("last_synced_at", models.DateTimeField(auto_now=True)),
(
"dataset",
models.OneToOneField(
on_delete=django.db.models.deletion.CASCADE,
related_name="source",
to="api.dataset",
),
),
(
"imported_by",
models.ForeignKey(
blank=True,
null=True,
on_delete=django.db.models.deletion.SET_NULL,
related_name="imported_dataset_sources",
to=settings.AUTH_USER_MODEL,
),
),
],
options={
"db_table": "dataset_source",
"indexes": [
models.Index(
fields=["platform", "source_identifier"],
name="dataset_sou_platfor_38ca23_idx",
)
],
},
),
]
21 changes: 21 additions & 0 deletions api/migrations/0049_resourcefiledetails_sha256.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
from django.db import migrations, models


class Migration(migrations.Migration):

dependencies = [
("api", "0048_platform_import"),
]

operations = [
migrations.AddField(
model_name="resourcefiledetails",
name="sha256",
field=models.CharField(
blank=True,
help_text="SHA-256 of the file contents, computed when the file is saved.",
max_length=64,
null=True,
),
),
]
26 changes: 17 additions & 9 deletions api/models/Dataset.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
import uuid
from typing import TYPE_CHECKING, Any
from typing import TYPE_CHECKING, Any, Optional

from django.db import models
from django.db.models import Sum
Expand Down Expand Up @@ -124,14 +124,22 @@ def formats_indexing(self) -> list[str]:

Used in Elasticsearch indexing.
"""
return list(
set(
[
resource.resourcefiledetails.format # type: ignore
for resource in self.resources.all()
]
).difference({""})
)
formats: set[str] = set()
for resource in self.resources.all():
# Link-only (EXTERNAL) resources have no file details; skip them.
file_details = getattr(resource, "resourcefiledetails", None)
if file_details is not None and file_details.format:
formats.add(file_details.format)
return list(formats)

@property
def source_platform_indexing(self) -> Optional[str]:
"""Platform this dataset was imported from, or None for native datasets.

Used in Elasticsearch indexing.
"""
source = getattr(self, "source", None)
return source.platform if source is not None else None

@property
def catalogs_indexing(self) -> list[str]:
Expand Down
69 changes: 69 additions & 0 deletions api/models/DatasetSource.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
import uuid

from django.db import models

from api.utils.enums import ImportPlatform


class DatasetSource(models.Model):
"""Provenance record for a dataset imported from a third-party platform.

Imports are link-only: DataSpace never copies the platform's files. This
row holds what the platform told us about the dataset, in typed columns.
Every column here has a reader: either a metadata standard on export
(DCAT / Croissant / Dublin Core) or the platform itself (attribution,
duplicate detection, licence review). No raw payload is kept.
"""

id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False)
dataset = models.OneToOneField("api.Dataset", on_delete=models.CASCADE, related_name="source")
platform = models.CharField(max_length=50, choices=ImportPlatform.choices)

# --- identity on the platform ------------------------------------------
# Platform-native identifier, e.g. "owner/dataset-slug" (Kaggle) or
# "namespace/name" (Hugging Face). Normalised by the importer.
source_identifier = models.CharField(max_length=300)
# Human-facing page on the platform. Export: schema:sameAs / prov:wasDerivedFrom.
source_url = models.URLField(max_length=500)
# Home page declared by the source, if any (GitHub `homepage`). Export: dcat:landingPage.
source_homepage = models.URLField(max_length=500, blank=True)
# Commit hash (Hugging Face / GitHub) or version number (Kaggle) at import time.
# Export: Croissant `version`. Later: what a sync compares against.
revision = models.CharField(max_length=64, blank=True)

# --- descriptive metadata the standards read ----------------------------
# Who made the data on the platform. Export: dcterms:creator / Croissant creator.
source_author = models.CharField(max_length=300, blank=True)
# License string exactly as the platform reported it (may not map onto
# DatasetLicense; the mapped value lives on Dataset.license).
source_license = models.CharField(max_length=300, blank=True)
# Full dataset card / README. Dataset.description keeps a 1,000-char cut.
source_readme = models.TextField(blank=True)
# BibTeX or free-text citation, when the platform provides one. Export: Croissant citeAs.
citation = models.TextField(blank=True)
# Language codes of the data, e.g. ["en", "hi"]. Export: dcterms:language / inLanguage.
languages = models.JSONField(default=list, blank=True)
# When the dataset was first published on the platform. Export: dcterms:issued.
source_created_at = models.DateTimeField(null=True, blank=True)
# When the platform last changed it. Export: dcterms:modified.
source_last_updated = models.DateTimeField(null=True, blank=True)
# Source is frozen / read-only upstream (GitHub `archived`). Shown as a hint.
is_archived = models.BooleanField(default=False)

# --- our side -------------------------------------------------------------
imported_by = models.ForeignKey(
"authorization.User",
on_delete=models.SET_NULL,
null=True,
blank=True,
related_name="imported_dataset_sources",
)
imported_at = models.DateTimeField(auto_now_add=True)
last_synced_at = models.DateTimeField(auto_now=True)

class Meta:
db_table = "dataset_source"
indexes = [models.Index(fields=["platform", "source_identifier"])]

def __str__(self) -> str:
return f"{self.platform}:{self.source_identifier}"
47 changes: 46 additions & 1 deletion api/models/Resource.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
import hashlib
import os
import random
import uuid
Expand All @@ -8,7 +9,7 @@
from django.conf import settings
from django.contrib.auth import get_user_model
from django.db import models
from django.db.models.signals import post_save
from django.db.models.signals import post_save, pre_save
from django.dispatch import receiver
from django.utils.text import slugify

Expand Down Expand Up @@ -74,6 +75,12 @@ class ResourceFileDetails(models.Model):
resource = models.OneToOneField(Resource, on_delete=models.CASCADE, null=False, blank=False)
file = models.FileField(upload_to="resources/", max_length=300)
size = models.FloatField(blank=True, null=True)
sha256 = models.CharField(
max_length=64,
blank=True,
null=True,
help_text="SHA-256 of the file contents, computed when the file is saved.",
)
created = models.DateTimeField(auto_now_add=True)
modified = models.DateTimeField(auto_now=True)
format = models.CharField(max_length=50)
Expand Down Expand Up @@ -123,6 +130,44 @@ class Meta:
db_table = "resource_version"


def compute_sha256(fieldfile: Any) -> Optional[str]:
"""Stream the file once and return its SHA-256 hex digest, or None if unreadable.

Works both for a file already in storage and for an upload that has not
been written yet (the position is reset afterwards so the write still
starts at byte 0).
"""
if not fieldfile:
return None
digest = hashlib.sha256()
try:
fieldfile.open("rb")
for chunk in fieldfile.chunks():
digest.update(chunk)
fieldfile.seek(0)
except Exception as exc: # a missing or unreadable file must never block a save
logger.warning(f"Could not hash resource file {getattr(fieldfile, 'name', '')}: {exc}")
return None
return digest.hexdigest()


@receiver(pre_save, sender=ResourceFileDetails)
def hash_resource_file(sender, instance: ResourceFileDetails, **kwargs):
"""Keep ``sha256`` in step with the file. Recomputed only when the file changes."""
if not instance.file:
instance.sha256 = None
return
if instance.pk and instance.sha256:
stored_name = (
ResourceFileDetails.objects.filter(pk=instance.pk)
.values_list("file", flat=True)
.first()
)
if stored_name == instance.file.name:
return # same file as before
instance.sha256 = compute_sha256(instance.file)


@receiver(post_save, sender=ResourceFileDetails)
def version_resource_with_dvc(sender, instance: ResourceFileDetails, created, **kwargs):
"""Create a new version using DVC when resource is updated"""
Expand Down
1 change: 1 addition & 0 deletions api/models/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
)
from api.models.Dataset import Dataset, Tag
from api.models.DatasetMetadata import DatasetMetadata
from api.models.DatasetSource import DatasetSource
from api.models.DataSpace import DataSpace
from api.models.Geography import Geography
from api.models.Metadata import Metadata
Expand Down
Loading
Loading