Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
152 changes: 152 additions & 0 deletions src/test/test_link_preview.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,152 @@
"""Unit tests for the link-preview page (get_preview / get_preview_image)."""

import html
import re

import pytest

from vfbquery import link_preview


def _neuron():
return {
"Id": "VFB_jrchjrch",
"Name": "5-HTPLP01_R",
"IsIndividual": True,
"IsClass": False,
"Meta": {
"Description": "",
"Comment": "tracing status-Roughly traced, cropped-False",
"Types": "[adult glutamatergic neuron](FBbt_00058220);[adult serotonergic PLP neuron](FBbt_00110925)",
},
"Technique": ["Hemibrain-Reconstruction"],
"Tags": ["Adult", "Nervous_system"],
"Images": {
"VFB_00101567": [
{"id": "VFB_00101384", "thumbnail": "https://www.virtualflybrain.org/data/VFB/i/jrch/jrch/VFB_00101384/thumbnail.png"},
{"id": "VFB_00101385", "thumbnail": "https://x/second.png"},
]
},
}


def _class_with_example():
return {
"Id": "FBbt_00003748",
"Name": "medulla",
"IsClass": True,
"IsIndividual": False,
"Meta": {"Description": "The second optic neuropil, [see](FBbt_00003701) *the* lobula."},
"Images": {},
"Examples": {"VFB_00017894": [{"id": "VFB_00030624", "thumbnail": "https://x/medulla.png"}]},
}


# --- id validation -----------------------------------------------------------

@pytest.mark.parametrize("value", ["FBbt_00003748", "VFB_jrchjrch", "VFBexp_FBtp0000001", "GO_0001872"])
def test_is_term_id_accepts_vfb_shapes(value):
assert link_preview.is_term_id(value)


@pytest.mark.parametrize("value", ["", None, "medulla", "<script>", "../x", "VFB jrch", "_leading"])
def test_is_term_id_rejects_free_text(value):
assert not link_preview.is_term_id(value)


# --- text helpers (mirror pageMetadata.js) -----------------------------------

def test_strip_markdown_keeps_link_labels_and_collapses_space():
assert link_preview.strip_markdown("[medulla](FBbt_1) is *big* and `odd`") == "medulla is big and odd"


def test_truncate_breaks_on_a_word_and_adds_ellipsis():
out = link_preview.truncate("word " * 100, 50)
assert len(out) <= 50
assert out.endswith("…")
assert not out[:-1].endswith(" ")


def test_truncate_leaves_short_text_alone():
assert link_preview.truncate("short", 50) == "short"


# --- thumbnail selection -----------------------------------------------------

def test_first_thumbnail_prefers_images_over_examples():
info = _neuron()
info["Examples"] = {"T": [{"thumbnail": "https://x/example.png"}]}
assert link_preview.first_thumbnail(info).endswith("VFB_00101384/thumbnail.png")


def test_first_thumbnail_falls_back_to_examples_for_a_class():
assert link_preview.first_thumbnail(_class_with_example()) == "https://x/medulla.png"


def test_first_thumbnail_none_when_nothing_has_one():
assert link_preview.first_thumbnail({"Images": {"T": [{"id": "x"}]}, "Examples": {}}) is None


# --- title / description -----------------------------------------------------

def test_title_has_name_id_and_site():
assert link_preview.build_title(_neuron()) == "5-HTPLP01_R [VFB_jrchjrch] - Virtual Fly Brain"


def test_description_for_individual_without_description_uses_types_comment_technique():
d = link_preview.build_description(_neuron())
assert d.startswith("5-HTPLP01_R is an instance of adult glutamatergic neuron, adult serotonergic PLP neuron.")
assert "tracing status-Roughly traced, cropped-False." in d
assert "Imaged by Hemibrain-Reconstruction." in d
assert d.endswith("for this image on Virtual Fly Brain.")


def test_description_for_class_uses_description_only():
d = link_preview.build_description(_class_with_example())
assert d == "The second optic neuropil, see the lobula."


def test_description_never_exceeds_limit():
info = _class_with_example()
info["Meta"]["Description"] = "long words " * 100
assert len(link_preview.build_description(info)) <= link_preview.MAX_DESCRIPTION


# --- metadata / page ---------------------------------------------------------

def test_metadata_urls_are_reports_canonical_and_viewer_target():
m = link_preview.preview_metadata(_neuron())
assert m["url"] == "https://virtualflybrain.org/reports/VFB_jrchjrch"
assert m["viewer_url"] == "https://v2.virtualflybrain.org/org.geppetto.frontend/geppetto?id=VFB_jrchjrch"
assert m["image"].endswith("VFB_00101384/thumbnail.png")


def test_metadata_image_defaults_to_logo():
info = _class_with_example()
info["Examples"] = {}
assert link_preview.preview_metadata(info)["image"] == link_preview.DEFAULT_IMAGE


def _tags(page):
return dict(re.findall(r'<meta (?:property|name)="([^"]+)" content="([^"]*)"', page))


def test_page_carries_every_tag_an_unfurler_reads():
page = link_preview.render_preview_html(_neuron())
tags = _tags(page)
for key in ("og:title", "og:description", "og:url", "og:image", "twitter:card", "twitter:image", "description"):
assert key in tags, key
assert tags["twitter:card"] == "summary_large_image"
assert tags["og:image"] == tags["twitter:image"]
assert 'rel="canonical" href="https://virtualflybrain.org/reports/VFB_jrchjrch"' in page
assert 'http-equiv="refresh" content="0; url=https://v2.virtualflybrain.org/org.geppetto.frontend/geppetto?id=VFB_jrchjrch"' in page


def test_page_escapes_hostile_ontology_text():
info = _class_with_example()
info["Name"] = 'x"><script>alert(1)</script>'
info["Meta"]["Description"] = "a & b < c"
page = link_preview.render_preview_html(info)
assert "<script>alert" not in page
assert html.escape(info["Name"], quote=True) in page
assert "a &amp; b &lt; c" in page
57 changes: 57 additions & 0 deletions src/vfbquery/ha_api.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,8 @@

import aiohttp
from aiohttp import web

from vfbquery import link_preview
import numpy as np

# Pure standard-library module: the /combine expression language, its set
Expand Down Expand Up @@ -4182,6 +4184,59 @@ async def handle_catmaid_command(request):
# Application factory
# ---------------------------------------------------------------------------


async def _term_info_for_preview(request, short_form):
"""The cached get_term_info result for a preview endpoint, or None.

Goes through the same L1 cache, coalescer and worker pool as
/get_term_info itself, so a preview never costs a second term lookup and
a burst of unfurlers for one link collapses to one query.
"""
key = f"term_info:{short_form}"
response = await _dispatch_to_pool(request, key, _run_term_info, short_form)
if response.status != 200:
return None
info = json.loads(response.body)
return info if info and info.get("Id") else None


async def handle_get_preview(request):
"""GET /get_preview?id=<short_form>

The term's link-preview page: og:*/twitter:* tags, title, description
and thumbnail rendered server-side, with a meta-refresh into the viewer.
Meant to be served to unfurlers (Slackbot, Twitterbot, ...) in place of
the viewer, which sets those tags only from JavaScript they never run.
"""
short_form = (request.query.get("id") or "").strip()
if not link_preview.is_term_id(short_form):
return web.Response(text="Error: a VFB term id is required", status=400)
info = await _term_info_for_preview(request, short_form)
if info is None:
return web.Response(text=f"No term found for id={short_form}", status=404)
return web.Response(
text=link_preview.render_preview_html(info),
content_type="text/html",
headers={"Cache-Control": "public, max-age=86400"},
)


async def handle_get_preview_image(request):
"""GET /get_preview_image?id=<short_form>

Redirects to the term's thumbnail, or the site logo when it has none:
a stable id-keyed image URL for og:image, sitemaps, cards and the MCP.
"""
short_form = (request.query.get("id") or "").strip()
if not link_preview.is_term_id(short_form):
return web.Response(text="Error: a VFB term id is required", status=400)
info = await _term_info_for_preview(request, short_form)
if info is None:
return web.Response(text=f"No term found for id={short_form}", status=404)
target = link_preview.first_thumbnail(info) or link_preview.DEFAULT_IMAGE
raise web.HTTPFound(target, headers={"Cache-Control": "public, max-age=86400"})


def create_app(max_workers=None, max_concurrent=None, max_queue_depth=None,
cache_ttl=None, search_concurrency=None, search_cpu_threads=None,
search_queue_wait=None):
Expand Down Expand Up @@ -4238,6 +4293,8 @@ def create_app(max_workers=None, max_concurrent=None, max_queue_depth=None,
app.router.add_get("/get_known_neurotransmitters", handle_get_known_neurotransmitters)
app.router.add_get("/get_hierarchy", handle_get_hierarchy)
app.router.add_get("/get_hierarchy_html", handle_get_hierarchy_html)
app.router.add_get("/get_preview", handle_get_preview)
app.router.add_get("/get_preview_image", handle_get_preview_image)

# Canonical free-text search (website-equivalent ranking)
app.router.add_get("/search", handle_search)
Expand Down
148 changes: 148 additions & 0 deletions src/vfbquery/link_preview.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,148 @@
"""Link-preview page for a VFB term: the HTML an unfurler sees.

Slack, Bluesky, Teams, Discord and the rest build a link card from the
``og:*`` / ``twitter:*`` tags in a page's raw HTML; none of them run
JavaScript. The Geppetto viewer sets those tags per term only from JS, so a
pasted viewer link unfurls as the site logo. This renders the same tags
server-side from ``get_term_info`` so a proxy can hand them to an unfurler
instead -- and a human who lands here anyway is sent on to the viewer.

The title and description logic mirrors ``pageMetadata.js`` in geppetto-vfb
so a term reads the same wherever its preview is built.
"""

import html
import re

SITE_NAME = "Virtual Fly Brain"
VIEWER_BASE = "https://v2.virtualflybrain.org/org.geppetto.frontend/geppetto"
REPORTS_BASE = "https://virtualflybrain.org/reports/"
DEFAULT_IMAGE = "https://www.virtualflybrain.org/favicons/vfb-logo-512.png"
MAX_DESCRIPTION = 300

_MARKDOWN_LINK = re.compile(r"\[([^\]]*)\]\([^)\s]*\)")
_ID_PATTERN = re.compile(r"^[A-Za-z][A-Za-z0-9]*_[A-Za-z0-9_]+$")


def is_term_id(value):
"""True for the id shapes VFB uses (FBbt_00003748, VFB_jrchjrch, ...)."""
return bool(value) and bool(_ID_PATTERN.match(value))


def strip_markdown(text):
if not isinstance(text, str):
return ""
text = _MARKDOWN_LINK.sub(r"\1", text)
text = re.sub(r"[*`]", "", text)
return re.sub(r"\s+", " ", text).strip()


def truncate(text, limit=MAX_DESCRIPTION):
if len(text) <= limit:
return text
cut = text[: limit - 1]
space = cut.rfind(" ")
return (cut[:space] if space > limit * 0.6 else cut) + "…"


def first_thumbnail(info):
"""First thumbnail across Images then Examples, whatever the template."""
for group in (info.get("Images"), info.get("Examples")):
if not isinstance(group, dict):
continue
for images in group.values():
if isinstance(images, list) and images and isinstance(images[0], dict):
thumbnail = images[0].get("thumbnail")
if thumbnail:
return thumbnail
return None


def kind_of(info):
if info.get("IsTemplate"):
return "template"
if info.get("IsIndividual"):
return "image"
if info.get("IsClass"):
return "class"
return "term"


def build_title(info):
return f"{info.get('Name', '')} [{info.get('Id', '')}] - {SITE_NAME}"


def build_description(info):
meta = info.get("Meta") or {}
name = info.get("Name", "")
parts = []
description = strip_markdown(meta.get("Description"))
if description:
parts.append(description)
# Types is ';'-separated; ", " (not ",") so the sentence reads as a list.
# pageMetadata.js still uses "," here -- worth aligning on its next release.
types = strip_markdown((meta.get("Types") or "").replace(";", ", "))
if types and info.get("IsIndividual"):
parts.append(f"{name} is an instance of {types}.")
if not description:
comment = strip_markdown(meta.get("Comment"))
if comment:
parts.append(comment + ".")
technique = info.get("Technique")
if isinstance(technique, list) and technique:
parts.append("Imaged by " + ", ".join(technique) + ".")
tags = info.get("Tags")
if not parts and isinstance(tags, list) and tags:
parts.append(f"{name} (" + ", ".join(tags).replace("_", " ") + ").")
parts.append(
f"View the 3D image, annotations and queries for this {kind_of(info)} on {SITE_NAME}."
)
return truncate(" ".join(parts))


def preview_metadata(info):
"""The fields a preview page or card needs, as plain strings."""
term_id = info.get("Id", "")
return {
"id": term_id,
"title": build_title(info),
"description": build_description(info),
"image": first_thumbnail(info) or DEFAULT_IMAGE,
"url": REPORTS_BASE + term_id,
"viewer_url": f"{VIEWER_BASE}?id={term_id}",
}


def render_preview_html(info):
"""A self-contained page carrying the term's link-preview tags.

Everything an unfurler reads is in the head; the body is a one-line
fallback for the odd human, with a meta-refresh into the viewer so they
still land where the link pointed. Values are HTML-escaped: names and
descriptions come from ontology text and can contain anything.
"""
m = {key: html.escape(str(value), quote=True) for key, value in preview_metadata(info).items()}
return f"""<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="utf-8">
<title>{m['title']}</title>
<meta name="description" content="{m['description']}">
<link rel="canonical" href="{m['url']}">
<meta property="og:type" content="website">
<meta property="og:site_name" content="{html.escape(SITE_NAME)}">
<meta property="og:title" content="{m['title']}">
<meta property="og:description" content="{m['description']}">
<meta property="og:url" content="{m['url']}">
<meta property="og:image" content="{m['image']}">
<meta name="twitter:card" content="summary_large_image">
<meta name="twitter:title" content="{m['title']}">
<meta name="twitter:description" content="{m['description']}">
<meta name="twitter:image" content="{m['image']}">
<meta http-equiv="refresh" content="0; url={m['viewer_url']}">
</head>
<body>
<p><a href="{m['viewer_url']}">{m['title']}</a></p>
</body>
</html>
"""
Loading