Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
264 changes: 264 additions & 0 deletions src/story2script/app/api/v1/adapt.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,20 +12,30 @@
AdaptRequest,
AdaptResponse,
Expansion,
ExportRequest,
ExportResponse,
Fidelity,
InlineEditRequest,
InlineEditResponse,
QualityMetrics,
RefinedBeat,
RefineSceneRequest,
RefineSceneResponse,
RefineSummaryRequest,
RefineSummaryResponse,
SchemaError,
SchemaValidationReport,
ValidateRequest,
ValidateResponse,
)
from story2script.export import to_fountain, to_markdown, to_txt
from story2script.llm.factory import get_provider_or_none
from story2script.llm.refine_inline import refine_inline_text
from story2script.llm.refine_summary import refine_scene_summary
from story2script.llm.stages.mode_refine import ai_refine_scene_beats
from story2script.parsing.character_extractor import Character
from story2script.parsing.dialogue_extractor import Beat
from story2script.parsing.scene_segmenter import Scene
from story2script.io.dispatch import load_from_bytes
from story2script.io.document import NormalizedDocument, UnsupportedFormatError
from story2script.pipeline.assemble import (
Expand Down Expand Up @@ -328,4 +338,258 @@ async def validate(request: ValidateRequest) -> ValidateResponse:
return ValidateResponse(valid=report.valid, errors=report.errors)


# MIME types per format. ``text/x-fountain`` is the Fountain spec's
# registered MIME; the other two use the standard plain-text MIMEs so
# the browser doesn't try to render them as HTML.
_EXPORT_MIME = {
"fountain": "text/x-fountain; charset=utf-8",
"txt": "text/plain; charset=utf-8",
"md": "text/markdown; charset=utf-8",
}
_EXPORT_EXTENSION = {"fountain": "fountain", "txt": "txt", "md": "md"}


def _slugify_for_filename(title: str) -> str:
"""Make a title safe for use in a Content-Disposition filename.
Strip path separators + control chars; collapse spaces to
underscores. CJK round-trips intact."""
cleaned = "".join(ch for ch in title if ch not in '\\/:*?"<>|\n\r\t')
cleaned = "_".join(cleaned.split()).strip("_")
return cleaned or "screenplay"


@router.post("/export", response_model=ExportResponse)
async def export_screenplay(request: ExportRequest) -> ExportResponse:
"""Convert the (possibly edited) screenplay YAML into an
author-facing format.

The endpoint is stateless: the request carries the YAML body and
the target format; the response carries the rendered text and a
Content-Disposition hint the frontend uses for the download. PDF
is intentionally not handled here — the frontend renders the
markdown variant in a print-friendly window and lets the browser's
native Print → Save as PDF do the conversion. That keeps server
dependencies CJK-font-free.
"""
try:
payload = yaml.safe_load(request.screenplay_yaml)
except yaml.YAMLError as exc:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={
"code": "INVALID_YAML",
"message": f"YAML 解析失败: {exc}",
},
) from exc

if not isinstance(payload, dict):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={
"code": "INVALID_YAML",
"message": "顶层结构必须是 mapping。",
},
)

# Validate against the schema before rendering; a malformed input
# would either crash the renderer or silently drop fields. Returning
# 422 with the schema errors gives the frontend something
# actionable instead of a half-rendered file.
report = _validate_against_schema(payload)
if not report.valid:
raise HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
detail={
"code": "INVALID_SCREENPLAY",
"message": "草稿未通过 schema 校验,请在编辑器中修正后再导出。",
"errors": [e.model_dump() for e in report.errors[:5]],
},
)

renderer = {
"fountain": to_fountain,
"txt": to_txt,
"md": to_markdown,
}[request.format]
content = renderer(payload)

title = (payload.get("metadata") or {}).get("title") or "screenplay"
filename = f"{_slugify_for_filename(title)}.{_EXPORT_EXTENSION[request.format]}"

return ExportResponse(
content=content,
format=request.format,
mime_type=_EXPORT_MIME[request.format],
suggested_filename=filename,
)


@router.post("/refine/scene", response_model=RefineSceneResponse)
async def refine_scene_endpoint(
request: RefineSceneRequest,
) -> RefineSceneResponse:
"""Run the mode-refine LLM stage on a single scene's beats.

Lets the author hit "AI 整体润色" on one scene from the editor and
get a polished pass with a possibly different mode than the
document's metadata.mode (so they can experiment without re-running
the whole pipeline). 503 when no LLM provider is configured, 422
on invalid YAML / unknown scene id, 502 when the LLM returns
something unusable.
"""
provider = get_provider_or_none()
if provider is None:
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail={
"code": "NO_LLM_PROVIDER",
"message": (
"未配置 LLM provider。设置 MIMO_API_KEY 或 "
"DEEPSEEK_API_KEY 后重启服务。"
),
},
)

try:
payload = yaml.safe_load(request.screenplay_yaml)
except yaml.YAMLError as exc:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={"code": "INVALID_YAML", "message": f"YAML 解析失败: {exc}"},
) from exc
if not isinstance(payload, dict):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={"code": "INVALID_YAML", "message": "顶层结构必须是 mapping。"},
)

scenes = payload.get("scenes") or []
scene_dict = next((s for s in scenes if s.get("id") == request.scene_id), None)
if scene_dict is None:
raise HTTPException(
status_code=status.HTTP_404_NOT_FOUND,
detail={
"code": "SCENE_NOT_FOUND",
"message": f"未找到场景 {request.scene_id}。",
},
)

beat_dicts = scene_dict.get("beats") or []
if not beat_dicts:
raise HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
detail={
"code": "EMPTY_SCENE",
"message": "场景没有节拍可以润色。",
},
)

# Build the minimal Scene / Beat / Character pydantic instances the
# refine stage expects. We don't have original Paragraphs (the YAML
# doesn't carry them), so we stub one from the summary — the refine
# prompt mostly uses beats + roster + summary, not the raw text.
from story2script.io.document import Paragraph

summary = (scene_dict.get("summary") or "(无)").strip() or "(无)"
stub_paragraph = Paragraph(
index=0,
text=summary,
char_start=0,
char_end=len(summary),
)

try:
scene_model = Scene(
chapter_index=scene_dict.get("chapter_index", 0),
index_in_chapter=0,
location=scene_dict.get("location"),
time=scene_dict.get("time"),
summary=summary,
paragraphs=[stub_paragraph],
char_start=0,
char_end=len(summary),
detection_method=scene_dict.get("detection_method", "fallback_single_scene"),
confidence=scene_dict.get("confidence", 0.5),
)
except Exception as exc:
raise HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
detail={
"code": "INVALID_SCENE",
"message": f"场景字段不合法: {exc}",
},
) from exc

beats: list[Beat] = []
for b in beat_dicts:
try:
beats.append(Beat(
kind=b.get("kind", "action"),
speaker=b.get("speaker"),
text=(b.get("text") or "").strip() or "(空)",
source_type=b.get("source_type", "generated"),
confidence=float(b.get("confidence", 0.5)),
needs_review=bool(b.get("needs_review", False)),
))
except Exception:
# Skip beats the schema check would catch elsewhere; refine
# is a best-effort polish, not a validation pass.
continue

characters: list[Character] = []
for c in payload.get("characters") or []:
try:
characters.append(Character(
id=c.get("id", ""),
canonical_name=c.get("name", ""),
aliases=c.get("aliases") or [],
first_appearance=c.get("first_appearance") or "scene_1",
source_type=c.get("source_type", "generated"),
confidence=float(c.get("confidence", 0.5)),
mentions=max(1, int(c.get("mentions", 1))),
))
except Exception:
continue

refined = ai_refine_scene_beats(
scene=scene_model,
beats=beats,
characters=characters,
mode_fidelity=request.fidelity,
mode_expansion=request.expansion,
provider=provider,
)
if refined is None:
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail={
"code": "REFINE_FAILED",
"message": (
"LLM 润色失败,请稍后重试。详细原因见 logs/story2script.log "
"中的 stage=refine 行。"
),
},
)

# Map char_id back to canonical name so the frontend can splice
# speaker into the YAML cleanly. Speaker may be None (the refine
# stage flags low-confidence dialogue lines with needs_review).
refined_payload: list[RefinedBeat] = []
for b in refined:
refined_payload.append(RefinedBeat(
kind=b.kind,
text=b.text,
speaker=b.speaker,
source_type=b.source_type,
confidence=b.confidence,
needs_review=b.needs_review,
))

return RefineSceneResponse(
scene_id=request.scene_id,
beats=refined_payload,
model=getattr(provider, "model", "unknown"),
)


__all__ = ["router", "_yaml_dump"]
70 changes: 70 additions & 0 deletions src/story2script/app/api/v1/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,76 @@ class ValidateResponse(BaseModel):
errors: list[SchemaError] = Field(default_factory=list)


# Author-facing export formats. ``yaml`` is intentionally NOT in this
# list — the validated schema YAML is downloadable directly from the
# editor; ``/api/v1/export`` exists to convert that into formats a
# director or actor can actually read.
ExportFormat = Literal["fountain", "txt", "md"]


class ExportRequest(BaseModel):
"""JSON body for POST /api/v1/export.

Sends the (possibly edited) screenplay YAML in, asks for the
target format back. The endpoint is stateless: no IDs, no
server-side cache; the author's draft is the single source of
truth and lives in the browser's sessionStorage.
"""

screenplay_yaml: str = Field(min_length=1)
format: ExportFormat


class ExportResponse(BaseModel):
"""Plain-text body of the rendered screenplay plus a MIME hint so
the front-end can wire a download link without re-deriving it."""

content: str
format: ExportFormat
mime_type: str
suggested_filename: str


class RefineSceneRequest(BaseModel):
"""JSON body for POST /api/v1/refine/scene.

Sends the full screenplay YAML plus the scene id we want refined,
along with the mode the LLM should apply (overriding the document's
metadata.mode so the author can try different fidelity / expansion
settings on the same scene without re-running the whole pipeline).
"""

screenplay_yaml: str = Field(min_length=1)
scene_id: str = Field(min_length=1)
fidelity: Fidelity
expansion: Expansion


class RefinedBeat(BaseModel):
"""One refined beat the frontend will splice back into the scene.

Mirrors the schema's beat shape but lives in API land so the
pydantic Literal types stay attached to the API surface.
"""

kind: Literal["dialogue", "action", "stage_direction", "narration", "transition"]
text: str = Field(min_length=1)
speaker: str | None = None
source_type: Literal["original", "inferred", "generated", "merged"] = "generated"
confidence: float
needs_review: bool = False


class RefineSceneResponse(BaseModel):
"""The full refined beat list for the scene plus the model used.
The frontend swaps these in atomically; partial application would
leave the scene in an inconsistent state mid-update."""

scene_id: str
beats: list[RefinedBeat]
model: str


class RefineSummaryRequest(BaseModel):
"""Body for POST /api/v1/refine/scene-summary.

Expand Down
23 changes: 23 additions & 0 deletions src/story2script/export/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
"""Screenplay export to author-facing formats.

The screenplay YAML is the contract format — schema-validated,
operator-friendly, but not what a director or actor wants to read.
This package turns a validated screenplay dict into the standard
industry formats authors actually share:

- **fountain**: plain-text industry standard, Final Draft / WriterDuet /
Highland can all import it.
- **txt**: human-readable centered-title screenplay layout.
- **md**: Markdown with the same structure, GitHub / Notion preview.

PDF is intentionally NOT generated server-side. CJK fonts pull in
either GTK (weasyprint) or a 5+ MB bundled font (fpdf2). The
frontend opens the markdown-rendered view in a new tab and lets the
browser's native Print → Save as PDF handle it, which works on
every system without extra deps.
"""
from story2script.export.fountain import to_fountain
from story2script.export.markdown import to_markdown
from story2script.export.txt import to_txt

__all__ = ["to_fountain", "to_markdown", "to_txt"]
Loading
Loading