refactor: make the abstract a paragraph instead of a paper field
A template's abstract paragraph (`0 Abstract`) *is* the paper's abstract: it has a position in the document, the heading and typography the template gives it, and the same sentence-by-sentence editing as everything else. `paper.abstract` was a second home for that text — one the document never reads, so a paper could show two different abstracts and the column could drift from the body. The column is gone, from the table, the model, the schemas, the API payloads, the paper form, the list subtitle and the paper page. Nothing else changed. The text already written into it is not gone. Revision a83f5c21d7b6 writes each stored abstract into the paper's body first — one sentence per 。 at the paragraph carrying the abstract heading, appended after anything already there rather than replacing it. A template without such a heading keeps the text too, one position above its first paragraph, where the document renders it under 未设定. Verified against the one paper that had an abstract: 450 characters in, 450 out, identical including `|J| ≤ α · I⁻ᵝ`, split into six sentences in the `0 Abstract` paragraph, still with its own edit button. The smoke test no longer assumes an empty database: it records the paper total before it starts and compares against that, so it can run on a real one.
This commit is contained in:
@@ -0,0 +1,148 @@
|
||||
"""move a paper's abstract into its body, then drop the column
|
||||
|
||||
``paper.abstract`` was a second home for something the outline already has a
|
||||
place for. A template's abstract paragraph (``0 Abstract`` in the seeded
|
||||
library) *is* the paper's abstract — it has a position in the document, a
|
||||
heading the template styles, and it sits in the same order as everything else.
|
||||
A separate column put the same text somewhere the document never reads, so a
|
||||
paper could show two different abstracts, or an abstract that no longer matched
|
||||
the paper it belonged to.
|
||||
|
||||
The column is therefore dropped. The text in it is not: this revision writes
|
||||
each stored abstract into the paper's body first, one sentence per ``。``, at
|
||||
the paragraph that carries the abstract heading. Papers whose template has no
|
||||
such heading keep their text too — it goes to the position just above the first
|
||||
paragraph, where the document renders it under 未设定 rather than discarding it.
|
||||
|
||||
Revision ID: a83f5c21d7b6
|
||||
Revises: f27a1c6d9e04
|
||||
Create Date: 2026-09-18
|
||||
|
||||
"""
|
||||
|
||||
import re
|
||||
from collections.abc import Sequence
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "a83f5c21d7b6"
|
||||
down_revision: str | None = "f27a1c6d9e04"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
#: Split *after* each full stop, so the stop stays with the sentence it ends.
|
||||
#: An abstract with no full stop at all stays one sentence, which is what a
|
||||
#: one-line abstract is.
|
||||
_SENTENCE_BREAK = re.compile(r"(?<=。)")
|
||||
|
||||
#: The heading that means "this paragraph is the abstract". Matched on either
|
||||
#: language because the library is user-editable and both spellings are in use.
|
||||
_ABSTRACT_HEADING = "(LOWER(name) LIKE '%abstract%' OR name LIKE '%摘要%')"
|
||||
|
||||
|
||||
def _split_sentences(text: str) -> list[str]:
|
||||
"""One line per sentence, with whitespace folded as the API folds it."""
|
||||
parts = (" ".join(part.split()) for part in _SENTENCE_BREAK.split(text.strip()))
|
||||
return [part for part in parts if part]
|
||||
|
||||
|
||||
def _target_position(bind: sa.Connection, template_id: int | None) -> int:
|
||||
"""Where the abstract should land in the paper's document.
|
||||
|
||||
The template's abstract paragraph when it has one; otherwise the position
|
||||
just before the first paragraph, so the text keeps its place at the top of
|
||||
the document even though no heading describes it; otherwise 1.
|
||||
"""
|
||||
if template_id is None:
|
||||
return 1
|
||||
|
||||
row = bind.execute(
|
||||
sa.text(
|
||||
f"""
|
||||
SELECT tf.sort
|
||||
FROM template_field AS tf
|
||||
JOIN template_field_library AS lib ON lib.id = tf.field_id
|
||||
WHERE tf.template_id = :template_id AND {_ABSTRACT_HEADING}
|
||||
ORDER BY tf.sort ASC, tf.id ASC
|
||||
LIMIT 1
|
||||
"""
|
||||
),
|
||||
{"template_id": template_id},
|
||||
).first()
|
||||
if row is not None:
|
||||
return int(row[0])
|
||||
|
||||
first = bind.execute(
|
||||
sa.text(
|
||||
"SELECT MIN(sort) FROM template_field WHERE template_id = :template_id"
|
||||
),
|
||||
{"template_id": template_id},
|
||||
).scalar()
|
||||
return int(first) - 1 if first is not None else 1
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
bind = op.get_bind()
|
||||
|
||||
papers = bind.execute(
|
||||
sa.text(
|
||||
"""
|
||||
SELECT id, template_id, abstract
|
||||
FROM paper
|
||||
WHERE abstract IS NOT NULL AND TRIM(abstract) <> ''
|
||||
"""
|
||||
)
|
||||
).all()
|
||||
|
||||
for paper_id, template_id, abstract in papers:
|
||||
sentences = _split_sentences(abstract)
|
||||
if not sentences:
|
||||
continue
|
||||
|
||||
position = _target_position(bind, template_id)
|
||||
# Appended rather than replacing: anything already written in that
|
||||
# paragraph is the writer's, and losing it to a migration would be a
|
||||
# far worse outcome than a duplicate they can delete in one click.
|
||||
start = bind.execute(
|
||||
sa.text(
|
||||
"""
|
||||
SELECT COALESCE(MAX(sort), 0)
|
||||
FROM paper_sentence
|
||||
WHERE paper_id = :paper_id AND paper_template_filed_sort = :position
|
||||
"""
|
||||
),
|
||||
{"paper_id": paper_id, "position": position},
|
||||
).scalar() or 0
|
||||
|
||||
for offset, content in enumerate(sentences, start=1):
|
||||
bind.execute(
|
||||
sa.text(
|
||||
"""
|
||||
INSERT INTO paper_sentence
|
||||
(paper_id, template_id, paper_template_filed_sort, sort, content)
|
||||
VALUES
|
||||
(:paper_id, :template_id, :position, :sort, :content)
|
||||
"""
|
||||
),
|
||||
{
|
||||
"paper_id": paper_id,
|
||||
"template_id": template_id,
|
||||
"position": position,
|
||||
"sort": int(start) + offset,
|
||||
"content": content,
|
||||
},
|
||||
)
|
||||
|
||||
op.drop_column("paper", "abstract")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Re-add the column, empty.
|
||||
|
||||
The text is not moved back: it is body content now, and the column it came
|
||||
from is the thing this revision exists to remove. Reverting the model is
|
||||
enough to leave the column unused again.
|
||||
"""
|
||||
op.add_column("paper", sa.Column("abstract", sa.Text(), nullable=True))
|
||||
@@ -224,7 +224,7 @@ def read(paper: Paper) -> PaperRead:
|
||||
else 0
|
||||
),
|
||||
)
|
||||
return PaperRead(**item.model_dump(), abstract=paper.abstract)
|
||||
return PaperRead(**item.model_dump())
|
||||
|
||||
|
||||
# --- document assembly -------------------------------------------------------
|
||||
@@ -425,7 +425,7 @@ def update(
|
||||
|
||||
``values`` holds only the keys the client actually sent — the route derives
|
||||
it from ``model_fields_set``, which is the only way to tell "clear the
|
||||
abstract" apart from "leave it alone"; both arrive as ``None``.
|
||||
author" apart from "leave it alone"; both arrive as ``None``.
|
||||
|
||||
When the template changes, every sentence is re-stamped with the new
|
||||
template id. The sentences themselves keep their positions, which is what
|
||||
|
||||
@@ -25,7 +25,7 @@ without a migration.
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from sqlalchemy import ForeignKey, Integer, String, Text, text
|
||||
from sqlalchemy import ForeignKey, Integer, String, text
|
||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||
|
||||
from app.db.base import Base
|
||||
@@ -69,9 +69,6 @@ class Paper(TimestampMixin, Base):
|
||||
nullable=True,
|
||||
)
|
||||
|
||||
#: 摘要 — the paper's own abstract, free text.
|
||||
abstract: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
|
||||
#: 作者. Free text, since authorship is written as it will be printed.
|
||||
author: Mapped[str | None] = mapped_column(String(255), nullable=True)
|
||||
|
||||
@@ -91,6 +88,12 @@ class Paper(TimestampMixin, Base):
|
||||
#: 投稿目标期刊.
|
||||
target_journal: Mapped[str | None] = mapped_column(String(255), nullable=True)
|
||||
|
||||
#: There is deliberately no ``abstract`` column. A paper's abstract is a
|
||||
#: paragraph of its body — the template's ``0 Abstract`` field — so it has a
|
||||
#: position in the document, the template's own heading, and the same
|
||||
#: sentence-by-sentence editing as everything else. A column here would be a
|
||||
#: second home for the same text, and the document never reads it.
|
||||
|
||||
#: Eager-loaded with the paper: every read of a paper shows its template
|
||||
#: name, and a lazy load there would be one query per row in the table.
|
||||
template: Mapped["Template | None"] = relationship(lazy="joined")
|
||||
|
||||
@@ -217,7 +217,6 @@ class PaperBase(BaseModel):
|
||||
|
||||
title: str = Field(min_length=1, max_length=255)
|
||||
template_id: int | None = None
|
||||
abstract: str | None = None
|
||||
author: str | None = Field(default=None, max_length=255)
|
||||
status: PaperStatus = "draft"
|
||||
keywords: str | None = Field(default=None, max_length=255)
|
||||
@@ -264,7 +263,6 @@ class PaperUpdate(BaseModel):
|
||||
|
||||
title: str | None = Field(default=None, min_length=1, max_length=255)
|
||||
template_id: int | None = None
|
||||
abstract: str | None = None
|
||||
author: str | None = Field(default=None, max_length=255)
|
||||
status: PaperStatus | None = None
|
||||
keywords: str | None = Field(default=None, max_length=255)
|
||||
@@ -317,9 +315,14 @@ class PaperListItem(BaseModel):
|
||||
|
||||
|
||||
class PaperRead(PaperListItem):
|
||||
"""A paper with everything the table does not need."""
|
||||
"""A paper as its own page needs it.
|
||||
|
||||
abstract: str | None
|
||||
Kept separate from :class:`PaperListItem` even though it currently adds no
|
||||
field, because the two are different contracts: the table's row and the
|
||||
page's subject. The paper's abstract is not here — it is a paragraph of the
|
||||
document (see ``app.crud.paper.build_document``), not a property of the
|
||||
paper.
|
||||
"""
|
||||
|
||||
|
||||
class PaperDocumentRead(BaseModel):
|
||||
|
||||
@@ -65,6 +65,11 @@ def main() -> int:
|
||||
created_papers: list[int] = []
|
||||
|
||||
try:
|
||||
# The database may already hold real papers, so "nothing of ours is
|
||||
# left" is measured against a baseline rather than against zero.
|
||||
baseline = call("GET", "/papers")["total"]
|
||||
print(f"0. baseline: {baseline} paper(s) already in the database")
|
||||
|
||||
print("1. pick two templates with different paragraph positions")
|
||||
templates = call("GET", "/templates?page_size=50")["items"]
|
||||
check("templates exist", len(templates) >= 2, f"got {len(templates)}")
|
||||
@@ -85,7 +90,6 @@ def main() -> int:
|
||||
"status": "writing",
|
||||
"keywords": "关键词A,关键词B;关键词A",
|
||||
"target_journal": "测试期刊",
|
||||
"abstract": "用于验证论文功能的临时数据。",
|
||||
},
|
||||
expect=201,
|
||||
)
|
||||
@@ -310,7 +314,11 @@ def main() -> int:
|
||||
print("14. delete the paper")
|
||||
call("DELETE", f"/papers/{paper['id']}", expect=204)
|
||||
created_papers.clear()
|
||||
check("gone from the list", call("GET", "/papers")["total"] == 0)
|
||||
check(
|
||||
"gone from the list",
|
||||
call("GET", "/papers")["total"] == baseline,
|
||||
f"expected {baseline} paper(s) to remain",
|
||||
)
|
||||
|
||||
print(f"\nall {_checks} checks passed")
|
||||
return 0
|
||||
|
||||
Reference in New Issue
Block a user