Publish snippets and Markdown posts through authenticated uploads
This commit is contained in:
@@ -0,0 +1,161 @@
|
||||
"""Markdown import with conservative content matching and explicit ID overrides."""
|
||||
from difflib import SequenceMatcher
|
||||
from datetime import date, datetime
|
||||
import hashlib
|
||||
import re
|
||||
import unicodedata
|
||||
|
||||
import yaml
|
||||
|
||||
|
||||
def split_markdown(text):
|
||||
text = text.replace("\r\n", "\n").lstrip("\ufeff")
|
||||
if text.startswith("---\n"):
|
||||
parts = text.split("\n---", 1)
|
||||
if len(parts) != 2 or (parts[1] and not parts[1].startswith("\n")):
|
||||
raise ValueError("Markdown front matter must end with a --- line")
|
||||
try:
|
||||
metadata = yaml.safe_load(parts[0][4:]) or {}
|
||||
except yaml.YAMLError:
|
||||
raise ValueError("Invalid YAML front matter") from None
|
||||
if not isinstance(metadata, dict):
|
||||
raise ValueError("Front matter must contain named fields")
|
||||
return metadata, parts[1].lstrip("\n")
|
||||
return {}, text
|
||||
|
||||
|
||||
def filename_key(filename):
|
||||
return unicodedata.normalize("NFC", filename).casefold()
|
||||
|
||||
|
||||
def parse_upload(filename, raw):
|
||||
if not filename or "/" in filename or "\\" in filename or not filename.lower().endswith((".md", ".markdown")):
|
||||
raise ValueError("Share one .md or .markdown file with a plain filename")
|
||||
if len(filename) > 240 or len(raw) > 1024 * 1024:
|
||||
raise ValueError("Markdown files must be at most 1 MiB with filenames up to 240 characters")
|
||||
try:
|
||||
text = raw.decode("utf-8-sig")
|
||||
except UnicodeDecodeError:
|
||||
raise ValueError("Markdown files must use UTF-8") from None
|
||||
metadata, body = split_markdown(text)
|
||||
if not body.strip():
|
||||
raise ValueError("The Markdown file is empty")
|
||||
clean = {}
|
||||
for name in ("title", "subtitle", "lang", "description", "permalink", "post_id"):
|
||||
if name in metadata:
|
||||
if not isinstance(metadata[name], str) or len(metadata[name]) > 1000:
|
||||
raise ValueError(f"{name} must be text of at most 1000 characters")
|
||||
clean[name] = metadata[name].strip()
|
||||
for name in ("categories", "tags"):
|
||||
if name in metadata:
|
||||
value = metadata[name]
|
||||
if isinstance(value, str):
|
||||
value = value.split()
|
||||
if not isinstance(value, list) or len(value) > 30 or any(not isinstance(item, str) or len(item) > 100 for item in value):
|
||||
raise ValueError(f"{name} must be a short list of names")
|
||||
clean[name] = value
|
||||
for name in ("toc", "math"):
|
||||
if name in metadata:
|
||||
if not isinstance(metadata[name], bool):
|
||||
raise ValueError(f"{name} must be true or false")
|
||||
clean[name] = metadata[name]
|
||||
if "date" in metadata:
|
||||
value = metadata["date"]
|
||||
if isinstance(value, (date, datetime)):
|
||||
value = value.isoformat()
|
||||
try:
|
||||
datetime.fromisoformat(str(value).replace("Z", "+00:00"))
|
||||
except ValueError:
|
||||
raise ValueError("date must be an ISO date or timestamp") from None
|
||||
clean["date"] = str(value)
|
||||
explicit_id = clean.pop("post_id", None)
|
||||
if explicit_id is not None and not re.fullmatch(r"[a-zA-Z0-9_-]{1,80}", explicit_id):
|
||||
raise ValueError("post_id must be 1–80 letters, digits, underscores, or hyphens")
|
||||
if "permalink" in clean and not re.fullmatch(r"/[\w/-]+/", clean["permalink"]):
|
||||
raise ValueError("permalink must be a local path such as /my-post/")
|
||||
title = clean.get("title")
|
||||
if not title:
|
||||
heading = re.search(r"^#\s+(.+)$", body, re.MULTILINE)
|
||||
clean["title"] = heading.group(1).strip() if heading else re.sub(r"\.(md|markdown)$", "", filename, flags=re.I)
|
||||
return dict(filename=filename, key=filename_key(filename), explicit_id=explicit_id,
|
||||
metadata=clean, body=body)
|
||||
|
||||
|
||||
def publish_post(repo, submission, manifest):
|
||||
import json
|
||||
incoming = json.loads((submission / "post.json").read_text())
|
||||
candidates = []
|
||||
scored = []
|
||||
records = []
|
||||
for path in sorted((repo / "_posts").glob("*")):
|
||||
if path.suffix.lower() not in {".md", ".markdown"}:
|
||||
continue
|
||||
metadata, body = split_markdown(path.read_text())
|
||||
records.append((path, metadata))
|
||||
if incoming["explicit_id"]:
|
||||
match = metadata.get("upload_id") == incoming["explicit_id"] or metadata.get("post_id") == incoming["explicit_id"]
|
||||
else:
|
||||
match = False
|
||||
score = content_similarity(incoming["body"], body)
|
||||
if score >= 0.84:
|
||||
scored.append((score, path, metadata))
|
||||
if match:
|
||||
candidates.append((path, metadata))
|
||||
if not incoming["explicit_id"] and scored:
|
||||
scored.sort(key=lambda item: item[0], reverse=True)
|
||||
if scored[0][0] >= 0.92:
|
||||
if len(scored) > 1 and scored[0][0] - scored[1][0] < 0.08:
|
||||
raise ValueError("Multiple posts have similar content; add the intended post_id to disambiguate")
|
||||
candidates.append(scored[0][1:])
|
||||
if len(candidates) > 1:
|
||||
raise ValueError("More than one post matches this post ID")
|
||||
identity = incoming["explicit_id"] or manifest["id"]
|
||||
if candidates:
|
||||
entry, metadata = candidates[0]
|
||||
identity = metadata.get("upload_id", identity)
|
||||
# Keep the public URL and original publication date during updates.
|
||||
metadata.update({key: value for key, value in incoming["metadata"].items() if key not in {"date", "permalink"}})
|
||||
permalink = metadata.get("permalink") or "/" + re.sub(r"^\d{4}-\d{2}-\d{2}-", "", entry.stem) + "/"
|
||||
else:
|
||||
metadata = dict(incoming["metadata"])
|
||||
metadata.setdefault("date", manifest["date"])
|
||||
slug = re.sub(r"[^\w-]+", "-", incoming["metadata"]["title"].lower()).strip("-")[:80] or "post"
|
||||
permalink = metadata.get("permalink", f"/{slug}-{identity[:8]}/")
|
||||
entry = repo / "_posts" / f"{str(metadata['date'])[:10]}-upload-{identity}.md"
|
||||
if entry.exists():
|
||||
raise ValueError("Post identity conflicts with another file")
|
||||
# Avoid taking over another page's URL, including non-post pages.
|
||||
for other in list(repo.glob("*.md")) + list(repo.glob("*.html")):
|
||||
other_metadata, _ = split_markdown(other.read_text())
|
||||
records.append((other, other_metadata))
|
||||
for other, other_metadata in records:
|
||||
url = other_metadata.get("permalink")
|
||||
if not url and other.parent.name == "_posts":
|
||||
url = re.sub(r"^\d{4}-\d{2}-\d{2}-", "", other.stem)
|
||||
if url and str(url).strip("/") == permalink.strip("/"):
|
||||
raise ValueError("Another page already uses this permalink")
|
||||
permalink = "/" + str(permalink).strip("/") + "/"
|
||||
metadata.update(layout="post", permalink=permalink, upload_id=identity,
|
||||
upload_filename=incoming["key"], upload_revision=manifest["id"],
|
||||
render_with_liquid=False)
|
||||
entry.parent.mkdir(parents=True, exist_ok=True)
|
||||
entry.write_text("---\n" + yaml.safe_dump(metadata, allow_unicode=True, sort_keys=False) +
|
||||
"---\n\n" + incoming["body"])
|
||||
return entry.relative_to(repo), permalink, "updated" if candidates else "created", identity
|
||||
|
||||
|
||||
def content_similarity(left, right):
|
||||
"""Ignore whitespace; require substantial text before accepting a fuzzy match."""
|
||||
left = " ".join(unicodedata.normalize("NFC", left).split())
|
||||
right = " ".join(unicodedata.normalize("NFC", right).split())
|
||||
if left == right:
|
||||
return 1.0 if len(left) >= 40 else 0.0
|
||||
if min(len(left), len(right)) < 200:
|
||||
return 0.0
|
||||
# Size bound avoids expensive comparisons to obviously different posts.
|
||||
if min(len(left), len(right)) / max(len(left), len(right)) < 0.84:
|
||||
return 0.0
|
||||
matcher = SequenceMatcher(None, left.split(), right.split())
|
||||
if matcher.quick_ratio() < 0.84:
|
||||
return 0.0
|
||||
return matcher.ratio()
|
||||
Reference in New Issue
Block a user