perf(citation): optimize CITATION.cff rendering (#39575)

Rendering a CITATION.cff could use memory far out of proportion to the
file, as every YAML alias copies its target into the formatted citation
and the parser copies `%TAG` prefixes into every node. Files past these
limits show no citation, like unparseable ones do today.

- Skip files over 256 KiB, largest real-world file found is 80 KiB
- Skip files with `%TAG` directives
- Skip files whose aliases add more than 64 Ki nodes and value bytes
- Skip self-referencing anchors, except a sequence listing itself

Co-authored-by: wxiaoguang <wxiaoguang@gmail.com>
This commit is contained in:
silverwindandwxiaoguang authored and GitHub committed 2026-10-04 23:50:46 +00:00
1 parent 9bb8751b29
commit 2705abf7f3
4 files changed
+85 -19

No files matched your search

+57 -8
View File
@@ -12,6 +12,7 @@ import (
"slices" "slices"
"strconv" "strconv"
"strings" "strings"
"sync"
"time" "time"
"unicode" "unicode"
"unicode/utf8" "unicode/utf8"
@@ -111,10 +112,16 @@ type reference struct {
Conference actor `yaml:"conference"` Conference actor `yaml:"conference"`
} }
const (
MaxContentSize = 256 * 1024 // parsing takes up to ~1000x the input, largest real-world file found is 80 KiB
maxAliasExpansion = 64 * 1024 // nodes plus value bytes aliases may add
)
// FormatCFF returns the APA and BibTeX citations of a CITATION.cff file, both empty if it has no title or authors // FormatCFF returns the APA and BibTeX citations of a CITATION.cff file, both empty if it has no title or authors
func FormatCFF(content string) (apa, bibtex string) { func FormatCFF(content string) (apa, bibtex string) {
var node yaml.Node var node yaml.Node
if yaml.Unmarshal([]byte(content), &node) != nil { // the parser copies %TAG prefixes into every node
if len(content) > MaxContentSize || strings.Contains(content, "%TAG") || yaml.Unmarshal([]byte(content), &node) != nil || aliasExpansion(&node) > maxAliasExpansion {
return "", "" return "", ""
} }
retagTimestamps(&node) retagTimestamps(&node)
@@ -144,6 +151,32 @@ func retagTimestamps(node *yaml.Node) {
} }
} }
func aliasExpansion(root *yaml.Node) int {
anchors := map[*yaml.Node]int{}
added := 0
var expandedSize func(node, parent *yaml.Node) int
expandedSize = func(node, parent *yaml.Node) int {
if node.Kind == yaml.AliasNode {
size, walked := anchors[node.Alias]
if !walked && (node.Alias != parent || parent.Kind != yaml.SequenceNode) { // decoders never expand a sequence listing itself
size = maxAliasExpansion + 1
}
added = min(added+size, maxAliasExpansion+1)
return size
}
size := 1 + len(node.Value)
for _, child := range node.Content {
size = min(size+expandedSize(child, node), maxAliasExpansion+1)
}
if node.Anchor != "" {
anchors[node] = size
}
return size
}
expandedSize(root, nil)
return added
}
func inspectNode(node *yaml.Node) string { func inspectNode(node *yaml.Node) string {
var parts []string var parts []string
switch node.Kind { switch node.Kind {
@@ -343,15 +376,25 @@ var bibtexTypeFields = map[string][]string{
"unpublished": {"note"}, "unpublished": {"note"},
} }
var ( var globalVars = sync.OnceValue(func() (ret struct {
bibtexEscaper = strings.NewReplacer("&", `\&`, "%", `\%`, "$", `\$`, "#", `\#`, "_", `\_`, "{", `\{`, "}", `\}`) bibtexEscaper *strings.Replacer
keyLetters = strings.NewReplacer( keyLetters *strings.Replacer
keyUnsafeChars *regexp.Regexp
bibtexPattern *regexp.Regexp
},
) {
ret.bibtexEscaper = strings.NewReplacer("&", `\&`, "%", `\%`, "$", `\$`, "#", `\#`, "_", `\_`, "{", `\{`, "}", `\}`)
ret.keyLetters = strings.NewReplacer(
"Æ", "AE", "æ", "ae", "Ð", "D", "ð", "d", "Ø", "O", "ø", "o", "Þ", "Th", "þ", "th", "ß", "ss", "×", "x", "Æ", "AE", "æ", "ae", "Ð", "D", "ð", "d", "Ø", "O", "ø", "o", "Þ", "Th", "þ", "th", "ß", "ss", "×", "x",
"Đ", "D", "đ", "d", "Ħ", "H", "ħ", "h", "ı", "i", "IJ", "IJ", "ij", "ij", "ĸ", "k", "Ŀ", "L", "ŀ", "l", "Đ", "D", "đ", "d", "Ħ", "H", "ħ", "h", "ı", "i", "IJ", "IJ", "ij", "ij", "ĸ", "k", "Ŀ", "L", "ŀ", "l",
"Ł", "L", "ł", "l", "ʼn", "'n", "Ŋ", "NG", "ŋ", "ng", "Œ", "OE", "œ", "oe", "Ŧ", "T", "ŧ", "t", "Ł", "L", "ł", "l", "ʼn", "'n", "Ŋ", "NG", "ŋ", "ng", "Œ", "OE", "œ", "oe", "Ŧ", "T", "ŧ", "t",
) )
keyUnsafeChars = regexp.MustCompile(`[^a-zA-Z0-9-]+`) ret.keyUnsafeChars = regexp.MustCompile(`[^a-zA-Z0-9-]+`)
)
// https://www.acm.org/publications/authors/bibtex-formatting
ret.bibtexPattern = regexp.MustCompile(`(?m)^\s*@?\w+\s*{`) // a simple and quick check, no need to be strict
return ret
})
func keyToASCII() transform.Transformer { func keyToASCII() transform.Transformer {
return transform.Chain( return transform.Chain(
@@ -372,6 +415,7 @@ func (r *reference) formatBibTeX() string {
if len(editors) == 0 { if len(editors) == 0 {
editors = r.EditorsSeries editors = r.EditorsSeries
} }
bibtexEscaper := globalVars().bibtexEscaper
typeFields := map[string]string{ typeFields := map[string]string{
"address": joinNonEmpty(", ", place.City, place.Region, place.Country), "address": joinNonEmpty(", ", place.City, place.Region, place.Country),
"booktitle": bibtexEscaper.Replace(r.CollectionTitle), "booktitle": bibtexEscaper.Replace(r.CollectionTitle),
@@ -441,6 +485,7 @@ func bibtexType(cffType string) string {
} }
func bibtexActors(actors []actor) string { func bibtexActors(actors []actor) string {
bibtexEscaper := globalVars().bibtexEscaper
names := make([]string, 0, len(actors)) names := make([]string, 0, len(actors))
for _, entry := range actors { for _, entry := range actors {
switch { switch {
@@ -463,6 +508,10 @@ func bibtexKey(fields map[string]string) string {
author, _, _ := strings.Cut(fields["author"], ",") author, _, _ := strings.Cut(fields["author"], ",")
titleWords := splitWords(fields["title"]) titleWords := splitWords(fields["title"])
key := joinNonEmpty("_", author, strings.Join(titleWords[:min(3, len(titleWords))], "_"), fields["year"]) key := joinNonEmpty("_", author, strings.Join(titleWords[:min(3, len(titleWords))], "_"), fields["year"])
key, _, _ = transform.String(keyToASCII(), keyLetters.Replace(key)) key, _, _ = transform.String(keyToASCII(), globalVars().keyLetters.Replace(key))
return strings.Trim(keyUnsafeChars.ReplaceAllString(key, "_"), "_") return strings.Trim(globalVars().keyUnsafeChars.ReplaceAllString(key, "_"), "_")
}
func IsLikelyBibTeX(content string) bool {
return globalVars().bibtexPattern.MatchString(content)
} }
+12
View File
@@ -4,6 +4,7 @@
package citation package citation
import ( import (
"strings"
"testing" "testing"
"github.com/stretchr/testify/assert" "github.com/stretchr/testify/assert"
@@ -125,6 +126,10 @@ year = {in press}
}`, }`,
}, },
{cff: "title: No authors\nauthor:\n - name: Typo\n"}, {cff: "title: No authors\nauthor:\n - name: Typo\n"},
{cff: "title: T\nauthors: [{name: A}]\nmessage: " + strings.Repeat("x", MaxContentSize)},
{cff: "title: T\nauthors: [{name: A}]\nmessage: &s " + strings.Repeat("x", maxAliasExpansion/2) + "\nlicense: [*s, *s]\n"},
{cff: "%TAG !e! tag:example.com,2000:\n---\ntitle: T\nauthors: [{name: A}]\n"},
{cff: "preferred-citation: &m {title: T, name: A, authors: [*m]}\n"},
} }
for _, tc := range cases { for _, tc := range cases {
t.Run("", func(t *testing.T) { t.Run("", func(t *testing.T) {
@@ -135,3 +140,10 @@ year = {in press}
}) })
} }
} }
func TestIsLikelyBibTeX(t *testing.T) {
assert.True(t, IsLikelyBibTeX("@article{key, title={Title}}"))
assert.True(t, IsLikelyBibTeX("Inproceedings\n{\n}\n"))
assert.True(t, IsLikelyBibTeX("% comment\n\n@misc{key}\n% comment\n"))
assert.False(t, IsLikelyBibTeX("not bib {}"))
}
+3
View File
@@ -124,6 +124,9 @@ func EntryFollowLinks(ctx context.Context, gitRepo *Repository, commit *Commit,
if treeEntry.IsLink() { if treeEntry.IsLink() {
return res, util.ErrorWrap(util.ErrUnprocessableContent, "%q has too many links", firstFullPath) return res, util.ErrorWrap(util.ErrUnprocessableContent, "%q has too many links", firstFullPath)
} }
if res == nil {
res = &EntryFollowResult{TargetEntry: treeEntry, TargetFullPath: fullPath} // in case limit=0
}
return res, nil return res, nil
} }
+13 -11
View File
@@ -22,7 +22,6 @@ import (
"gitea.dev/modules/git" "gitea.dev/modules/git"
"gitea.dev/modules/htmlutil" "gitea.dev/modules/htmlutil"
"gitea.dev/modules/httplib" "gitea.dev/modules/httplib"
"gitea.dev/modules/lfs"
"gitea.dev/modules/log" "gitea.dev/modules/log"
repo_module "gitea.dev/modules/repository" repo_module "gitea.dev/modules/repository"
"gitea.dev/modules/setting" "gitea.dev/modules/setting"
@@ -102,33 +101,36 @@ func prepareHomeSidebarCitationFile(ctx *context.Context) {
ctx.ServerError("ListEntries", err) ctx.ServerError("ListEntries", err)
return return
} }
isBlob := func(entry *git.TreeEntry) bool { return !entry.IsDir() && !entry.IsSubModule() } isBlobSupported := func(entry *git.TreeEntry) bool {
for _, name := range []string{"CITATION.cff", "CITATION.bib"} { return entry.IsRegular() || entry.IsExecutable() || entry.IsLink()
idx := slices.IndexFunc(allEntries, func(entry *git.TreeEntry) bool { return isBlob(entry) && strings.EqualFold(entry.Name(), name) }) }
const nameCff = "CITATION.cff"
const nameBib = "CITATION.bib"
for _, name := range []string{nameCff, nameBib} {
idx := slices.IndexFunc(allEntries, func(entry *git.TreeEntry) bool {
return isBlobSupported(entry) && util.AsciiEqualFold(entry.Name(), name)
})
if idx == -1 { if idx == -1 {
continue continue
} }
entry := allEntries[idx] entry := allEntries[idx]
if entry.IsLink() { if entry.IsLink() {
res, err := git.EntryFollowLinks(ctx, ctx.Repo.GitRepo, ctx.Repo.Commit, entry.Name(), entry) res, err := git.EntryFollowLinks(ctx, ctx.Repo.GitRepo, ctx.Repo.Commit, entry.Name(), entry)
if err != nil || !isBlob(res.TargetEntry) { if err != nil || !isBlobSupported(res.TargetEntry) {
continue continue
} }
entry = res.TargetEntry entry = res.TargetEntry
} }
content, err := entry.Blob(ctx.Repo.GitRepo).GetBlobContent(ctx, setting.UI.MaxDisplayFileSize) content, err := entry.Blob(ctx.Repo.GitRepo).GetBlobContent(ctx, citation.MaxContentSize+1)
if err != nil { if err != nil {
log.Error("prepareHomeSidebarCitationFile: GetBlobContent: %v", err) log.Error("prepareHomeSidebarCitationFile: GetBlobContent: %v", err)
continue continue
} }
if pointer, _ := lfs.ReadPointerFromBuffer([]byte(content)); pointer.IsValid() {
continue
}
apa, bibtex := "", content apa, bibtex := "", content
if name == "CITATION.cff" { if name == nameCff {
apa, bibtex = citation.FormatCFF(content) apa, bibtex = citation.FormatCFF(content)
} }
if bibtex != "" { if citation.IsLikelyBibTeX(bibtex) {
ctx.Data["CitationFileName"] = allEntries[idx].Name() ctx.Data["CitationFileName"] = allEntries[idx].Name()
ctx.Data["CitationAPA"] = apa ctx.Data["CitationAPA"] = apa
ctx.Data["CitationBibTeX"] = bibtex ctx.Data["CitationBibTeX"] = bibtex