perf(citation): optimize CITATION.cff rendering (#39575)

Rendering a CITATION.cff could use memory far out of proportion to the
file, as every YAML alias copies its target into the formatted citation
and the parser copies `%TAG` prefixes into every node. Files past these
limits show no citation, like unparseable ones do today.

- Skip files over 256 KiB, largest real-world file found is 80 KiB
- Skip files with `%TAG` directives
- Skip files whose aliases add more than 64 Ki nodes and value bytes
- Skip self-referencing anchors, except a sequence listing itself

Co-authored-by: wxiaoguang <wxiaoguang@gmail.com>
This commit is contained in:
silverwindandwxiaoguang authored and GitHub committed 2026-10-04 23:50:46 +00:00
1 parent 9bb8751b29
commit 2705abf7f3
4 files changed
+85 -19

No files matched your search

+57 -8
View File
@@ -12,6 +12,7 @@ import (
"slices"
"strconv"
"strings"
"sync"
"time"
"unicode"
"unicode/utf8"
@@ -111,10 +112,16 @@ type reference struct {
Conference actor `yaml:"conference"`
}
const (
MaxContentSize = 256 * 1024 // parsing takes up to ~1000x the input, largest real-world file found is 80 KiB
maxAliasExpansion = 64 * 1024 // nodes plus value bytes aliases may add
)
// FormatCFF returns the APA and BibTeX citations of a CITATION.cff file, both empty if it has no title or authors
func FormatCFF(content string) (apa, bibtex string) {
var node yaml.Node
if yaml.Unmarshal([]byte(content), &node) != nil {
// the parser copies %TAG prefixes into every node
if len(content) > MaxContentSize || strings.Contains(content, "%TAG") || yaml.Unmarshal([]byte(content), &node) != nil || aliasExpansion(&node) > maxAliasExpansion {
return "", ""
}
retagTimestamps(&node)
@@ -144,6 +151,32 @@ func retagTimestamps(node *yaml.Node) {
}
}
func aliasExpansion(root *yaml.Node) int {
anchors := map[*yaml.Node]int{}
added := 0
var expandedSize func(node, parent *yaml.Node) int
expandedSize = func(node, parent *yaml.Node) int {
if node.Kind == yaml.AliasNode {
size, walked := anchors[node.Alias]
if !walked && (node.Alias != parent || parent.Kind != yaml.SequenceNode) { // decoders never expand a sequence listing itself
size = maxAliasExpansion + 1
}
added = min(added+size, maxAliasExpansion+1)
return size
}
size := 1 + len(node.Value)
for _, child := range node.Content {
size = min(size+expandedSize(child, node), maxAliasExpansion+1)
}
if node.Anchor != "" {
anchors[node] = size
}
return size
}
expandedSize(root, nil)
return added
}
func inspectNode(node *yaml.Node) string {
var parts []string
switch node.Kind {
@@ -343,15 +376,25 @@ var bibtexTypeFields = map[string][]string{
"unpublished": {"note"},
}
var (
bibtexEscaper = strings.NewReplacer("&", `\&`, "%", `\%`, "$", `\$`, "#", `\#`, "_", `\_`, "{", `\{`, "}", `\}`)
keyLetters = strings.NewReplacer(
var globalVars = sync.OnceValue(func() (ret struct {
bibtexEscaper *strings.Replacer
keyLetters *strings.Replacer
keyUnsafeChars *regexp.Regexp
bibtexPattern *regexp.Regexp
},
) {
ret.bibtexEscaper = strings.NewReplacer("&", `\&`, "%", `\%`, "$", `\$`, "#", `\#`, "_", `\_`, "{", `\{`, "}", `\}`)
ret.keyLetters = strings.NewReplacer(
"Æ", "AE", "æ", "ae", "Ð", "D", "ð", "d", "Ø", "O", "ø", "o", "Þ", "Th", "þ", "th", "ß", "ss", "×", "x",
"Đ", "D", "đ", "d", "Ħ", "H", "ħ", "h", "ı", "i", "IJ", "IJ", "ij", "ij", "ĸ", "k", "Ŀ", "L", "ŀ", "l",
"Ł", "L", "ł", "l", "ʼn", "'n", "Ŋ", "NG", "ŋ", "ng", "Œ", "OE", "œ", "oe", "Ŧ", "T", "ŧ", "t",
)
keyUnsafeChars = regexp.MustCompile(`[^a-zA-Z0-9-]+`)
)
ret.keyUnsafeChars = regexp.MustCompile(`[^a-zA-Z0-9-]+`)
// https://www.acm.org/publications/authors/bibtex-formatting
ret.bibtexPattern = regexp.MustCompile(`(?m)^\s*@?\w+\s*{`) // a simple and quick check, no need to be strict
return ret
})
func keyToASCII() transform.Transformer {
return transform.Chain(
@@ -372,6 +415,7 @@ func (r *reference) formatBibTeX() string {
if len(editors) == 0 {
editors = r.EditorsSeries
}
bibtexEscaper := globalVars().bibtexEscaper
typeFields := map[string]string{
"address": joinNonEmpty(", ", place.City, place.Region, place.Country),
"booktitle": bibtexEscaper.Replace(r.CollectionTitle),
@@ -441,6 +485,7 @@ func bibtexType(cffType string) string {
}
func bibtexActors(actors []actor) string {
bibtexEscaper := globalVars().bibtexEscaper
names := make([]string, 0, len(actors))
for _, entry := range actors {
switch {
@@ -463,6 +508,10 @@ func bibtexKey(fields map[string]string) string {
author, _, _ := strings.Cut(fields["author"], ",")
titleWords := splitWords(fields["title"])
key := joinNonEmpty("_", author, strings.Join(titleWords[:min(3, len(titleWords))], "_"), fields["year"])
key, _, _ = transform.String(keyToASCII(), keyLetters.Replace(key))
return strings.Trim(keyUnsafeChars.ReplaceAllString(key, "_"), "_")
key, _, _ = transform.String(keyToASCII(), globalVars().keyLetters.Replace(key))
return strings.Trim(globalVars().keyUnsafeChars.ReplaceAllString(key, "_"), "_")
}
func IsLikelyBibTeX(content string) bool {
return globalVars().bibtexPattern.MatchString(content)
}
+12
View File
@@ -4,6 +4,7 @@
package citation
import (
"strings"
"testing"
"github.com/stretchr/testify/assert"
@@ -125,6 +126,10 @@ year = {in press}
}`,
},
{cff: "title: No authors\nauthor:\n - name: Typo\n"},
{cff: "title: T\nauthors: [{name: A}]\nmessage: " + strings.Repeat("x", MaxContentSize)},
{cff: "title: T\nauthors: [{name: A}]\nmessage: &s " + strings.Repeat("x", maxAliasExpansion/2) + "\nlicense: [*s, *s]\n"},
{cff: "%TAG !e! tag:example.com,2000:\n---\ntitle: T\nauthors: [{name: A}]\n"},
{cff: "preferred-citation: &m {title: T, name: A, authors: [*m]}\n"},
}
for _, tc := range cases {
t.Run("", func(t *testing.T) {
@@ -135,3 +140,10 @@ year = {in press}
})
}
}
func TestIsLikelyBibTeX(t *testing.T) {
assert.True(t, IsLikelyBibTeX("@article{key, title={Title}}"))
assert.True(t, IsLikelyBibTeX("Inproceedings\n{\n}\n"))
assert.True(t, IsLikelyBibTeX("% comment\n\n@misc{key}\n% comment\n"))
assert.False(t, IsLikelyBibTeX("not bib {}"))
}
+3
View File
@@ -124,6 +124,9 @@ func EntryFollowLinks(ctx context.Context, gitRepo *Repository, commit *Commit,
if treeEntry.IsLink() {
return res, util.ErrorWrap(util.ErrUnprocessableContent, "%q has too many links", firstFullPath)
}
if res == nil {
res = &EntryFollowResult{TargetEntry: treeEntry, TargetFullPath: fullPath} // in case limit=0
}
return res, nil
}
+13 -11
View File
@@ -22,7 +22,6 @@ import (
"gitea.dev/modules/git"
"gitea.dev/modules/htmlutil"
"gitea.dev/modules/httplib"
"gitea.dev/modules/lfs"
"gitea.dev/modules/log"
repo_module "gitea.dev/modules/repository"
"gitea.dev/modules/setting"
@@ -102,33 +101,36 @@ func prepareHomeSidebarCitationFile(ctx *context.Context) {
ctx.ServerError("ListEntries", err)
return
}
isBlob := func(entry *git.TreeEntry) bool { return !entry.IsDir() && !entry.IsSubModule() }
for _, name := range []string{"CITATION.cff", "CITATION.bib"} {
idx := slices.IndexFunc(allEntries, func(entry *git.TreeEntry) bool { return isBlob(entry) && strings.EqualFold(entry.Name(), name) })
isBlobSupported := func(entry *git.TreeEntry) bool {
return entry.IsRegular() || entry.IsExecutable() || entry.IsLink()
}
const nameCff = "CITATION.cff"
const nameBib = "CITATION.bib"
for _, name := range []string{nameCff, nameBib} {
idx := slices.IndexFunc(allEntries, func(entry *git.TreeEntry) bool {
return isBlobSupported(entry) && util.AsciiEqualFold(entry.Name(), name)
})
if idx == -1 {
continue
}
entry := allEntries[idx]
if entry.IsLink() {
res, err := git.EntryFollowLinks(ctx, ctx.Repo.GitRepo, ctx.Repo.Commit, entry.Name(), entry)
if err != nil || !isBlob(res.TargetEntry) {
if err != nil || !isBlobSupported(res.TargetEntry) {
continue
}
entry = res.TargetEntry
}
content, err := entry.Blob(ctx.Repo.GitRepo).GetBlobContent(ctx, setting.UI.MaxDisplayFileSize)
content, err := entry.Blob(ctx.Repo.GitRepo).GetBlobContent(ctx, citation.MaxContentSize+1)
if err != nil {
log.Error("prepareHomeSidebarCitationFile: GetBlobContent: %v", err)
continue
}
if pointer, _ := lfs.ReadPointerFromBuffer([]byte(content)); pointer.IsValid() {
continue
}
apa, bibtex := "", content
if name == "CITATION.cff" {
if name == nameCff {
apa, bibtex = citation.FormatCFF(content)
}
if bibtex != "" {
if citation.IsLikelyBibTeX(bibtex) {
ctx.Data["CitationFileName"] = allEntries[idx].Name()
ctx.Data["CitationAPA"] = apa
ctx.Data["CitationBibTeX"] = bibtex