summaryrefslogtreecommitdiffci
path: root/internal/blog/s3.go
diff refs
from: back
to: back
| flip
diff options
context:
space:
mode:
Diffstat (limited to 'internal/blog/s3.go')
-rw-r--r--internal/blog/s3.go269
1 files changed, 157 insertions, 112 deletions
diff --git a/internal/blog/s3.go b/internal/blog/s3.go
index 3c1a932..bbd5238 100644
--- a/internal/blog/s3.go
+++ b/internal/blog/s3.go
@@ -1,22 +1,29 @@
package blog
import (
- "bytes"
"context"
"encoding/json"
+ "errors"
"fmt"
"io"
+ "log/slog"
+ "net/url"
+ "slices"
"strings"
+ "sync"
+ "time"
"github.com/SayaAndy/saya-today-web/config"
"github.com/SayaAndy/saya-today-web/internal/frontmatter"
- "github.com/SayaAndy/saya-today-web/l10n"
"github.com/aws/aws-sdk-go-v2/aws"
awsconfig "github.com/aws/aws-sdk-go-v2/config"
"github.com/aws/aws-sdk-go-v2/credentials"
"github.com/aws/aws-sdk-go-v2/service/s3"
+ s3types "github.com/aws/aws-sdk-go-v2/service/s3/types"
)
+const s3ScanConcurrency = 32
+
type S3Client struct {
prefix string
bucketName string
@@ -59,38 +66,39 @@ func NewS3Client(cfg *config.StorageConfig) (Client, error) {
return &S3Client{s3cfg.Prefix, s3cfg.BucketName, s3cl}, nil
}
-func (c *S3Client) GetMedleys() ([]MedleyEntry, error) {
- idxRaw, err := c.readAll(MedleysIndexFileName)
- if err != nil {
- return nil, fmt.Errorf("read %s: %w", MedleysIndexFileName, err)
+func (c *S3Client) Scan(prefix string) ([]*Page, error) {
+ pages, err := c.scanFromIndex(prefix)
+ if err == nil {
+ return pages, nil
}
- var idx []MedleyEntry
- if err := json.Unmarshal(idxRaw, &idx); err != nil {
- return nil, fmt.Errorf("unmarshal %s: %w", MedleysIndexFileName, err)
+ var nsk *s3types.NoSuchKey
+ if !errors.As(err, &nsk) {
+ return nil, err
}
- return idx, nil
+ slog.Warn("index.json missing, falling back to listing", slog.String("prefix", c.prefix))
+ return c.scanByListing(prefix)
}
-func (c *S3Client) Scan(prefix string) ([]*Page, error) {
+func (c *S3Client) scanFromIndex(prefix string) ([]*Page, error) {
out, err := c.s3cl.GetObject(context.Background(), &s3.GetObjectInput{
Bucket: aws.String(c.bucketName),
Key: aws.String(IndexFileName),
})
if err != nil {
- return nil, fmt.Errorf("get %s: %w", IndexFileName, err)
+ return nil, fmt.Errorf("get index.json: %w", err)
}
defer out.Body.Close()
raw, err := io.ReadAll(out.Body)
if err != nil {
- return nil, fmt.Errorf("read %s: %w", IndexFileName, err)
+ return nil, fmt.Errorf("read index.json: %w", err)
}
var idx Index
if err := json.Unmarshal(raw, &idx); err != nil {
- return nil, fmt.Errorf("unmarshal %s: %w", IndexFileName, err)
+ return nil, fmt.Errorf("unmarshal index.json: %w", err)
}
wantLang := ""
@@ -100,93 +108,160 @@ func (c *S3Client) Scan(prefix string) ([]*Page, error) {
fullPrefix := c.prefix + prefix
pages := make([]*Page, 0)
-
- switch idx.SchemaVersion {
- case 1:
- for catKey, cat := range *idx.Categories.(*map[string]*IndexV1Category) {
- lang, ok := strings.CutPrefix(catKey, c.prefix)
- if !ok {
+ for catKey, cat := range idx.Categories {
+ lang, ok := strings.CutPrefix(catKey, c.prefix)
+ if !ok {
+ continue
+ }
+ if wantLang != "" && wantLang != lang {
+ continue
+ }
+ for _, e := range cat.Pages {
+ if !strings.HasPrefix(e.Link, fullPrefix) {
continue
}
- if wantLang != "" && wantLang != lang {
+ linkParts := strings.Split(e.Link, "/")
+ nameParts := strings.Split(linkParts[len(linkParts)-1], ".")
+ fileName := strings.Join(nameParts[:len(nameParts)-1], ".")
+ pages = append(pages, &Page{
+ Link: e.Link,
+ FileName: fileName,
+ Lang: lang,
+ ModifiedTime: e.ModifiedTime,
+ Metadata: &frontmatter.Metadata{
+ Title: e.Title,
+ ShortDescription: e.ShortDescription,
+ ActionDate: e.ActionDate,
+ PublishedTime: e.PublishedTime,
+ Thumbnail: e.Thumbnail,
+ Tags: e.Tags,
+ Geolocation: e.Geolocation,
+ Medley: e.Medley,
+ MedleyPart: e.MedleyPart,
+ },
+ })
+ }
+ }
+ return pages, nil
+}
+
+func (c *S3Client) scanByListing(prefix string) ([]*Page, error) {
+ fullPrefix := c.prefix + prefix
+
+ type candidate struct {
+ key string
+ lastModified time.Time
+ }
+ var candidates []candidate
+
+ paginator := s3.NewListObjectsV2Paginator(c.s3cl, &s3.ListObjectsV2Input{
+ Bucket: aws.String(c.bucketName),
+ Prefix: aws.String(fullPrefix),
+ })
+ for paginator.HasMorePages() {
+ output, err := paginator.NextPage(context.Background())
+ if err != nil {
+ return nil, fmt.Errorf("list S3 objects: %w", err)
+ }
+ for _, obj := range output.Contents {
+ key := aws.ToString(obj.Key)
+ if !strings.HasSuffix(key, ".md") {
continue
}
- for _, e := range cat.Pages {
- if !strings.HasPrefix(e.Link, fullPrefix) {
- continue
+ candidates = append(candidates, candidate{key, aws.ToTime(obj.LastModified)})
+ }
+ }
+
+ pages := make([]*Page, len(candidates))
+ sem := make(chan struct{}, s3ScanConcurrency)
+ var wg sync.WaitGroup
+ var firstErr error
+ var errMu sync.Mutex
+
+ for i, cand := range candidates {
+ sem <- struct{}{}
+ wg.Go(func() {
+ defer func() { <-sem }()
+
+ head, err := c.s3cl.HeadObject(context.Background(), &s3.HeadObjectInput{
+ Bucket: aws.String(c.bucketName),
+ Key: aws.String(cand.key),
+ })
+ if err != nil {
+ errMu.Lock()
+ if firstErr == nil {
+ firstErr = fmt.Errorf("head S3 object %s: %w", cand.key, err)
}
- fileName := e.Link[strings.LastIndex(e.Link, "/")+1 : strings.LastIndex(e.Link, ".")]
- pages = append(pages, &Page{
- Link: e.Link,
- FileName: fileName,
- Lang: lang,
- ModifiedTime: e.ModifiedTime,
- Metadata: &frontmatter.Metadata{
- Title: e.Title,
- ShortDescription: e.ShortDescription,
- ActionDate: e.ActionDate,
- PublishedTime: e.PublishedTime,
- Thumbnail: e.Thumbnail,
- Tags: e.Tags,
- Geolocation: e.Geolocation,
- Medley: e.Medley,
- MedleyPart: e.MedleyPart,
- },
- })
+ errMu.Unlock()
+ return
}
- }
- case 2:
- for catKey, cat := range *idx.Categories.(*map[string]*IndexV2Category) {
- lang, ok := strings.CutPrefix(catKey, c.prefix)
- if !ok {
- continue
+
+ if head.ContentType == nil || !strings.Contains(*head.ContentType, "text/markdown") {
+ return
}
- if wantLang != "" && wantLang != lang {
- continue
+ meta := head.Metadata
+ if meta["title"] == "" {
+ return
}
- for codename, e := range cat.Pages {
- if !strings.HasPrefix(e.Link, fullPrefix) {
- continue
+
+ publishedTime, err := time.Parse(time.RFC3339, meta["published-time"])
+ if err != nil {
+ errMu.Lock()
+ if firstErr == nil {
+ firstErr = fmt.Errorf("failed to parse published time metadata field: %w", err)
}
- pages = append(pages, &Page{
- Link: e.Link,
- FileName: codename,
- Lang: lang,
- ModifiedTime: e.ModifiedTime,
- Metadata: &frontmatter.Metadata{
- Title: e.Title,
- ShortDescription: e.ShortDescription,
- ActionDate: e.ActionDate,
- PublishedTime: e.PublishedTime,
- Thumbnail: e.Thumbnail,
- Tags: e.Tags,
- Geolocation: e.Geolocation,
- Medley: e.Medley,
- MedleyPart: e.MedleyPart,
- },
- })
+ errMu.Unlock()
+ return
}
- }
+
+ linkParts := strings.Split(cand.key, "/")
+ nameParts := strings.Split(linkParts[len(linkParts)-1], ".")
+ fileName := strings.Join(nameParts[:len(nameParts)-1], ".")
+ lang, _ := strings.CutPrefix(linkParts[0], c.prefix)
+
+ tags := strings.Split(meta["tags"], ",")
+ slices.Sort(tags)
+
+ title, _ := url.QueryUnescape(meta["title"])
+ shortDescription, _ := url.QueryUnescape(meta["short-description"])
+ thumbnail, _ := url.QueryUnescape(meta["thumbnail"])
+
+ pages[i] = &Page{
+ Link: cand.key,
+ FileName: fileName,
+ Lang: lang,
+ ModifiedTime: cand.lastModified,
+ Metadata: &frontmatter.Metadata{
+ Title: title,
+ ShortDescription: shortDescription,
+ ActionDate: meta["action-date"],
+ PublishedTime: publishedTime,
+ Thumbnail: thumbnail,
+ Tags: tags,
+ Geolocation: meta["geolocation"],
+ },
+ }
+ })
}
+ wg.Wait()
- medleys, _ := c.GetMedleys()
- for _, medley := range medleys {
- for locale, localname := range medley.Localnames {
- l10n.T.SetPath(localname, true, locale, "Medleys", medley.Codename)
- }
+ if firstErr != nil {
+ return nil, firstErr
}
- return pages, nil
+ out := pages[:0]
+ for _, p := range pages {
+ if p != nil {
+ out = append(out, p)
+ }
+ }
+ return out, nil
}
func (c *S3Client) ReadAll(path string) ([]byte, error) {
- return c.readAll(c.prefix + path)
-}
-
-func (c *S3Client) readAll(path string) ([]byte, error) {
output, err := c.s3cl.GetObject(context.Background(), &s3.GetObjectInput{
Bucket: aws.String(c.bucketName),
- Key: aws.String(path),
+ Key: aws.String(c.prefix + path),
})
if err != nil {
return nil, fmt.Errorf("get S3 object: %w", err)
@@ -202,40 +277,10 @@ func (c *S3Client) readAll(path string) ([]byte, error) {
}
func (c *S3Client) ReadFrontmatter(path string) (metadata *frontmatter.Metadata, markdown []byte, err error) {
- idxRaw, err := c.readAll(IndexFileName)
- if err != nil {
- return nil, nil, fmt.Errorf("read %s: %w", IndexFileName, err)
- }
-
- var idx Index
- if err := json.Unmarshal(idxRaw, &idx); err != nil {
- return nil, nil, fmt.Errorf("unmarshal %s: %w", IndexFileName, err)
- }
-
contentBytes, err := c.ReadAll(path)
if err != nil {
return nil, nil, fmt.Errorf("failed to read file for frontmatter parsing: %w", err)
}
- switch idx.SchemaVersion {
- case 1:
- return frontmatter.ParseFrontmatter(contentBytes)
- case 2:
- fullPath := c.prefix + path
- page := (*idx.Categories.(*map[string]*IndexV2Category))[fullPath[:strings.LastIndex(fullPath, "/")]].Pages[fullPath[strings.LastIndex(fullPath, "/")+1:strings.LastIndex(fullPath, ".")]]
- metadata = page.Metadata()
-
- if !bytes.HasPrefix(contentBytes, []byte("---\n")) {
- return metadata, contentBytes, nil
- }
-
- end := bytes.Index(contentBytes[4:], []byte("\n---\n"))
- if end == -1 {
- return metadata, contentBytes, nil
- }
-
- return metadata, contentBytes[end+9:], nil
- }
-
return frontmatter.ParseFrontmatter(contentBytes)
}