2021-12-05 13:47:33 +00:00
|
|
|
package upstream
|
|
|
|
|
|
|
|
import (
|
2022-06-11 21:02:06 +00:00
|
|
|
"errors"
|
2022-11-12 19:37:20 +00:00
|
|
|
"fmt"
|
2021-12-05 13:47:33 +00:00
|
|
|
"io"
|
2022-11-12 19:37:20 +00:00
|
|
|
"net/http"
|
2021-12-05 13:47:33 +00:00
|
|
|
"strings"
|
|
|
|
"time"
|
|
|
|
|
|
|
|
"github.com/rs/zerolog/log"
|
2021-12-05 14:53:46 +00:00
|
|
|
|
|
|
|
"codeberg.org/codeberg/pages/html"
|
2023-03-30 21:36:31 +00:00
|
|
|
"codeberg.org/codeberg/pages/server/cache"
|
2022-11-12 19:37:20 +00:00
|
|
|
"codeberg.org/codeberg/pages/server/context"
|
2022-06-11 21:02:06 +00:00
|
|
|
"codeberg.org/codeberg/pages/server/gitea"
|
2021-12-05 13:47:33 +00:00
|
|
|
)
|
|
|
|
|
2022-11-12 19:37:20 +00:00
|
|
|
const (
|
|
|
|
headerLastModified = "Last-Modified"
|
|
|
|
headerIfModifiedSince = "If-Modified-Since"
|
|
|
|
|
|
|
|
rawMime = "text/plain; charset=utf-8"
|
|
|
|
)
|
|
|
|
|
2021-12-05 13:47:33 +00:00
|
|
|
// upstreamIndexPages lists pages that may be considered as index pages for directories.
|
|
|
|
var upstreamIndexPages = []string{
|
|
|
|
"index.html",
|
|
|
|
}
|
|
|
|
|
2022-06-12 01:50:00 +00:00
|
|
|
// upstreamNotFoundPages lists pages that may be considered as custom 404 Not Found pages.
|
|
|
|
var upstreamNotFoundPages = []string{
|
|
|
|
"404.html",
|
|
|
|
}
|
|
|
|
|
2021-12-05 13:47:33 +00:00
|
|
|
// Options provides various options for the upstream request.
|
|
|
|
type Options struct {
|
2022-11-12 19:43:44 +00:00
|
|
|
TargetOwner string
|
|
|
|
TargetRepo string
|
|
|
|
TargetBranch string
|
|
|
|
TargetPath string
|
2021-12-05 18:53:23 +00:00
|
|
|
|
2022-08-12 03:06:26 +00:00
|
|
|
// Used for debugging purposes.
|
|
|
|
Host string
|
|
|
|
|
2022-11-12 19:37:20 +00:00
|
|
|
TryIndexPages bool
|
|
|
|
BranchTimestamp time.Time
|
2021-12-05 16:57:54 +00:00
|
|
|
// internal
|
|
|
|
appendTrailingSlash bool
|
|
|
|
redirectIfExists string
|
2022-11-12 19:37:20 +00:00
|
|
|
|
|
|
|
ServeRaw bool
|
2021-12-05 13:47:33 +00:00
|
|
|
}
|
|
|
|
|
|
|
|
// Upstream requests a file from the Gitea API at GiteaRoot and writes it to the request context.
|
2024-02-15 16:08:29 +00:00
|
|
|
func (o *Options) Upstream(ctx *context.Context, giteaClient *gitea.Client, redirectsCache cache.ICache) bool {
|
2022-11-12 19:37:20 +00:00
|
|
|
log := log.With().Strs("upstream", []string{o.TargetOwner, o.TargetRepo, o.TargetBranch, o.TargetPath}).Logger()
|
|
|
|
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Debug().Msg("Start")
|
|
|
|
|
2022-11-12 19:37:20 +00:00
|
|
|
if o.TargetOwner == "" || o.TargetRepo == "" {
|
2024-02-11 12:43:25 +00:00
|
|
|
html.ReturnErrorPage(ctx, "forge client: either repo owner or name info is missing", http.StatusBadRequest)
|
2022-11-12 19:37:20 +00:00
|
|
|
return true
|
|
|
|
}
|
2021-12-05 13:47:33 +00:00
|
|
|
|
|
|
|
// Check if the branch exists and when it was modified
|
2021-12-05 18:53:23 +00:00
|
|
|
if o.BranchTimestamp.IsZero() {
|
2022-11-12 19:43:44 +00:00
|
|
|
branchExist, err := o.GetBranchTimestamp(giteaClient)
|
|
|
|
// handle 404
|
|
|
|
if err != nil && errors.Is(err, gitea.ErrorNotFound) || !branchExist {
|
|
|
|
html.ReturnErrorPage(ctx,
|
2023-11-16 17:11:35 +00:00
|
|
|
fmt.Sprintf("branch <code>%q</code> for <code>%s/%s</code> not found", o.TargetBranch, o.TargetOwner, o.TargetRepo),
|
2022-11-12 19:43:44 +00:00
|
|
|
http.StatusNotFound)
|
|
|
|
return true
|
|
|
|
}
|
2021-12-05 13:47:33 +00:00
|
|
|
|
2022-11-12 19:43:44 +00:00
|
|
|
// handle unexpected errors
|
|
|
|
if err != nil {
|
2022-11-12 19:37:20 +00:00
|
|
|
html.ReturnErrorPage(ctx,
|
2023-11-16 17:11:35 +00:00
|
|
|
fmt.Sprintf("could not get timestamp of branch <code>%q</code>: '%v'", o.TargetBranch, err),
|
2022-11-12 19:37:20 +00:00
|
|
|
http.StatusFailedDependency)
|
2021-12-05 13:47:33 +00:00
|
|
|
return true
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
// Check if the browser has a cached version
|
2022-11-12 19:37:20 +00:00
|
|
|
if ctx.Response() != nil {
|
2022-11-15 15:15:11 +00:00
|
|
|
if ifModifiedSince, err := time.Parse(time.RFC1123, ctx.Response().Header.Get(headerIfModifiedSince)); err == nil {
|
|
|
|
if ifModifiedSince.After(o.BranchTimestamp) {
|
2022-11-12 19:37:20 +00:00
|
|
|
ctx.RespWriter.WriteHeader(http.StatusNotModified)
|
|
|
|
log.Trace().Msg("check response against last modified: valid")
|
|
|
|
return true
|
|
|
|
}
|
2021-12-05 13:47:33 +00:00
|
|
|
}
|
2022-11-12 19:37:20 +00:00
|
|
|
log.Trace().Msg("check response against last modified: outdated")
|
2021-12-05 13:47:33 +00:00
|
|
|
}
|
2022-08-12 03:06:26 +00:00
|
|
|
|
|
|
|
log.Debug().Msg("Preparing")
|
2021-12-05 13:47:33 +00:00
|
|
|
|
2022-11-12 19:37:20 +00:00
|
|
|
reader, header, statusCode, err := giteaClient.ServeRawContent(o.TargetOwner, o.TargetRepo, o.TargetBranch, o.TargetPath)
|
|
|
|
if reader != nil {
|
|
|
|
defer reader.Close()
|
2021-12-05 13:47:33 +00:00
|
|
|
}
|
2022-11-07 22:09:41 +00:00
|
|
|
|
2022-08-12 03:06:26 +00:00
|
|
|
log.Debug().Msg("Aquisting")
|
2021-12-05 13:47:33 +00:00
|
|
|
|
2022-11-12 19:37:20 +00:00
|
|
|
// Handle not found error
|
|
|
|
if err != nil && errors.Is(err, gitea.ErrorNotFound) {
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Debug().Msg("Handling not found error")
|
2023-03-30 21:36:31 +00:00
|
|
|
// Get and match redirects
|
|
|
|
redirects := o.getRedirects(giteaClient, redirectsCache)
|
|
|
|
if o.matchRedirects(ctx, giteaClient, redirects, redirectsCache) {
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Trace().Msg("redirect")
|
2023-03-30 21:36:31 +00:00
|
|
|
return true
|
|
|
|
}
|
|
|
|
|
2021-12-05 16:57:54 +00:00
|
|
|
if o.TryIndexPages {
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Trace().Msg("try index page")
|
2021-12-05 16:57:54 +00:00
|
|
|
// copy the o struct & try if an index page exists
|
|
|
|
optionsForIndexPages := *o
|
2021-12-05 13:47:33 +00:00
|
|
|
optionsForIndexPages.TryIndexPages = false
|
2021-12-05 16:57:54 +00:00
|
|
|
optionsForIndexPages.appendTrailingSlash = true
|
2021-12-05 13:47:33 +00:00
|
|
|
for _, indexPage := range upstreamIndexPages {
|
2021-12-05 18:53:23 +00:00
|
|
|
optionsForIndexPages.TargetPath = strings.TrimSuffix(o.TargetPath, "/") + "/" + indexPage
|
2023-03-30 21:36:31 +00:00
|
|
|
if optionsForIndexPages.Upstream(ctx, giteaClient, redirectsCache) {
|
2021-12-05 13:47:33 +00:00
|
|
|
return true
|
|
|
|
}
|
|
|
|
}
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Trace().Msg("try html file with path name")
|
2021-12-05 13:47:33 +00:00
|
|
|
// compatibility fix for GitHub Pages (/example → /example.html)
|
2021-12-05 16:57:54 +00:00
|
|
|
optionsForIndexPages.appendTrailingSlash = false
|
2022-11-12 19:37:20 +00:00
|
|
|
optionsForIndexPages.redirectIfExists = strings.TrimSuffix(ctx.Path(), "/") + ".html"
|
2021-12-05 18:53:23 +00:00
|
|
|
optionsForIndexPages.TargetPath = o.TargetPath + ".html"
|
2023-03-30 21:36:31 +00:00
|
|
|
if optionsForIndexPages.Upstream(ctx, giteaClient, redirectsCache) {
|
2021-12-05 13:47:33 +00:00
|
|
|
return true
|
|
|
|
}
|
|
|
|
}
|
2022-11-12 19:37:20 +00:00
|
|
|
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Trace().Msg("not found")
|
|
|
|
|
2022-11-12 19:37:20 +00:00
|
|
|
ctx.StatusCode = http.StatusNotFound
|
2022-06-12 01:50:00 +00:00
|
|
|
if o.TryIndexPages {
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Trace().Msg("try not found page")
|
2022-06-12 01:50:00 +00:00
|
|
|
// copy the o struct & try if a not found page exists
|
|
|
|
optionsForNotFoundPages := *o
|
|
|
|
optionsForNotFoundPages.TryIndexPages = false
|
|
|
|
optionsForNotFoundPages.appendTrailingSlash = false
|
|
|
|
for _, notFoundPage := range upstreamNotFoundPages {
|
|
|
|
optionsForNotFoundPages.TargetPath = "/" + notFoundPage
|
2023-03-30 21:36:31 +00:00
|
|
|
if optionsForNotFoundPages.Upstream(ctx, giteaClient, redirectsCache) {
|
2022-06-12 01:50:00 +00:00
|
|
|
return true
|
|
|
|
}
|
|
|
|
}
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Trace().Msg("not found page missing")
|
2022-06-12 01:50:00 +00:00
|
|
|
}
|
2023-03-30 21:36:31 +00:00
|
|
|
|
2021-12-05 13:47:33 +00:00
|
|
|
return false
|
|
|
|
}
|
2022-11-12 19:37:20 +00:00
|
|
|
|
|
|
|
// handle unexpected client errors
|
|
|
|
if err != nil || reader == nil || statusCode != http.StatusOK {
|
|
|
|
log.Debug().Msg("Handling error")
|
|
|
|
var msg string
|
|
|
|
|
|
|
|
if err != nil {
|
2024-02-11 12:43:25 +00:00
|
|
|
msg = "forge client: returned unexpected error"
|
2022-11-12 19:37:20 +00:00
|
|
|
log.Error().Err(err).Msg(msg)
|
2023-11-16 17:11:35 +00:00
|
|
|
msg = fmt.Sprintf("%s: '%v'", msg, err)
|
2022-11-12 19:37:20 +00:00
|
|
|
}
|
|
|
|
if reader == nil {
|
2024-02-11 12:43:25 +00:00
|
|
|
msg = "forge client: returned no reader"
|
2022-11-12 19:37:20 +00:00
|
|
|
log.Error().Msg(msg)
|
|
|
|
}
|
|
|
|
if statusCode != http.StatusOK {
|
2024-02-11 12:43:25 +00:00
|
|
|
msg = fmt.Sprintf("forge client: couldn't fetch contents: <code>%d - %s</code>", statusCode, http.StatusText(statusCode))
|
2022-11-12 19:37:20 +00:00
|
|
|
log.Error().Msg(msg)
|
|
|
|
}
|
|
|
|
|
|
|
|
html.ReturnErrorPage(ctx, msg, http.StatusInternalServerError)
|
2021-12-05 13:47:33 +00:00
|
|
|
return true
|
|
|
|
}
|
|
|
|
|
|
|
|
// Append trailing slash if missing (for index files), and redirect to fix filenames in general
|
2021-12-05 16:57:54 +00:00
|
|
|
// o.appendTrailingSlash is only true when looking for index pages
|
2022-11-12 19:37:20 +00:00
|
|
|
if o.appendTrailingSlash && !strings.HasSuffix(ctx.Path(), "/") {
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Trace().Msg("append trailing slash and redirect")
|
2022-11-12 19:37:20 +00:00
|
|
|
ctx.Redirect(ctx.Path()+"/", http.StatusTemporaryRedirect)
|
2021-12-05 13:47:33 +00:00
|
|
|
return true
|
|
|
|
}
|
2023-02-11 03:12:42 +00:00
|
|
|
if strings.HasSuffix(ctx.Path(), "/index.html") && !o.ServeRaw {
|
2024-02-26 22:21:42 +00:00
|
|
|
log.Trace().Msg("remove index.html from path and redirect")
|
2022-11-12 19:37:20 +00:00
|
|
|
ctx.Redirect(strings.TrimSuffix(ctx.Path(), "index.html"), http.StatusTemporaryRedirect)
|
2021-12-05 13:47:33 +00:00
|
|
|
return true
|
|
|
|
}
|
2021-12-05 16:57:54 +00:00
|
|
|
if o.redirectIfExists != "" {
|
2022-11-12 19:37:20 +00:00
|
|
|
ctx.Redirect(o.redirectIfExists, http.StatusTemporaryRedirect)
|
2021-12-05 13:47:33 +00:00
|
|
|
return true
|
|
|
|
}
|
2022-11-07 22:09:41 +00:00
|
|
|
|
2022-11-12 19:37:20 +00:00
|
|
|
// Set ETag & MIME
|
2022-11-12 19:43:44 +00:00
|
|
|
o.setHeader(ctx, header)
|
2021-12-05 13:47:33 +00:00
|
|
|
|
2022-08-12 03:06:26 +00:00
|
|
|
log.Debug().Msg("Prepare response")
|
2021-12-05 13:47:33 +00:00
|
|
|
|
2022-11-12 19:37:20 +00:00
|
|
|
ctx.RespWriter.WriteHeader(ctx.StatusCode)
|
|
|
|
|
2021-12-05 13:47:33 +00:00
|
|
|
// Write the response body to the original request
|
2022-11-12 19:37:20 +00:00
|
|
|
if reader != nil {
|
|
|
|
_, err := io.Copy(ctx.RespWriter, reader)
|
|
|
|
if err != nil {
|
|
|
|
log.Error().Err(err).Msgf("Couldn't write body for %q", o.TargetPath)
|
|
|
|
html.ReturnErrorPage(ctx, "", http.StatusInternalServerError)
|
|
|
|
return true
|
2021-12-05 13:47:33 +00:00
|
|
|
}
|
|
|
|
}
|
2022-11-07 22:09:41 +00:00
|
|
|
|
2022-08-12 03:06:26 +00:00
|
|
|
log.Debug().Msg("Sending response")
|
2021-12-05 13:47:33 +00:00
|
|
|
|
|
|
|
return true
|
|
|
|
}
|