mirror of
https://codeberg.org/Codeberg/pages-server.git
synced 2024-11-26 05:55:27 +00:00
a6e9510c07
Hello 👋 since it affected my deployment of the pages server I started to look into the problem of the blank pages and think I found a solution for it: 1. There is no check if the file response is empty, neither in cache retrieval nor in writing of a cache. Also the provided method for checking for empty responses had a bug. 2. I identified the redirect response to be the issue here. There is a cache write with the full cache key (e. g. rawContent/user/repo|branch|route/index.html) happening in the handling of the redirect response. But the written body here is empty. In the triggered request from the redirect response the server then finds a cache item to the key and serves the empty body. A quick fix is the check for empty file responses mentioned in 1. 3. The decision to redirect the user comes quite far down in the upstream function. Before that happens a lot of stuff that may not be important since after the redirect response comes a new request anyway. Also, I suspect that this causes the caching problem because there is a request to the forge server and its error handling with some recursions happening before. I propose to move two of the redirects before "Preparing" 4. The recursion in the upstream function makes it difficult to understand what is actually happening. I added some more logging to have an easier time with that. 5. I changed the default behaviour to append a trailing slash to the path to true. In my tested scenarios it happened anyway. This way there is no recursion happening before the redirect. I am not developing in go frequently and rarely contribute to open source -> so feedback of all kind is appreciated closes #164 Reviewed-on: https://codeberg.org/Codeberg/pages-server/pulls/292 Reviewed-by: 6543 <6543@obermui.de> Reviewed-by: crapStone <codeberg@crapstone.dev> Co-authored-by: Hoernschen <julian.hoernschemeyer@mailbox.org> Co-committed-by: Hoernschen <julian.hoernschemeyer@mailbox.org>
221 lines
6.5 KiB
Go
221 lines
6.5 KiB
Go
package upstream
|
|
|
|
import (
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/rs/zerolog/log"
|
|
|
|
"codeberg.org/codeberg/pages/html"
|
|
"codeberg.org/codeberg/pages/server/cache"
|
|
"codeberg.org/codeberg/pages/server/context"
|
|
"codeberg.org/codeberg/pages/server/gitea"
|
|
)
|
|
|
|
const (
|
|
headerLastModified = "Last-Modified"
|
|
headerIfModifiedSince = "If-Modified-Since"
|
|
|
|
rawMime = "text/plain; charset=utf-8"
|
|
)
|
|
|
|
// upstreamIndexPages lists pages that may be considered as index pages for directories.
|
|
var upstreamIndexPages = []string{
|
|
"index.html",
|
|
}
|
|
|
|
// upstreamNotFoundPages lists pages that may be considered as custom 404 Not Found pages.
|
|
var upstreamNotFoundPages = []string{
|
|
"404.html",
|
|
}
|
|
|
|
// Options provides various options for the upstream request.
|
|
type Options struct {
|
|
TargetOwner string
|
|
TargetRepo string
|
|
TargetBranch string
|
|
TargetPath string
|
|
|
|
// Used for debugging purposes.
|
|
Host string
|
|
|
|
TryIndexPages bool
|
|
BranchTimestamp time.Time
|
|
// internal
|
|
appendTrailingSlash bool
|
|
redirectIfExists string
|
|
|
|
ServeRaw bool
|
|
}
|
|
|
|
// Upstream requests a file from the Gitea API at GiteaRoot and writes it to the request context.
|
|
func (o *Options) Upstream(ctx *context.Context, giteaClient *gitea.Client, redirectsCache cache.ICache) bool {
|
|
log := log.With().Strs("upstream", []string{o.TargetOwner, o.TargetRepo, o.TargetBranch, o.TargetPath}).Logger()
|
|
|
|
log.Debug().Msg("Start")
|
|
|
|
if o.TargetOwner == "" || o.TargetRepo == "" {
|
|
html.ReturnErrorPage(ctx, "forge client: either repo owner or name info is missing", http.StatusBadRequest)
|
|
return true
|
|
}
|
|
|
|
// Check if the branch exists and when it was modified
|
|
if o.BranchTimestamp.IsZero() {
|
|
branchExist, err := o.GetBranchTimestamp(giteaClient)
|
|
// handle 404
|
|
if err != nil && errors.Is(err, gitea.ErrorNotFound) || !branchExist {
|
|
html.ReturnErrorPage(ctx,
|
|
fmt.Sprintf("branch <code>%q</code> for <code>%s/%s</code> not found", o.TargetBranch, o.TargetOwner, o.TargetRepo),
|
|
http.StatusNotFound)
|
|
return true
|
|
}
|
|
|
|
// handle unexpected errors
|
|
if err != nil {
|
|
html.ReturnErrorPage(ctx,
|
|
fmt.Sprintf("could not get timestamp of branch <code>%q</code>: '%v'", o.TargetBranch, err),
|
|
http.StatusFailedDependency)
|
|
return true
|
|
}
|
|
}
|
|
|
|
// Check if the browser has a cached version
|
|
if ctx.Response() != nil {
|
|
if ifModifiedSince, err := time.Parse(time.RFC1123, ctx.Response().Header.Get(headerIfModifiedSince)); err == nil {
|
|
if ifModifiedSince.After(o.BranchTimestamp) {
|
|
ctx.RespWriter.WriteHeader(http.StatusNotModified)
|
|
log.Trace().Msg("check response against last modified: valid")
|
|
return true
|
|
}
|
|
}
|
|
log.Trace().Msg("check response against last modified: outdated")
|
|
}
|
|
|
|
log.Debug().Msg("Preparing")
|
|
|
|
reader, header, statusCode, err := giteaClient.ServeRawContent(o.TargetOwner, o.TargetRepo, o.TargetBranch, o.TargetPath)
|
|
if reader != nil {
|
|
defer reader.Close()
|
|
}
|
|
|
|
log.Debug().Msg("Aquisting")
|
|
|
|
// Handle not found error
|
|
if err != nil && errors.Is(err, gitea.ErrorNotFound) {
|
|
log.Debug().Msg("Handling not found error")
|
|
// Get and match redirects
|
|
redirects := o.getRedirects(giteaClient, redirectsCache)
|
|
if o.matchRedirects(ctx, giteaClient, redirects, redirectsCache) {
|
|
log.Trace().Msg("redirect")
|
|
return true
|
|
}
|
|
|
|
if o.TryIndexPages {
|
|
log.Trace().Msg("try index page")
|
|
// copy the o struct & try if an index page exists
|
|
optionsForIndexPages := *o
|
|
optionsForIndexPages.TryIndexPages = false
|
|
optionsForIndexPages.appendTrailingSlash = true
|
|
for _, indexPage := range upstreamIndexPages {
|
|
optionsForIndexPages.TargetPath = strings.TrimSuffix(o.TargetPath, "/") + "/" + indexPage
|
|
if optionsForIndexPages.Upstream(ctx, giteaClient, redirectsCache) {
|
|
return true
|
|
}
|
|
}
|
|
log.Trace().Msg("try html file with path name")
|
|
// compatibility fix for GitHub Pages (/example → /example.html)
|
|
optionsForIndexPages.appendTrailingSlash = false
|
|
optionsForIndexPages.redirectIfExists = strings.TrimSuffix(ctx.Path(), "/") + ".html"
|
|
optionsForIndexPages.TargetPath = o.TargetPath + ".html"
|
|
if optionsForIndexPages.Upstream(ctx, giteaClient, redirectsCache) {
|
|
return true
|
|
}
|
|
}
|
|
|
|
log.Trace().Msg("not found")
|
|
|
|
ctx.StatusCode = http.StatusNotFound
|
|
if o.TryIndexPages {
|
|
log.Trace().Msg("try not found page")
|
|
// copy the o struct & try if a not found page exists
|
|
optionsForNotFoundPages := *o
|
|
optionsForNotFoundPages.TryIndexPages = false
|
|
optionsForNotFoundPages.appendTrailingSlash = false
|
|
for _, notFoundPage := range upstreamNotFoundPages {
|
|
optionsForNotFoundPages.TargetPath = "/" + notFoundPage
|
|
if optionsForNotFoundPages.Upstream(ctx, giteaClient, redirectsCache) {
|
|
return true
|
|
}
|
|
}
|
|
log.Trace().Msg("not found page missing")
|
|
}
|
|
|
|
return false
|
|
}
|
|
|
|
// handle unexpected client errors
|
|
if err != nil || reader == nil || statusCode != http.StatusOK {
|
|
log.Debug().Msg("Handling error")
|
|
var msg string
|
|
|
|
if err != nil {
|
|
msg = "forge client: returned unexpected error"
|
|
log.Error().Err(err).Msg(msg)
|
|
msg = fmt.Sprintf("%s: '%v'", msg, err)
|
|
}
|
|
if reader == nil {
|
|
msg = "forge client: returned no reader"
|
|
log.Error().Msg(msg)
|
|
}
|
|
if statusCode != http.StatusOK {
|
|
msg = fmt.Sprintf("forge client: couldn't fetch contents: <code>%d - %s</code>", statusCode, http.StatusText(statusCode))
|
|
log.Error().Msg(msg)
|
|
}
|
|
|
|
html.ReturnErrorPage(ctx, msg, http.StatusInternalServerError)
|
|
return true
|
|
}
|
|
|
|
// Append trailing slash if missing (for index files), and redirect to fix filenames in general
|
|
// o.appendTrailingSlash is only true when looking for index pages
|
|
if o.appendTrailingSlash && !strings.HasSuffix(ctx.Path(), "/") {
|
|
log.Trace().Msg("append trailing slash and redirect")
|
|
ctx.Redirect(ctx.Path()+"/", http.StatusTemporaryRedirect)
|
|
return true
|
|
}
|
|
if strings.HasSuffix(ctx.Path(), "/index.html") && !o.ServeRaw {
|
|
log.Trace().Msg("remove index.html from path and redirect")
|
|
ctx.Redirect(strings.TrimSuffix(ctx.Path(), "index.html"), http.StatusTemporaryRedirect)
|
|
return true
|
|
}
|
|
if o.redirectIfExists != "" {
|
|
ctx.Redirect(o.redirectIfExists, http.StatusTemporaryRedirect)
|
|
return true
|
|
}
|
|
|
|
// Set ETag & MIME
|
|
o.setHeader(ctx, header)
|
|
|
|
log.Debug().Msg("Prepare response")
|
|
|
|
ctx.RespWriter.WriteHeader(ctx.StatusCode)
|
|
|
|
// Write the response body to the original request
|
|
if reader != nil {
|
|
_, err := io.Copy(ctx.RespWriter, reader)
|
|
if err != nil {
|
|
log.Error().Err(err).Msgf("Couldn't write body for %q", o.TargetPath)
|
|
html.ReturnErrorPage(ctx, "", http.StatusInternalServerError)
|
|
return true
|
|
}
|
|
}
|
|
|
|
log.Debug().Msg("Sending response")
|
|
|
|
return true
|
|
}
|