diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ebf9be7..bfe68a6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -103,12 +103,38 @@ jobs: with: fetch-depth: 0 + # cache: true is actions/cache underneath, keyed on the lock files. That needs a cache server, + # which act_runner ships and can be turned off — and when it is off, this does nothing at all and + # says nothing about it. See the step below, which is how you find out. - uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0 with: global-json-file: global.json cache: true cache-dependency-path: '**/packages.lock.json' + # ◆ ONE LINE, AND IT EXISTS BECAUSE THE ALTERNATIVE IS ASSUMING. + # + # A restore that is cold every single run and a cache working perfectly look identical from + # outside: both end in a green build, and the only difference is minutes. This job's whole + # package cache is the step above, so whether that step works is worth knowing as a fact rather + # than as a belief — and it is one `find` before anything has written to the folder. + # + # It does not fail the build. A runner with no cache server is slow rather than wrong, and a + # pipeline that refused to run on one would be a worse trade than the minutes. + - name: what the package cache brought + run: | + set -eu + + root="${NUGET_PACKAGES:-$HOME/.nuget/packages}" + + if [ -d "$root" ]; then + echo "$root holds $(find "$root" -mindepth 1 -maxdepth 1 -type d | wc -l) package(s) already." + else + echo "$root is not there, so this restore is cold." + echo "If every run says that, the cache above is doing nothing: check that this runner's" + echo "config.yaml has the cache server enabled, and that ACTIONS_CACHE_URL reaches it." + fi + # Locked mode fails if packages.lock.json does not match the project files, so a # dependency cannot change without the lock file change being reviewed. - name: restore @@ -381,12 +407,31 @@ jobs: # does and never otherwise. This runner is persistent, so the first build pays for a JDK, an # Android SDK and the android workload — several minutes — and every run after it is a cache hit # that prints two lines. A change to that Dockerfile is what buys a new one. + # ◆ BUILT ONCE PER DOCKERFILE, NOT ONCE PER RUN. + # + # The tag is the Dockerfile's own digest, so a tag that exists is by construction an image built + # from exactly this Dockerfile — and docker applies a tag only when the build succeeded, so it + # cannot be a half-built one. That makes `docker image inspect` a sound cache check rather than a + # guess, and it skips the context upload and the layer walk that a no-op `docker build` still does. + # + # The staleness this trades away is real and bounded. Everything that decides what is *in* the + # image — the command-line tools zip, the build-tools version, the workload — is pinned in the + # Dockerfile, so changing any of them changes the digest and rebuilds. What drifts is the JDK from + # apt. This is a build tool rather than something shipped, so that is the right way round; the + # image job takes the opposite trade with --pull, because what it builds is what users run. + # + # Deleting the image, or editing the Dockerfile, is how you force it. - name: the android toolchain image run: | set -eu tag="dodossh-android-build:$(sha256sum build/android-build.Dockerfile | cut -c1-16)" - docker build -t "$tag" -f build/android-build.Dockerfile build + + if docker image inspect "$tag" >/dev/null 2>&1; then + echo "$tag is already on this runner" + else + docker build -t "$tag" -f build/android-build.Dockerfile build + fi echo "ANDROID_BUILD_IMAGE=$tag" >> "$GITHUB_ENV" echo "$tag" diff --git a/scripts/ci-android.sh b/scripts/ci-android.sh index b9651fa..683fb1e 100644 --- a/scripts/ci-android.sh +++ b/scripts/ci-android.sh @@ -26,14 +26,37 @@ echo "==> restore" # Locked mode, the same gate the solution restore gives every other project. This head is not in # DodoSSH.slnx — it needs a workload the other jobs have no reason to install — so this is the only # place its lock file is ever checked. It has silently gone stale before. +# +# Restored without a channel and built with one, which is safe and worth one line: the two channel +# PropertyGroups change the application id, the version and some assembly metadata, and none of those +# is in the restore graph. If a channel ever gains a PackageReference of its own, this splits and the +# restore has to move under the same -p: as the build. +# +# Into /root/.nuget/packages, which the job mounts as a named volume — see the workflow. That is the +# whole of this job's package caching and it is what took a run from eight minutes to two and a half. dotnet restore "$project" --locked-mode -echo "==> build" -dotnet build "$project" \ - --no-restore --configuration Release \ - "-p:Aapt2ToolPath=$build_tools" - echo "==> package the nightly" +# ◆ ONE BUILD, WHERE THERE WERE TWO, AND THE FIRST WAS NOT A CHECK. +# +# There used to be a plain `dotnet build` above this. It looked like a cheap compile before the +# expensive packaging and it was neither: SignAndroidPackage depends on Build, so this line compiles +# everything anyway, and the build above ran with no -p:DodoChannel — which means it ran as the +# *release* channel. Different ApplicationId, different ApplicationVersion, different assembly +# metadata; MSBuild treats a different set of global properties as a different project instance, so +# not one output was reused. It was a complete second compile of the whole reference closure, +# producing an APK for a channel this job must never build, thrown away unread. +# +# Removing it does not lose a check. Anything that fails in a compile fails in this one, and the +# link-time failures below are the ones only packaging finds. +# +# Measured in the toolchain image with a warm package volume, same commit, same APK out of both: +# +# two builds 2m52 + 2m26 5m21 total +# one build 3m01 3m04 total +# +# versionCode 203, versionName 0.0.0-alpha.0.136, dev.dodotech.dodossh.nightly, either way. +# # Packaging rather than only compiling, because the two failures this head is most exposed to are both # link-time: a native library with no android ABI, and a managed assembly that resolves for net10.0 but # has nothing to dex. Neither shows up in a compile.