mirror of
https://github.com/alexbelgium/hassio-addons.git
synced 2026-08-16 18:12:29 +02:00
fix(claude_desktop): address CodeRabbit findings on TokenSave startup hardening
flock -n silently skipped TokenSave prep on lock contention with no retry until next restart; wait up to 60s instead (kernel releases flock the instant its owner exits, so only a truly stuck lock can't clear within that window). Quarantine fired on any sync failure after 3 retries, including transient causes (permissions, disk full, missing binary) unrelated to corruption. Now only quarantines when stderr names actual database corruption (SQLite malformed/not-a-database/disk-image wording); other failures leave the index untouched and retry next start. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -203,12 +203,17 @@ if $TOKENSAVE_ENABLED; then
|
|||||||
# Prepare the per-repo semantic graph defensively so a hard add-on stop or storage
|
# Prepare the per-repo semantic graph defensively so a hard add-on stop or storage
|
||||||
# hiccup can never leave a broken index that fails every subsequent boot:
|
# hiccup can never leave a broken index that fails every subsequent boot:
|
||||||
# * a startup-scoped flock serializes against an overlapping restart (and any git
|
# * a startup-scoped flock serializes against an overlapping restart (and any git
|
||||||
# post-commit/checkout sync hook that fires mid-boot), so two writers never race;
|
# post-commit/checkout sync hook that fires mid-boot); waits up to 60s for the
|
||||||
|
# other writer to finish rather than silently skipping, since a held lock clears
|
||||||
|
# itself the moment its holder exits or dies (the kernel releases flock on exit);
|
||||||
# * an existing index is refreshed with a cheap incremental `sync`, retried a few
|
# * an existing index is refreshed with a cheap incremental `sync`, retried a few
|
||||||
# times because SQLITE_BUSY under lock contention is transient, not corruption;
|
# times because SQLITE_BUSY under lock contention is transient, not corruption;
|
||||||
# * only a genuinely unreadable/malformed index (sync still failing after retries)
|
# * quarantine is reserved for sync failures whose stderr actually names database
|
||||||
# or a half-written one from an interrupted `init` is quarantined and rebuilt,
|
# corruption (SQLite's own "malformed"/"not a database"/"disk image" wording) or
|
||||||
# so the graph self-heals instead of propagating corruption;
|
# a half-written index from an interrupted `init` (sentinel-flagged). Any other
|
||||||
|
# failure (permissions, disk full, missing binary, ...) leaves the existing index
|
||||||
|
# untouched and simply retries on the next start — corruption should self-heal,
|
||||||
|
# a transient environment problem should not nuke a healthy graph;
|
||||||
# * `init` is bracketed by a sentinel file so an interrupted full build is detected
|
# * `init` is bracketed by a sentinel file so an interrupted full build is detected
|
||||||
# as incomplete on the next start and rebuilt rather than trusted.
|
# as incomplete on the next start and rebuilt rather than trusted.
|
||||||
# All file operations run as the abc runtime user because the repo `.tokensave`
|
# All file operations run as the abc runtime user because the repo `.tokensave`
|
||||||
@@ -223,10 +228,13 @@ if $TOKENSAVE_ENABLED; then
|
|||||||
initflag="$ts_dir/.init-incomplete"
|
initflag="$ts_dir/.init-incomplete"
|
||||||
mkdir -p "$ts_dir"
|
mkdir -p "$ts_dir"
|
||||||
exec 9>"$lock"
|
exec 9>"$lock"
|
||||||
if ! flock -n 9; then
|
if ! flock -w 60 9; then
|
||||||
echo "TokenSave: index busy for $repo_root; skipping startup sync" >&2
|
echo "TokenSave: index still locked for $repo_root after 60s; skipping startup sync" >&2
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
|
is_corruption() {
|
||||||
|
printf "%s" "$1" | grep -qiE "malformed|not a database|file is encrypted|disk image|database.*corrupt"
|
||||||
|
}
|
||||||
quarantine() {
|
quarantine() {
|
||||||
stamp="$(date +%Y%m%d-%H%M%S)"
|
stamp="$(date +%Y%m%d-%H%M%S)"
|
||||||
bdir="$ts_dir/corrupt-$stamp"
|
bdir="$ts_dir/corrupt-$stamp"
|
||||||
@@ -239,14 +247,20 @@ if $TOKENSAVE_ENABLED; then
|
|||||||
if [ -f "$db" ] && [ ! -f "$initflag" ]; then
|
if [ -f "$db" ] && [ ! -f "$initflag" ]; then
|
||||||
attempt=1
|
attempt=1
|
||||||
while :; do
|
while :; do
|
||||||
tokensave sync "$repo_root" && exit 0
|
sync_err="$(tokensave sync "$repo_root" 2>&1 1>/dev/null)" && exit 0
|
||||||
[ "$attempt" -ge 3 ] && break
|
[ "$attempt" -ge 3 ] && break
|
||||||
echo "TokenSave: sync attempt $attempt failed for $repo_root; retrying" >&2
|
echo "TokenSave: sync attempt $attempt failed for $repo_root; retrying" >&2
|
||||||
attempt=$((attempt + 1))
|
attempt=$((attempt + 1))
|
||||||
sleep 2
|
sleep 2
|
||||||
done
|
done
|
||||||
echo "TokenSave: sync failed after retries for $repo_root; rebuilding index" >&2
|
if is_corruption "$sync_err"; then
|
||||||
quarantine
|
echo "TokenSave: sync failed after retries for $repo_root (corruption detected); rebuilding index" >&2
|
||||||
|
quarantine
|
||||||
|
else
|
||||||
|
echo "TokenSave: sync failed after retries for $repo_root (no corruption signature); leaving index in place, will retry next start" >&2
|
||||||
|
echo "TokenSave: last sync error: $sync_err" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
elif [ -f "$db" ]; then
|
elif [ -f "$db" ]; then
|
||||||
echo "TokenSave: previous init did not finish for $repo_root; rebuilding index" >&2
|
echo "TokenSave: previous init did not finish for $repo_root; rebuilding index" >&2
|
||||||
quarantine
|
quarantine
|
||||||
|
|||||||
Reference in New Issue
Block a user