package com.example.snake import android.content.Context import android.graphics.* import android.media.AudioAttributes import android.media.AudioFormat import android.media.AudioManager import android.media.AudioTrack import android.media.ToneGenerator import android.os.Build import android.os.Bundle import android.os.Parcelable import android.os.VibrationEffect import android.os.Vibrator import android.os.VibratorManager import android.util.AttributeSet import android.util.Log import android.view.Choreographer import android.view.MotionEvent import android.view.View import java.io.BufferedInputStream import java.io.BufferedOutputStream import java.io.DataInputStream import java.io.DataOutputStream import java.io.FileInputStream import java.io.FileOutputStream import java.util.ArrayDeque import java.util.concurrent.Callable import java.util.concurrent.ExecutorService import java.util.concurrent.Executors import java.util.concurrent.Future import kotlin.math.abs import kotlin.math.max import kotlin.math.min import kotlin.random.Random class SnakeView @JvmOverloads constructor( context: Context, attrs: AttributeSet? = null ) : View(context, attrs) { private val cols = 15 private val rows = 15 private val total = cols * rows private var cell = 0f private var ox = 0f private var oy = 0f private data class P(val x: Int, val y: Int) private data class Candidate( val d: P, val score: Float, val reason: String, val legal: Boolean, val regionScore: Float = 0f, val mobilityScore: Float = 0f, val tailScore: Float = 0f, val foodScore: Float = 0f, val edgeScore: Float = 0f, val spaceScore: Float = 0f, val region: Int = 0, val mobility: Int = 0, val tailOk: Boolean = false, val foodDist: Int = -1, val ate: Boolean = false, val qValue: Float = 0f ) private data class Snapshot( var strategy: String = "V2 AUTO", var reason: String = "初始化", var danger: Int = 1, var region: Int = 1, var spaceRatio: Float = 1f, var tailReachable: Boolean = false, var foodReachable: Boolean = false, var foodDistance: Int = -1, var hunger: Int = 0, var chosen: P = P(0, 0), var candidates: List = emptyList(), var depth: Int = 0, var nodes: Int = 0, var hungerFactor: Float = 1f, var regionWeight: Float = 11f, var strategyId: Int = -1, var qValue: Float = 0f, var nVisits: Int = 0, var forceEatActive: Boolean = false, var forceEatSafe: Boolean = false, var safeFollowMode: Boolean = false ) private data class Sim(val body: ArrayDeque

, val ate: Boolean) private data class Particle(var x: Float, var y: Float, var vx: Float, var vy: Float, var life: Float, val color: Int) private data class FloatText(var x: Float, var y: Float, var life: Float, val text: String) private data class BeamNode( val body: ArrayDeque

, val dir: P, val food: P, val score: Float, val steps: Int, val ateFood: Boolean, val dead: Boolean, val loopPenalty: Float, val firstAction: P, val recentHeadCells: IntArray ) private val snake = ArrayDeque

() private val queue = ArrayDeque

() private val dirs = listOf(P(0, -1), P(0, 1), P(-1, 0), P(1, 0)) private var dir = P(1, 0) private var food = P(5, 5) private var score = 0 private var highScore = 0 private var money = 0 private var gameOver = false private var running = false private var aiMode = 1 private var forcedStrategy = -1 private var gameSpeed = 220L private val gameSpeedMin = 45L private val gameSpeedStart = 220L private var accumulator = 0L private var lastFrame = 0L private var combo = 0 private var hunger = 0 private var lastFreeRegion = 0f private var deathCause = "无" private var lastDeathInfo = "" private var ai = Snapshot(chosen = dir) private val sharedLock = Any() private val evolveLock = Any() private val POPULATION_SIZE = 50 @Volatile private var generation = 0 private val population = MutableList(POPULATION_SIZE) { TinyBrain() } @Volatile private var currentAgentIndex = 0 private var completedAgents = 0 private val currentScores = FloatArray(POPULATION_SIZE) @Volatile private var qWeight = 1.0f @Volatile private var nnWeight = 0.0f @Volatile private var bestScoreThisGen = 0f @Volatile private var bestScoreAllTime = 0f @Volatile private var deadEndPredicted = false @Volatile private var rolloutActive = false @Volatile private var rolloutSteps = 0 @Volatile private var foodSpaceRatio = 1f @Volatile private var evolving = false @Volatile private var generationTransitioning = false @Volatile private var beamCalls = 0L @Volatile private var beamSuccess = 0L @Volatile private var beamSelected = 0L @Volatile private var beamTotalNodes = 0L @Volatile private var beamBestScore = 0f private val visitedStates = java.util.Collections.newSetFromMap( java.util.concurrent.ConcurrentHashMap() ) private class ThreadStatus { @Volatile var alive = false @Volatile var agentId = -1 @Volatile var score = 0 @Volatile var steps = 0 @Volatile var phase = "空闲" } private val threadStatus = Array(8) { ThreadStatus() } private class EvoRecord( val gen: Int, val bestScore: Float, val eliteN: Int, val crossN: Int, val breedN: Int, val personality: String, val deaths: IntArray, val timestamp: Long ) private val evoHistory = java.util.concurrent.ConcurrentLinkedDeque() private val completedAgentIds = HashSet() private val agentCompleted = BooleanArray(POPULATION_SIZE) @Volatile private var generationToken = 0L @Volatile private var uniqueCompletedAgents = 0 private enum class V3Goal(val label: String, val short: String, val color: Int) { GET_FOOD("获取食物", "食", Color.rgb(46, 204, 113)), PRESERVE_SPACE("保持空间", "空", Color.rgb(52, 152, 219)), REACH_TAIL("追踪尾巴", "尾", Color.rgb(155, 89, 182)), ESCAPE_DANGER("逃离危险", "危", Color.rgb(231, 76, 60)), BREAK_LOOP("打破循环", "循", Color.rgb(241, 196, 15)), SURVIVE("极限求生", "生", Color.rgb(230, 126, 34)) } private data class V3Lesson( var badAction: Int, var cause: String, var penalty: Float, var confidence: Float, var occurrences: Int, var altAction: Int ) private class V3FailureMemory(private val capacity: Int = 8000) { private val memory = LinkedHashMap() @Synchronized fun remember(ctx: Long, action: Int, cause: String, penalty: Float, alt: Int): V3Lesson { val old = memory[ctx] return if (old == null) { val l = V3Lesson(action, cause, penalty, 0.30f, 1, alt) memory[ctx] = l if (memory.size > capacity) memory.remove(memory.keys.first()) l } else { old.occurrences++ old.cause = cause old.penalty = max(old.penalty, penalty) old.confidence = min(0.95f, 0.40f + old.occurrences * 0.18f) if (alt >= 0) old.altAction = alt old } } @Synchronized fun penaltyFor(ctx: Long, action: Int): Float { val l = memory[ctx] ?: return 0f if (l.badAction != action) return 0f return l.penalty * l.confidence } @Synchronized fun hardLesson(ctx: Long, action: Int): Boolean { val l = memory[ctx] ?: return false return l.badAction == action && l.occurrences >= 3 && l.confidence >= 0.70f } @Synchronized fun worstLessonText(): String { var best: V3Lesson? = null for (l in memory.values) if (best == null || l.occurrences > best!!.occurrences) best = l val b = best ?: return "暂无教训" return "死因=${b.cause} 重复${b.occurrences}次 置信=${"%.2f".format(b.confidence)}" } @Synchronized fun size(): Int = memory.size @Synchronized fun clear() = memory.clear() } private class V3StrategyGenome { var foodPriority = 1.00f var spacePriority = 1.25f var tailPriority = 0.90f var dangerAversion = 1.40f var loopAversion = 0.80f var hungerUrgency = 0.80f private var backup = floatArrayOf(1.00f, 1.25f, 0.90f, 1.40f, 0.80f, 0.80f) fun snapshot() { backup = floatArrayOf(foodPriority, spacePriority, tailPriority, dangerAversion, loopAversion, hungerUrgency) } fun revert() { foodPriority = backup[0]; spacePriority = backup[1]; tailPriority = backup[2] dangerAversion = backup[3]; loopAversion = backup[4]; hungerUrgency = backup[5] } fun normalize() { foodPriority = foodPriority.coerceIn(0.4f, 2.2f) spacePriority = spacePriority.coerceIn(0.6f, 2.4f) tailPriority = tailPriority.coerceIn(0.4f, 2.0f) dangerAversion = dangerAversion.coerceIn(0.8f, 2.6f) loopAversion = loopAversion.coerceIn(0.3f, 2.2f) hungerUrgency = hungerUrgency.coerceIn(0.4f, 2.0f) } fun selfAdjust(cause: String) { when (cause) { "SELF" -> { spacePriority += 0.05f; dangerAversion += 0.04f; tailPriority += 0.02f } "TRAP" -> { dangerAversion += 0.05f; spacePriority += 0.03f } "WALL" -> { dangerAversion += 0.03f; tailPriority += 0.03f } "HUNGER" -> { hungerUrgency += 0.04f; foodPriority += 0.03f; spacePriority -= 0.02f } } normalize() } fun evolve() { val sigma = 0.06f fun noise() = Random.nextFloat() * 2f * sigma - sigma foodPriority += noise(); spacePriority += noise(); tailPriority += noise() dangerAversion += noise(); loopAversion += noise(); hungerUrgency += noise() normalize() } fun values(): FloatArray = floatArrayOf(foodPriority, spacePriority, tailPriority, dangerAversion, loopAversion, hungerUrgency) } @Volatile private var v3FusionEnabled = true @Volatile private var v3Goal: V3Goal = V3Goal.GET_FOOD @Volatile private var v3LayerActive: Int = 4 @Volatile private var v3LayerWhy: String = "V4融合待命" @Volatile private var v3Regret: Float = 0f @Volatile private var v3LessonFires: Int = 0 @Volatile private var v3LoopHits: Int = 0 private val v3GoalCounter = IntArray(V3Goal.values().size) private val v3RecentHeads = ArrayDeque() private val v3Memory = V3FailureMemory(8000) private val v3Genome = V3StrategyGenome() private var v3GenomeBest = 0f private var v3LessonFlash = 0f private val v3CtxSalt = 0x5157L private val v3TrainCtxSalt = 0x5157L fun setV3FusionEnabled(on: Boolean) { v3FusionEnabled = on; v3LayerWhy = if (on) "V4融合已接管" else "V2基线模式" } fun isV3FusionEnabled(): Boolean = v3FusionEnabled private fun v3ActionIndexOf(d: P): Int = dirs.indexOfFirst { it == d } private fun v3DirLabel(d: P): String = when (d) { P(0, -1) -> "上"; P(0, 1) -> "下"; P(-1, 0) -> "左"; P(1, 0) -> "右"; else -> "?" } private fun v3RecordHead() { val h = snake.firstOrNull() ?: return v3RecentHeads.addLast(h.y * cols + h.x) while (v3RecentHeads.size > 24) v3RecentHeads.removeFirst() } private fun v3LoopRisk(): Boolean { if (v3RecentHeads.size < 8) return false val last8 = if (v3RecentHeads.size > 8) v3RecentHeads.drop(v3RecentHeads.size - 8) else v3RecentHeads return last8.toSet().size <= 5 } private fun v3SelectGoal(): V3Goal { if (legalDirectionMask() == 0) return V3Goal.ESCAPE_DANGER if (v3LoopRisk()) { v3LoopHits++; return V3Goal.BREAK_LOOP } val region = freeRegion(snake) val ratio = region.toFloat() / max(1, snake.size) val danger = calculateDanger() if (danger >= 4) return V3Goal.ESCAPE_DANGER if (ratio < 1.05f) return V3Goal.PRESERVE_SPACE if (snake.size >= total * 0.45f && tailReachable(snake)) return V3Goal.REACH_TAIL if (snake.size >= total * 0.55f) return V3Goal.SURVIVE return V3Goal.GET_FOOD } private fun legalDirectionMask(): Int { var m = 0 val head = snake.firstOrNull() ?: return 0 dirs.forEachIndexed { i, d -> val nx = head.x + d.x; val ny = head.y + d.y if (nx in 0 until cols && ny in 0 until rows && !snake.contains(P(nx, ny))) m = m or (1 shl i) } return m } private fun v3ContextHash(): Long { val head = snake.firstOrNull() ?: return 0L val regionBucket = (freeRegion(snake) / 4).coerceIn(0, 60) val lenBucket = (snake.size / 8).coerceIn(0, 30) val foodDxBucket = ((food.x - head.x) / 3).coerceIn(-5, 5) val foodDyBucket = ((food.y - head.y) / 3).coerceIn(-5, 5) val headingBucket = directionIndex(dir) val dangerBucket = calculateDanger().coerceIn(1, 5) val tailBucket = if (tailReachable(snake)) 1 else 0 var h = 1125899906842597L h = h * 31 + head.x; h = h * 31 + head.y h = h * 31 + foodDxBucket; h = h * 31 + foodDyBucket h = h * 31 + headingBucket; h = h * 31 + dangerBucket h = h * 31 + regionBucket; h = h * 31 + lenBucket; h = h * 31 + tailBucket return h xor v3CtxSalt } private fun v3PostCheck(originalDir: P, legal: List, state: Int): P { if (!v3FusionEnabled || legal.size <= 1) return originalDir val best = legal.firstOrNull { it.d == originalDir } ?: return originalDir val ctx = v3ContextHash() if (v3Memory.hardLesson(ctx, v3ActionIndexOf(originalDir))) { val alt = legal .filter { it.d != originalDir && v3ActionIndexOf(it.d) >= 0 && !v3Memory.hardLesson(ctx, v3ActionIndexOf(it.d)) } .maxByOrNull { it.region * v3Genome.spacePriority * 10f + (if (it.tailOk) v3Genome.tailPriority * 40f else 0f) } if (alt != null && alt.tailOk && alt.region >= best.region * 0.8f) { Log.d(LOG_TAG, "L5 TRIGGERED! Cause: 防止重蹈覆辙") v3LayerActive = 5; v3LayerWhy = "L5反事实:阻止重蹈覆辙" v3LessonFires++ return alt.d } } if (v3LoopRisk()) { val alt = legal .filter { it.d != originalDir && it.tailOk } .maxByOrNull { it.region * v3Genome.spacePriority + v3Genome.loopAversion * 8f } if (alt != null && alt.tailOk && alt.region >= best.region * 0.8f) { Log.d(LOG_TAG, "L6 TRIGGERED! Cause: 打破循环") v3LayerActive = 6; v3LayerWhy = "L6基因:循环厌恶接管" v3LoopHits++ return alt.d } } return originalDir } private fun v3AnalyzeDeath(cause: String) { if (!v3FusionEnabled) return val chosenAction = v3ActionIndexOf(ai.chosen) if (chosenAction < 0) return val ctx = v3ContextHash() val alt = ai.candidates .filter { it.legal && it.d != ai.chosen && v3ActionIndexOf(it.d) >= 0 } .maxByOrNull { it.region * 10f + (if (it.tailOk) 50f else 0f) } val severity = when (cause) { "SELF" -> 1.0f; "TRAP" -> 0.9f; "WALL" -> 0.85f; "HUNGER" -> 0.35f; else -> 0.5f } val lesson = v3Memory.remember(ctx, chosenAction, cause, severity, v3ActionIndexOf(alt?.d ?: ai.chosen)) v3Regret = severity * lesson.confidence v3LessonFires++; v3LessonFlash = 1f v3Genome.selfAdjust(cause) } private fun v3TrainRecordLesson(cause: String, head: P, tailOk: Boolean, len: Int, action: Int) { if (!v3FusionEnabled || action < 0) return var h = 1125899906842597L h = h * 31 + head.x; h = h * 31 + head.y h = h * 31 + len; h = h * 31 + (if (tailOk) 1 else 0) val ctx = h xor v3TrainCtxSalt val severity = when (cause) { "SELF" -> 0.9f; "TRAP" -> 0.8f; "WALL" -> 0.7f; else -> 0.3f } v3Memory.remember(ctx, action, cause, severity, -1) } private fun v3EvolveGenome() { synchronized(v3Genome) { v3Genome.snapshot() val before = v3GenomeBest v3Genome.evolve() if (before > 0f && bestScoreAllTime < before) v3Genome.revert() else v3GenomeBest = bestScoreAllTime } } private var wRegion = 11f private var wMobility = 35f private var wTailGood = 180f private var wTailBad = -250f private var wFoodNear = 300f private var wFoodAte = 850f private var wEdge = 55f private var wSpace = 30f private val wRegion0 = 11f private val wMobility0 = 35f private val wTailGood0 = 180f private val wTailBad0 = -250f private val wFoodNear0 = 300f private val wFoodAte0 = 850f private val wEdge0 = 55f private val wSpace0 = 30f private var aggression = 1.15f private var safetyMargin = 1.08f private var shortcutBonus = 0.90f private var deathWall = 0 private var deathSelf = 0 private var deathTrap = 0 private var totalGames = 0 private var lastLearnAction = "初始化" private val recentScores = ArrayDeque() private var bestRecentScore = 0 private var generationsWithoutImprovement = 0 private var previousGenerationBest = Float.NEGATIVE_INFINITY private companion object { const val V2_FOOD_DIR = 4 const val V2_DANGER = 16 const val V2_MOBILITY = 5 const val V2_SPACE = 4 const val V2_HUNGER = 5 const val V2_LENGTH = 4 const val V2_TAIL = 2 const val V2_HEADING = 1 const val V2_STATE_COUNT = V2_FOOD_DIR * V2_DANGER * V2_MOBILITY * V2_SPACE * V2_HUNGER * V2_LENGTH * V2_TAIL * V2_HEADING const val V2_ACTIONS = 4 const val V2_Q_SIZE = V2_STATE_COUNT * V2_ACTIONS const val V2_LOCKS = 64 const val GAMMA = 0.94f const val ALPHA_FAST = 0.18f const val ALPHA_NORMAL = 0.055f const val EPS_START = 0.22f const val EPS_MIN = 0.015f const val REWARD_STEP = -0.05f const val REWARD_FOOD = 10.0f const val REWARD_TAIL = 0.55f const val REWARD_DANGER = -0.30f const val REWARD_SPACE_DELTA = 0.08f const val DEATH_WALL = -14.0f const val DEATH_SELF = -18.0f const val DEATH_TRAP = -20.0f const val DEATH_HUNGER = -12.0f const val LOG_TAG = "SnakeTrain" const val HEARTBEAT_TIMEOUT_MS = 7_200_000L const val BEAM_ENABLED = true const val BEAM_WIDTH = 3 const val BEAM_DEPTH_SHORT = 5 const val BEAM_DEPTH_LONG = 7 const val BEAM_MIN_LENGTH = 10 const val BEAM_MAX_NODES = 80 const val BEAM_WEIGHT = 0.30f const val BEAM_FOOD_WEIGHT = 1.0f const val BEAM_SPACE_WEIGHT = 1.2f const val BEAM_TAIL_WEIGHT = 1.1f const val BEAM_SURVIVAL_WEIGHT = 2.0f const val BEAM_DANGER_WEIGHT = 1.5f const val BEAM_LOOP_WEIGHT = 0.8f const val BEAM_LONG_LENGTH = 35 const val BEAM_DEPTH_MAX = 8 } private val qV2 = FloatArray(V2_Q_SIZE) private val qTarget = FloatArray(V2_Q_SIZE) private val nV2 = IntArray(V2_Q_SIZE) @Volatile private var trainingRunId = 0L private val trainingLifecycleLock = Any() private val targetLock = Any() private val targetUpdateCounter = java.util.concurrent.atomic.AtomicInteger(0) private fun isTrainingRunActive(runId: Long): Boolean = trainActive && trainingRunId == runId private val qLocks = Array(V2_LOCKS) { Any() } @Volatile private var v2LearningSteps = 0L @Volatile private var v2Episodes = 0L @Volatile private var v2Epsilon = EPS_START private var lastV2State = -1 private var lastV2Action = -1 private fun qIndex(state: Int, action: Int): Int = state * V2_ACTIONS + action private fun qLock(index: Int): Any = qLocks[index and (V2_LOCKS - 1)] private fun qStateLock(state: Int): Any = qLocks[state and (V2_LOCKS - 1)] private fun qRead(state: Int, action: Int): Float { if (state !in 0 until V2_STATE_COUNT || action !in 0 until V2_ACTIONS) return 0f return synchronized(qLock(state)) { qV2[qIndex(state, action)] } } private fun qVisit(state: Int, action: Int): Int { if (state !in 0 until V2_STATE_COUNT || action !in 0 until V2_ACTIONS) return 0 return synchronized(qLock(state)) { nV2[qIndex(state, action)] } } private fun qMax(state: Int, legalMask: Int = 15): Float { if (state !in 0 until V2_STATE_COUNT) return 0f var bestAction = -1; var bestQ = -Float.MAX_VALUE synchronized(qLock(state)) { for (a in 0 until 4) { if ((legalMask and (1 shl a)) == 0) continue val v = qV2[qIndex(state, a)] if (v > bestQ) { bestQ = v; bestAction = a } } if (bestAction < 0) return 0f synchronized(targetLock) { return qTarget[qIndex(state, bestAction)] } } } private data class Experience( val state: Int, val action: Int, val reward: Float, val nextState: Int, val nextMask: Int, val terminal: Boolean, val tdError: Float = 0f ) private val replayBuffer = ArrayList(5000) private val REPLAY_CAPACITY = 5000 private fun qUpdate(state: Int, action: Int, reward: Float, nextState: Int, nextMask: Int, terminal: Boolean) { if (state !in 0 until V2_STATE_COUNT || action !in 0 until V2_ACTIONS) return visitedStates.add(state) val effectiveReward = reward val idx = qIndex(state, action) val nextBest = if (terminal) 0f else qMax(nextState, nextMask) val oldValue = synchronized(qStateLock(state)) { qV2[idx] } val tdErr = if (terminal) effectiveReward - oldValue else effectiveReward + GAMMA * nextBest - oldValue synchronized(replayBuffer) { replayBuffer.add(Experience(state, action, effectiveReward, nextState, nextMask, terminal, tdErr)) if (replayBuffer.size > REPLAY_CAPACITY) replayBuffer.removeAt(0) } synchronized(qStateLock(state)) { val visits = nV2[idx] val alpha = if (visits < 12) ALPHA_FAST else ALPHA_NORMAL val target = if (terminal) effectiveReward else effectiveReward + GAMMA * nextBest val old = qV2[idx] val updated = old + alpha * (target - old) qV2[idx] = updated.coerceIn(-10f, 10f) nV2[idx] = if (visits < Int.MAX_VALUE) visits + 1 else visits } synchronized(replayBuffer) { if (replayBuffer.size > 10) repeat(2) { val e = replayBuffer[Random.nextInt(replayBuffer.size)] evoUpdate(e.state, e.action, e.reward, e.nextState, e.nextMask, e.terminal) } } v2LearningSteps++ val cnt = targetUpdateCounter.incrementAndGet() if (cnt >= 50 && targetUpdateCounter.compareAndSet(cnt, 0)) { synchronized(targetLock) { for (i in qTarget.indices) qTarget[i] = qTarget[i] + 0.01f * (qV2[i] - qTarget[i]) } } } private fun evoUpdate(state: Int, action: Int, reward: Float, nextState: Int, nextMask: Int, terminal: Boolean) { if (state !in 0 until V2_STATE_COUNT || action !in 0 until V2_ACTIONS) return val idx = qIndex(state, action) val nextBest = if (terminal) 0f else qMax(nextState, nextMask) synchronized(qStateLock(state)) { val visits = nV2[idx] val alpha = if (visits < 12) ALPHA_FAST else ALPHA_NORMAL val target = if (terminal) reward else reward + GAMMA * nextBest val old = qV2[idx] val updated = old + alpha * (target - old) qV2[idx] = updated.coerceIn(-10f, 10f) nV2[idx] = if (visits < Int.MAX_VALUE) visits + 1 else visits } } private fun currentEpsilon(): Float { if (!trainingMode && !reinforceTraining) return 0f val e = v2Epsilon if (v2LearningSteps > 0L && v2LearningSteps % 4096L == 0L) v2Epsilon = max(EPS_MIN, v2Epsilon * 0.985f) return e } private inner class TinyBrain { var w1 = FloatArray(8 * 32) { Random.nextFloat() * 2f - 1f } var w2 = FloatArray(32 * 16) { Random.nextFloat() * 2f - 1f } var w3 = FloatArray(16 * 4) { Random.nextFloat() * 2f - 1f } fun think(inputs: FloatArray): FloatArray { val h1 = FloatArray(32) for (i in 0 until 32) { var s = 0f; for (j in 0 until 8) s += inputs[j] * w1[j * 32 + i]; h1[i] = max(0f, s) } val h2 = FloatArray(16) for (i in 0 until 16) { var s = 0f; for (j in 0 until 32) s += h1[j] * w2[j * 16 + i]; h2[i] = max(0f, s) } val outputs = FloatArray(4) for (i in 0 until 4) { var s = 0f; for (j in 0 until 16) s += h2[j] * w3[j * 4 + i]; outputs[i] = s } return outputs } fun copyFrom(src: TinyBrain) { System.arraycopy(src.w1, 0, w1, 0, w1.size) System.arraycopy(src.w2, 0, w2, 0, w2.size) System.arraycopy(src.w3, 0, w3, 0, w3.size) } fun breed(child: TinyBrain) { val mut = if (generation < 10) 0.18f else if (generation < 30) 0.10f else 0.05f for (i in w1.indices) child.w1[i] = w1[i] + (Random.nextFloat() - 0.5f) * mut for (i in w2.indices) child.w2[i] = w2[i] + (Random.nextFloat() - 0.5f) * mut for (i in w3.indices) child.w3[i] = w3[i] + (Random.nextFloat() - 0.5f) * mut } fun crossover(other: TinyBrain, child: TinyBrain, dynamicMut: Float) { val s1 = w1.size / 2; val s2 = w2.size / 2; val s3 = w3.size / 2 System.arraycopy(w1, 0, child.w1, 0, s1) System.arraycopy(other.w1, s1, child.w1, s1, w1.size - s1) System.arraycopy(w2, 0, child.w2, 0, s2) System.arraycopy(other.w2, s2, child.w2, s2, w2.size - s2) System.arraycopy(w3, 0, child.w3, 0, s3) System.arraycopy(other.w3, s3, child.w3, s3, w3.size - s3) val baseMut = if (generation < 10) 0.12f else 0.06f val mut = max(baseMut, dynamicMut.coerceIn(0.06f, 0.30f)) for (i in child.w1.indices) child.w1[i] += (Random.nextFloat() - 0.5f) * mut for (i in child.w2.indices) child.w2[i] += (Random.nextFloat() - 0.5f) * mut for (i in child.w3.indices) child.w3[i] += (Random.nextFloat() - 0.5f) * mut } } private fun evolveNextGeneration() { evolving = true try { v3EvolveGenome() generation++ val scoresCopy = synchronized(sharedLock) { currentScores.copyOf() } val sortedIndices = (0 until POPULATION_SIZE).filter { !scoresCopy[it].isNaN() }.sortedByDescending { scoresCopy[it] } val bestBrains = if (sortedIndices.isNotEmpty()) { sortedIndices.take(minOf(10, sortedIndices.size)).map { TinyBrain().also { c -> c.copyFrom(population[it]) } } } else { (0 until minOf(10, POPULATION_SIZE)).map { TinyBrain().also { c -> c.copyFrom(population[it]) } } } if (bestBrains.isEmpty()) return val generationBest = if (sortedIndices.isNotEmpty()) scoresCopy[sortedIndices[0]] else 0f if (generationBest > previousGenerationBest + 0.001f) generationsWithoutImprovement = 0 else generationsWithoutImprovement++ previousGenerationBest = max(previousGenerationBest, generationBest) if (generationBest > bestScoreThisGen) bestScoreThisGen = generationBest if (generationBest > bestScoreAllTime) bestScoreAllTime = generationBest val dynamicMut = when { generationsWithoutImprovement >= 20 -> 0.30f generationsWithoutImprovement >= 10 -> 0.22f generationsWithoutImprovement >= 5 -> 0.16f else -> 0.12f } val eliteCount = 10 val crossoverCount = 25 val mutatedEliteCount = 10 val immigrantCount = 5 for (i in 0 until POPULATION_SIZE) { try { when { i < eliteCount -> population[i] = TinyBrain().also { it.copyFrom(bestBrains[i % bestBrains.size]) } i < eliteCount + crossoverCount -> { val p1 = bestBrains[Random.nextInt(bestBrains.size)] var p2 = bestBrains[Random.nextInt(bestBrains.size)] if (bestBrains.size > 1) while (p2 === p1) p2 = bestBrains[Random.nextInt(bestBrains.size)] val child = TinyBrain(); p1.crossover(p2, child, dynamicMut); population[i] = child } i < eliteCount + crossoverCount + mutatedEliteCount -> { val parent = bestBrains[Random.nextInt(bestBrains.size)] val child = TinyBrain(); parent.breed(child); population[i] = child } else -> population[i] = TinyBrain() } } catch (t: Throwable) { Log.e(LOG_TAG, "breed agent=$i failed", t); population[i] = TinyBrain() } } synchronized(sharedLock) { for (i in 0 until POPULATION_SIZE) currentScores[i] = 0f } val personality = when { v3Genome.dangerAversion > 1.8f && v3Genome.spacePriority > 1.6f -> "稳健保守型" v3Genome.foodPriority > 1.6f && v3Genome.hungerUrgency > 1.4f -> "激进捕食型" v3Genome.loopAversion > 1.6f -> "反循环型" v3Genome.tailPriority > 1.5f -> "追尾保命型" else -> "均衡探索型" } evoHistory.addLast(EvoRecord(generation, generationBest, eliteCount, crossoverCount, mutatedEliteCount + immigrantCount, personality, intArrayOf(deathWall, deathSelf, deathTrap), System.currentTimeMillis())) while (evoHistory.size > 24) evoHistory.pollFirst() val baseRatio = (generation / 30f).coerceIn(0f, 0.45f) val scoreBonus = (generationBest / 25000f).coerceIn(0f, 0.10f) nnWeight = (baseRatio + scoreBonus).coerceIn(0f, 0.50f) qWeight = (1.0f - nnWeight).coerceIn(0.50f, 1.0f) bestScoreThisGen = 0f Log.d(LOG_TAG, "EVOLVE gen=$generation best=$generationBest nnW=$nnWeight qW=$qWeight stall=$generationsWithoutImprovement") Thread { try { saveTrainingState() } catch (_: Exception) {} }.apply { isDaemon = true; start() } } catch (t: Throwable) { Log.e(LOG_TAG, "evolveNextGeneration failed", t) } finally { evolving = false } } private fun buildInputs(head: P, target: P, currentBody: ArrayDeque

): FloatArray { val inputs = FloatArray(8) inputs[0] = if (isSafeForBody(head.x, head.y - 1, currentBody)) 0f else 1f inputs[1] = if (isSafeForBody(head.x, head.y + 1, currentBody)) 0f else 1f inputs[2] = if (isSafeForBody(head.x - 1, head.y, currentBody)) 0f else 1f inputs[3] = if (isSafeForBody(head.x + 1, head.y, currentBody)) 0f else 1f inputs[4] = if (target.y < head.y) 1f else 0f inputs[5] = if (target.y > head.y) 1f else 0f inputs[6] = if (target.x < head.x) 1f else 0f inputs[7] = if (target.x > head.x) 1f else 0f return inputs } private fun isSafeForBody(nx: Int, ny: Int, body: ArrayDeque

): Boolean { val p = P(nx, ny) if (!inside(p)) return false if (body.contains(p) && p != body.last()) return false return true } var onScoreChanged: ((Int) -> Unit)? = null var onMoneyChanged: ((Int) -> Unit)? = null private val prefs = context.getSharedPreferences("snake_prefs", Context.MODE_PRIVATE) private val vibrator: Vibrator? = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) { (context.getSystemService(Context.VIBRATOR_MANAGER_SERVICE) as? VibratorManager)?.defaultVibrator } else { @Suppress("DEPRECATION") context.getSystemService(Context.VIBRATOR_SERVICE) as? Vibrator } private val toneGen: ToneGenerator? = try { ToneGenerator(AudioManager.STREAM_MUSIC, 85) } catch (_: Throwable) { null } private var bgm: BgmPlayer? = null private var bodyColor = Color.rgb(46, 204, 113) private var headColor = Color.rgb(39, 174, 96) private var bgColor = Color.BLACK private var gridColor = Color.CYAN private var rainbowSkin = false private val hsv = IntArray(360) { Color.HSVToColor(floatArrayOf(it.toFloat(), 1f, 1f)) } private val particles = mutableListOf() private val floats = mutableListOf() private val paint = Paint(Paint.ANTI_ALIAS_FLAG) private val text = Paint(Paint.ANTI_ALIAS_FLAG) private val panel = Paint(Paint.ANTI_ALIAS_FLAG) private val border = Paint(Paint.ANTI_ALIAS_FLAG) private val gridPaint = Paint(Paint.ANTI_ALIAS_FLAG) private val snakePaint = Paint(Paint.ANTI_ALIAS_FLAG) private val headPaint = Paint(Paint.ANTI_ALIAS_FLAG) private val foodPaint = Paint(Paint.ANTI_ALIAS_FLAG) private val barPaint = Paint(Paint.ANTI_ALIAS_FLAG) private var flash = 0f private val hungerKillLimit = 500 private val safeFollowLength = 32 private var restartCountdown = 0L private var trainingMode = false private var reinforceTraining = false @Volatile private var lastHeartbeat = 0L @Volatile private var watchDogThread: Thread? = null @Volatile private var watchdogRestartCount = 0 private fun heartbeat() { lastHeartbeat = System.currentTimeMillis() } private fun startWatchDog() { stopWatchDog() lastHeartbeat = System.currentTimeMillis() watchDogThread = Thread { while (true) { try { Thread.sleep(1000) } catch (_: Exception) {} if (!reinforceTraining) continue val now = System.currentTimeMillis() if (now - lastHeartbeat > HEARTBEAT_TIMEOUT_MS) { watchdogRestartCount++ try { restartTrainingSafely() } catch (_: Exception) {} try { Thread.sleep(3000) } catch (_: Exception) {} } } }.also { it.priority = Thread.MIN_PRIORITY; it.start() } } private fun stopWatchDog() { watchDogThread?.interrupt(); watchDogThread = null } override fun onSaveInstanceState(): Parcelable? { val superState = super.onSaveInstanceState() val bundle = Bundle() bundle.putParcelable("super", superState) bundle.putBoolean("reinforceTraining", reinforceTraining) bundle.putBoolean("trainingMode", trainingMode) return bundle } override fun onRestoreInstanceState(state: Parcelable?) { if (state is Bundle) { val wasReinforce = state.getBoolean("reinforceTraining", false) trainingMode = state.getBoolean("trainingMode", false) super.onRestoreInstanceState(state.getParcelable("super")) if (wasReinforce) { startParallelTraining(resetCounters = false); startWatchDog() } } else super.onRestoreInstanceState(state) } private var renderSkipCounter = 0 private val reinforceButtonRect = RectF() private var lastReinforceTap = 0L @Volatile private var trainActive = false private val trainThreads: MutableList = mutableListOf() private val TRAIN_THREADS = 8 private val aiPool: ExecutorService = run { val cores = Runtime.getRuntime().availableProcessors().coerceIn(4, 8) Executors.newFixedThreadPool(cores) { r -> Thread(r, "snake-ai").apply { isDaemon = true; priority = Thread.NORM_PRIORITY } } } private val tlVisited: ThreadLocal = ThreadLocal.withInitial { BooleanArray(cols * rows) } private val tlQueue: ThreadLocal = ThreadLocal.withInitial { IntArray(cols * rows) } private var aiStepStartNs = 0L private var aiStepBudgetNs = 15_000_000L private fun budgetFor(speedMs: Long): Long = when { speedMs >= 150L -> 35_000_000L speedMs >= 80L -> 18_000_000L speedMs >= 55L -> 12_000_000L else -> 8_000_000L } private fun hungerForceEat(): Int = when { snake.size < 20 -> 50; snake.size < 35 -> 70; snake.size < 55 -> 100; snake.size < 80 -> 140; else -> 200 } private fun vibrateEat() { if (reinforceTraining || trainingMode) return vibrator?.let { try { if (Build.VERSION.SDK_INT >= 26) it.vibrate(VibrationEffect.createOneShot(18, VibrationEffect.DEFAULT_AMPLITUDE)) else { @Suppress("DEPRECATION") it.vibrate(18) } } catch (_: Throwable) {} } } private fun vibrateDeath() { if (reinforceTraining || trainingMode) return vibrator?.let { try { if (Build.VERSION.SDK_INT >= 26) it.vibrate(VibrationEffect.createOneShot(120, VibrationEffect.DEFAULT_AMPLITUDE)) else { @Suppress("DEPRECATION") it.vibrate(120) } } catch (_: Throwable) {} } } private fun playEatSound() { if (reinforceTraining || trainingMode) return try { toneGen?.startTone(ToneGenerator.TONE_PROP_BEEP, 55) } catch (_: Throwable) {} } private val frame = object : Choreographer.FrameCallback { override fun doFrame(ns: Long) { if (!running) return if (lastFrame == 0L) lastFrame = ns val dt = ((ns - lastFrame) / 1_000_000L).coerceAtMost(100L) lastFrame = ns if (reinforceTraining) { if ((++renderSkipCounter % 2) == 0) invalidate(); Choreographer.getInstance().postFrameCallback(this); return } if (gameOver) restartCountdown = 0L else { accumulator += dt var stepCount = 0 val maxSteps = if (trainingMode) 10 else Int.MAX_VALUE while (accumulator >= gameSpeed && stepCount < maxSteps) { updateGame() if (gameOver && !trainingMode) break accumulator -= gameSpeed stepCount++ } if (stepCount >= maxSteps) accumulator = 0L } updateEffects(dt / 16f) val shouldRender = if (trainingMode) (++renderSkipCounter % 2) == 0 else true if (shouldRender) invalidate() Choreographer.getInstance().postFrameCallback(this) } } init { highScore = prefs.getInt("high_score", 0) money = prefs.getInt("money", 0) trainingMode = prefs.getBoolean("training_mode", false) aggression = prefs.getFloat("learn_aggression", 1.15f) safetyMargin = prefs.getFloat("learn_safety", 1.08f) shortcutBonus = prefs.getFloat("learn_shortcut", 0.90f) wRegion = prefs.getFloat("w_region", wRegion0) wMobility = prefs.getFloat("w_mobility", wMobility0) wTailGood = prefs.getFloat("w_tail_good", wTailGood0) wTailBad = prefs.getFloat("w_tail_bad", wTailBad0) wFoodNear = prefs.getFloat("w_food_near", wFoodNear0) wFoodAte = prefs.getFloat("w_food_ate", wFoodAte0) wEdge = prefs.getFloat("w_edge", wEdge0) wSpace = prefs.getFloat("w_space", wSpace0) fun fixW(v: Float, base: Float): Float = v.coerceIn(base * 0.85f, base * 1.15f) wRegion = fixW(wRegion, wRegion0); wMobility = fixW(wMobility, wMobility0) wTailGood = fixW(wTailGood, wTailGood0); wFoodNear = fixW(wFoodNear, wFoodNear0) wFoodAte = fixW(wFoodAte, wFoodAte0); wEdge = fixW(wEdge, wEdge0) wSpace = fixW(wSpace, wSpace0) wTailBad = wTailBad.coerceIn(wTailBad0 * 1.15f, wTailBad0 * 0.85f) aggression = aggression.coerceIn(0.95f, 1.35f) safetyMargin = safetyMargin.coerceIn(1.0f, 1.20f) lastLearnAction = prefs.getString("last_learn_action", "初始化") ?: "初始化" v3Genome.foodPriority = prefs.getFloat("v3_g_food", 1.00f) v3Genome.spacePriority = prefs.getFloat("v3_g_space", 1.25f) v3Genome.tailPriority = prefs.getFloat("v3_g_tail", 0.90f) v3Genome.dangerAversion = prefs.getFloat("v3_g_danger", 1.40f) v3Genome.loopAversion = prefs.getFloat("v3_g_loop", 0.80f) v3Genome.hungerUrgency = prefs.getFloat("v3_g_hunger", 0.80f) v3Genome.normalize() v3LessonFires = prefs.getInt("v3_lesson_fires", 0) v3LoopHits = prefs.getInt("v3_loop_hits", 0) deathWall = prefs.getInt("stat_wall", 0) deathSelf = prefs.getInt("stat_self", 0) deathTrap = prefs.getInt("stat_trap", 0) totalGames = prefs.getInt("stat_total", 0) bgm = BgmPlayer() updateCurrentSkin() updateCurrentBoard() loadTrainingState() reset() } fun getAIMode(): Int = aiMode fun setAIMode(m: Int) { aiMode = m; invalidate() } fun setTrainingMode(b: Boolean) { trainingMode = b if (b) { aiMode = 1; bgm?.stop(); reset() } else { stopParallelTraining(); if (running) bgm?.start(); reset() } invalidate() } fun setReinforceTraining(enable: Boolean) { if (enable) { startParallelTraining(); startWatchDog() } else { stopParallelTraining(); stopWatchDog() } invalidate() } fun isReinforceTraining(): Boolean = reinforceTraining fun isTrainingMode(): Boolean = trainingMode private fun restartTrainingSafely() { synchronized(sharedLock) { trainActive = false; reinforceTraining = false } val oldThreads = synchronized(trainThreads) { trainThreads.toList() } for (t in oldThreads) { try { t.interrupt() } catch (_: Throwable) {} } for (t in oldThreads) { try { t.join(500) } catch (_: InterruptedException) { break } } synchronized(trainThreads) { trainThreads.clear() } Thread.sleep(200) synchronized(sharedLock) { reinforceTraining = true } startParallelTraining(resetCounters = false) startWatchDog() } private fun startParallelTraining(resetCounters: Boolean = true) { synchronized(trainingLifecycleLock) { if (trainThreads.any { it.isAlive }) return if (trainActive) return trainingRunId++ val myRunId = trainingRunId reinforceTraining = true trainingMode = true aiMode = 1 bgm?.stop() gameOver = false trainActive = true trainThreads.clear() if (resetCounters) { currentAgentIndex = 0 completedAgents = 0 uniqueCompletedAgents = 0 bestScoreThisGen = 0f completedAgentIds.clear() java.util.Arrays.fill(agentCompleted, false) synchronized(sharedLock) { for (i in 0 until POPULATION_SIZE) currentScores[i] = 0f } } generationToken++ val myGenerationToken = generationToken generationTransitioning = false for (i in threadStatus.indices) { threadStatus[i].alive = false; threadStatus[i].agentId = -1 threadStatus[i].score = 0; threadStatus[i].steps = 0; threadStatus[i].phase = "空闲" } Log.d(LOG_TAG, "GEN_START gen=$generation token=$myGenerationToken run=$myRunId pop=$POPULATION_SIZE th=$TRAIN_THREADS") for (i in 0 until TRAIN_THREADS) { val threadIdx = i val t = Thread({ val game = TrainGame(seed = System.nanoTime() + i * 999983L, runId = myRunId, statusIndex = threadIdx) threadStatus[threadIdx].alive = true while (isTrainingRunActive(myRunId) && !Thread.currentThread().isInterrupted) { if (evolving || generationTransitioning || generationToken != myGenerationToken) { threadStatus[threadIdx].phase = "进化"; Thread.yield(); continue } val agentId: Int synchronized(sharedLock) { if (currentAgentIndex >= POPULATION_SIZE) agentId = -1 else { agentId = currentAgentIndex; currentAgentIndex++ } } if (agentId < 0) { threadStatus[threadIdx].phase = "等待"; Thread.yield(); continue } synchronized(sharedLock) { if (agentId !in 0 until POPULATION_SIZE || agentCompleted[agentId]) { Log.e(LOG_TAG, "INVALID_AGENT_ASSIGNMENT agent=$agentId gen=$generation") continue } } threadStatus[threadIdx].agentId = agentId threadStatus[threadIdx].phase = "训练" Log.d(LOG_TAG, "AGENT_START gen=$generation agent=$agentId thread=$threadIdx") var episodeCompleted = false try { episodeCompleted = game.playOneGame(agentId = agentId, expectedGenerationToken = myGenerationToken) } catch (t: Throwable) { Log.e(LOG_TAG, "Agent $agentId crashed outside playOneGame", t) } if (!episodeCompleted) { threadStatus[threadIdx].phase = "等待"; continue } var shouldEvolve = false synchronized(sharedLock) { if (generationToken != myGenerationToken) { Log.w(LOG_TAG, "STALE_AGENT_RESULT agent=$agentId expected=$myGenerationToken actual=$generationToken") } else if (agentId in 0 until POPULATION_SIZE && !agentCompleted[agentId]) { agentCompleted[agentId] = true completedAgentIds.add(agentId) completedAgents++ uniqueCompletedAgents++ Log.d(LOG_TAG, "AGENT_COMPLETE gen=$generation agent=$agentId done=$completedAgents/$POPULATION_SIZE") if (completedAgents == POPULATION_SIZE && currentAgentIndex == POPULATION_SIZE && !generationTransitioning) { generationTransitioning = true shouldEvolve = true Log.d(LOG_TAG, "GEN_COMPLETE gen=$generation done=$completedAgents/$POPULATION_SIZE") } } } if (shouldEvolve) { synchronized(evolveLock) { if (isTrainingRunActive(myRunId) && generationToken == myGenerationToken) { try { evolveNextGeneration() } catch (t: Throwable) { Log.e(LOG_TAG, "evolveNextGeneration failed", t) } finally { synchronized(sharedLock) { currentAgentIndex = 0; completedAgents = 0; uniqueCompletedAgents = 0 completedAgentIds.clear(); java.util.Arrays.fill(agentCompleted, false) generationToken++; generationTransitioning = false } Log.d(LOG_TAG, "GEN_NEXT_READY gen=$generation") } } else synchronized(sharedLock) { generationTransitioning = false } } } } threadStatus[threadIdx].alive = false; threadStatus[threadIdx].phase = "空闲"; threadStatus[threadIdx].agentId = -1 }, "snake-train-$i") t.isDaemon = true t.priority = Thread.NORM_PRIORITY t.start() trainThreads.add(t) } } } private fun stopParallelTraining() { val threadsToJoin: List synchronized(trainingLifecycleLock) { if (!trainActive && trainThreads.none { it.isAlive }) { reinforceTraining = false; trainingMode = false; return } trainingRunId++; trainActive = false; reinforceTraining = false; trainingMode = false threadsToJoin = trainThreads.toList() } for (t in threadsToJoin) { try { t.interrupt() } catch (_: Throwable) {} } for (t in threadsToJoin) { try { t.join(5000L) } catch (_: Throwable) {} } synchronized(trainingLifecycleLock) { trainThreads.removeAll { !it.isAlive } } saveLearning(); saveTrainingState() if (running) bgm?.start() reset() } fun setForcedStrategy(s: Int) { forcedStrategy = s; invalidate() } fun updateCurrentSkin(id: String? = null) { when (prefs.getString("equipped_skin", "green")) { "blue" -> setSnakeColors(Color.rgb(52, 152, 219), Color.rgb(41, 128, 185), false) "red" -> setSnakeColors(Color.rgb(46, 204, 113), Color.rgb(192, 57, 43), false) "purple" -> setSnakeColors(Color.rgb(155, 89, 182), Color.rgb(142, 68, 173), false) "gold" -> setSnakeColors(Color.rgb(241, 196, 15), Color.rgb(243, 156, 18), false) "rainbow" -> setSnakeColors(Color.WHITE, Color.WHITE, true) else -> setSnakeColors(Color.rgb(46, 204, 113), Color.rgb(39, 174, 96), false) } invalidate() } private fun setSnakeColors(body: Int, head: Int, rainbow: Boolean) { bodyColor = body; headColor = head; rainbowSkin = rainbow snakePaint.color = body snakePaint.style = Paint.Style.STROKE snakePaint.strokeCap = Paint.Cap.ROUND snakePaint.strokeJoin = Paint.Join.ROUND snakePaint.strokeWidth = 0f headPaint.color = head } fun updateCurrentBoard(id: String? = null) { when (prefs.getString("equipped_board", "dark")) { "light" -> { bgColor = Color.rgb(240, 240, 240); gridColor = Color.rgb(200, 200, 200) } "neon" -> { bgColor = Color.rgb(10, 25, 47); gridColor = Color.CYAN } "forest" -> { bgColor = Color.rgb(27, 46, 26); gridColor = Color.rgb(46, 74, 45) } "cyberpunk" -> { bgColor = Color.rgb(43, 15, 59); gridColor = Color.MAGENTA } "rainbow_board" -> { bgColor = Color.BLACK; gridColor = Color.WHITE } else -> { bgColor = Color.BLACK; gridColor = Color.CYAN } } invalidate() } fun reset() { snake.clear(); queue.clear() if (aiMode == 2) { snake.add(P(0, 0)); dir = P(0, 1) } else { snake.add(P(cols / 2, rows / 2)); dir = P(1, 0) } score = 0; combo = 0; hunger = 0 gameOver = false; deathCause = "无"; lastDeathInfo = "" gameSpeed = if (trainingMode && !reinforceTraining) 2L else gameSpeedStart accumulator = 0L; lastFrame = 0L particles.clear(); floats.clear() flash = 0f; restartCountdown = 0L lastV2State = -1; lastV2Action = -1 ai = Snapshot(chosen = dir) placeFood() lastFreeRegion = freeRegion(snake).toFloat() onScoreChanged?.invoke(score) invalidate() } fun resume() { if (running) return running = true; lastFrame = 0L if (!trainingMode) bgm?.start() Choreographer.getInstance().postFrameCallback(frame) } fun pause() { running = false; bgm?.stop() Choreographer.getInstance().removeFrameCallback(frame) } private fun updateGame() { if (gameOver) return heartbeat() if (hunger >= hungerKillLimit) { die("HUNGER"); return } if (aiMode != 0) { queue.clear(); queue.add(chooseMove()) } if (queue.isNotEmpty()) { val requested = queue.removeFirst() if (!isReverse(requested, dir) && legalDirection(requested)) dir = requested } if (dir == P(0, 0)) return val oldState = buildState(snake, dir, food, hunger) val oldAction = dirs.indexOfFirst { it == dir }.coerceAtLeast(0) val nh = P(snake.first().x + dir.x, snake.first().y + dir.y) if (!inside(nh)) { qTerminalFromMove(oldState, oldAction, DEATH_WALL); die("WALL"); return } val ate = nh == food val body = snake.toList() val hitIndex = body.indexOf(nh) val tail = snake.last() if (hitIndex >= 0 && !(nh == tail && !ate)) { val sim = simulateOn(ArrayDeque(snake), dir) val region = freeRegion(sim.body) val cause = if (region < max(2, (snake.size * safetyMargin).toInt())) "TRAP" else "SELF" val reward = if (cause == "TRAP") DEATH_TRAP else DEATH_SELF qTerminalFromMove(oldState, oldAction, reward); die(cause); return } snake.addFirst(nh); v3RecordHead() var reward = REWARD_STEP if (ate) { val comboBonus = min(combo, 30) * 3 val gain = 10 + comboBonus + snake.size score += gain; combo++; hunger = 0 money += gain * 3 + 30 if (score > highScore) highScore = score onScoreChanged?.invoke(score); onMoneyChanged?.invoke(money) prefs.edit().putInt("money", money).putInt("high_score", highScore).apply() gameSpeed = max(gameSpeedMin, gameSpeedStart - snake.size * 2L) vibrateEat(); playEatSound(); spawnFoodEffect(nh); placeFood() reward += REWARD_FOOD + combo * 0.06f } else { snake.removeLast(); hunger++; combo = max(0, combo - 1) val curFree = freeRegion(snake).toFloat() val spaceDelta = curFree - lastFreeRegion lastFreeRegion = curFree reward += spaceDelta * REWARD_SPACE_DELTA if (tailReachable(snake)) reward += REWARD_TAIL if (calculateDanger() >= 4) reward += REWARD_DANGER val hungerPenalty = -(hunger * hunger * 0.001f).coerceAtMost(0.15f) reward += hungerPenalty } val nextState = buildState(snake, dir, food, hunger) val nextMask = legalActionMask(snake, dir, food) qUpdate(oldState, oldAction, reward, nextState, nextMask, false) lastV2State = nextState; lastV2Action = oldAction } private fun directionIndex(d: P): Int = when (d) { P(0, -1) -> 0; P(0, 1) -> 1; P(-1, 0) -> 2; else -> 3 } private fun foodDirection(head: P, target: P): Int { val dx = target.x - head.x; val dy = target.y - head.y return if (abs(dx) >= abs(dy)) { if (dx >= 0) 3 else 2 } else { if (dy >= 0) 1 else 0 } } private fun dangerMask(body: ArrayDeque

, heading: P, target: P): Int { if (body.isEmpty()) return 15 val h = body.first(); var mask = 0 for (i in dirs.indices) { val d = dirs[i] if (isReverse(d, heading)) { mask = mask or (1 shl i); continue } val np = P(h.x + d.x, h.y + d.y) if (!inside(np)) { mask = mask or (1 shl i); continue } val ate = np == target if (body.contains(np) && !(np == body.last() && !ate)) { mask = mask or (1 shl i); continue } val sim = simulateForBody(body, d, target) if (sim.body.isEmpty()) { mask = mask or (1 shl i); continue } val region = freeRegion(sim.body) val tail = tailReachable(sim.body) val ratio = region.toFloat() / max(1, sim.body.size) if (!tail && sim.body.size > 8) mask = mask or (1 shl i) else if (ratio < 0.55f && sim.body.size > 12) mask = mask or (1 shl i) } return mask and 15 } private fun buildState(body: ArrayDeque

, heading: P, target: P, hungerValue: Int): Int { if (body.isEmpty()) return 0 val h = body.first() val foodD = foodDirection(h, target) val danger = dangerMask(body, heading, target) val mobility = countSafeMovesFor(body, heading, target).coerceIn(0, 4) val region = freeRegion(body) val ratio = region.toFloat() / max(1, body.size) val spaceBucket = when { ratio < 0.8f -> 0; ratio < 1.3f -> 1; ratio < 2.0f -> 2; else -> 3 } val hungerBucket = when { hungerValue < 10 -> 0; hungerValue < 25 -> 1; hungerValue < 50 -> 2; hungerValue < 100 -> 3; else -> 4 } val lengthBucket = when { body.size < 12 -> 0; body.size < 30 -> 1; body.size < 60 -> 2; else -> 3 } val tail = if (tailReachable(body)) 1 else 0 val headingIndex = directionIndex(heading) var s = foodD s = s * V2_DANGER + danger s = s * V2_MOBILITY + mobility s = s * V2_SPACE + spaceBucket s = s * V2_HUNGER + hungerBucket s = s * V2_LENGTH + lengthBucket s = s * V2_TAIL + tail s = s * V2_HEADING + headingIndex return s.coerceIn(0, V2_STATE_COUNT - 1) } private fun legalActionMask(body: ArrayDeque

, heading: P, target: P): Int { if (body.isEmpty()) return 0 val h = body.first(); var mask = 0 for (i in dirs.indices) { val d = dirs[i] if (isReverse(d, heading)) continue val np = P(h.x + d.x, h.y + d.y) if (!inside(np)) continue val ate = np == target if (body.contains(np) && !(np == body.last() && !ate)) continue mask = mask or (1 shl i) } return mask } private fun qTerminalFromMove(state: Int, action: Int, reward: Float) { qUpdate(state, action, reward, state, 0, true) } private fun selectV2Action(state: Int, legal: List, training: Boolean): Candidate { if (legal.isEmpty()) return Candidate(dir, -1e9f, "无合法动作", false) if (forcedStrategy >= 0) return bestSurvivalStep(legal) val epsilon = if (training) currentEpsilon() else 0f if (training && Random.nextFloat() < epsilon) { val safe = legal.filter { it.tailOk || it.region >= max(5, snake.size / 2) } return if (safe.isNotEmpty()) safe[Random.nextInt(safe.size)] else legal[Random.nextInt(legal.size)] } var best = legal.first(); var bestValue = -Float.MAX_VALUE for (candidate in legal) { val action = dirs.indexOfFirst { it == candidate.d } if (action < 0) continue val q = qRead(state, action); val visits = qVisit(state, action) val safetyBonus = when { candidate.tailOk && candidate.region >= snake.size * 1.5f -> 2.2f candidate.tailOk -> 1.2f candidate.region >= snake.size -> 0.35f else -> -1.5f } val foodBonus = if (candidate.ate) 4.0f else if (candidate.foodDist >= 0) 0.7f / (candidate.foodDist + 1) else 0f val visitBonus = 0.12f / kotlin.math.sqrt((visits + 1).toFloat()) val value = q + safetyBonus + foodBonus + visitBonus if (value > bestValue) { bestValue = value; best = candidate.copy(qValue = q) } } return best } private fun chooseMove(): P { aiStepStartNs = System.nanoTime(); aiStepBudgetNs = budgetFor(gameSpeed) if (v3FusionEnabled) { val v3g = v3SelectGoal(); v3Goal = v3g; v3GoalCounter[v3g.ordinal]++ } val futures: List> = dirs.map { d -> aiPool.submit(Callable { evaluate(d) }) } val candidates = futures.map { try { it.get() } catch (_: Throwable) { Candidate(P(0, 0), -1e9f, "AI异常", false) } } val legal = candidates.filter { it.legal } if (legal.isEmpty()) { ai = Snapshot(strategy = "NO MOVE", reason = "四个方向都无法前进", danger = 5, chosen = dir, candidates = candidates); return dir } val danger = calculateDanger() val regionNow = freeRegion(snake) val foodDistNow = distance(snake.first(), food, snake, true) val state = buildState(snake, dir, food, hunger) val safeFood = findSafeFoodStep() if (safeFood != null) { val action = dirs.indexOfFirst { it == safeFood }.coerceAtLeast(0) val q = qRead(state, action) lastV2State = state; lastV2Action = action v3LayerActive = 1; v3LayerWhy = "L1安全食物层(硬安全)" ai = Snapshot(strategy = "V2 SAFE FOOD", reason = "安全路径可吃食物,吃后尾巴仍可达", danger = danger, region = regionNow, spaceRatio = regionNow.toFloat() / max(1, snake.size), tailReachable = tailReachable(snake), foodReachable = true, foodDistance = foodDistNow, hunger = hunger, chosen = safeFood, candidates = candidates, depth = 2, nodes = legal.size, hungerFactor = hungerFactorValue(), regionWeight = wRegion, strategyId = 10, qValue = q, nVisits = qVisit(state, action), forceEatActive = false, forceEatSafe = true, safeFollowMode = false) return v3PostCheck(safeFood, legal, state) } val tailStep = followTailStep(legal) if (tailStep != null && hunger < hungerForceEat()) { val action = dirs.indexOfFirst { it == tailStep }.coerceAtLeast(0) val q = qRead(state, action) lastV2State = state; lastV2Action = action v3LayerActive = 2; v3LayerWhy = "L2尾巴追踪层(空间循环)" ai = Snapshot(strategy = "V2 TAIL", reason = "食物路线风险较高,沿尾巴保持循环空间", danger = danger, region = regionNow, spaceRatio = regionNow.toFloat() / max(1, snake.size), tailReachable = true, foodReachable = foodDistNow >= 0, foodDistance = foodDistNow, hunger = hunger, chosen = tailStep, candidates = candidates, depth = 1, nodes = legal.size, hungerFactor = hungerFactorValue(), regionWeight = wRegion, strategyId = 11, qValue = q, nVisits = qVisit(state, action), forceEatActive = false, forceEatSafe = false, safeFollowMode = true) return v3PostCheck(tailStep, legal, state) } val forceThreshold = hungerForceEat() if (hunger >= forceThreshold && foodDistNow >= 0) { val safeFoodCandidates = legal.filter { it.tailOk && it.foodDist >= 0 } if (safeFoodCandidates.isNotEmpty()) { val selected = selectV2Action(state, safeFoodCandidates, training = trainingMode || reinforceTraining) val action = dirs.indexOfFirst { it == selected.d }.coerceAtLeast(0) lastV2State = state; lastV2Action = action v3LayerActive = 3; v3LayerWhy = "L3饥饿压制层(强制进食)" ai = Snapshot(strategy = "V2 HUNGER", reason = "饥饿压力提高食物收益,但仍受安全屏蔽", danger = max(danger, 3), region = regionNow, spaceRatio = regionNow.toFloat() / max(1, snake.size), tailReachable = selected.tailOk, foodReachable = true, foodDistance = selected.foodDist, hunger = hunger, chosen = selected.d, candidates = candidates.map { if (it.d == selected.d) it.copy(qValue = qRead(state, action)) else it }, depth = 1, nodes = legal.size, hungerFactor = hungerFactorValue(), regionWeight = wRegion, strategyId = 12, qValue = qRead(state, action), nVisits = qVisit(state, action), forceEatActive = true, forceEatSafe = true, safeFollowMode = false) return v3PostCheck(selected.d, legal, state) } } val shielded = legal.filter { val ratio = it.region.toFloat() / max(1, snake.size); it.tailOk || ratio >= 1.0f || (snake.size < 15 && ratio >= 0.75f) } val pool = if (shielded.isNotEmpty()) shielded else legal val brain = if (trainingMode || reinforceTraining) population[currentAgentIndex.coerceAtLeast(0) % POPULATION_SIZE] else population[0] val nnInputs = buildInputs(snake.first(), food, snake) val nnVals = brain.think(nnInputs) val de = checkFoodDeadEnd() deadEndPredicted = de.first; foodSpaceRatio = de.second var deadEndPenalty = 0f; if (de.first) deadEndPenalty = -5f val rSteps = when { snake.size > 60 -> 5; snake.size > 25 -> 3; else -> 0 } rolloutActive = rSteps > 0; rolloutSteps = rSteps var best = pool.first(); var bestValue = -Float.MAX_VALUE for (candidate in pool) { val action = dirs.indexOfFirst { it == candidate.d }; if (action < 0) continue val q = qRead(state, action); val nn = nnVals[action]; val rule = candidate.score val rolloutScore = if (rSteps > 0) rolloutN(candidate.d, rSteps) else 0f val mixed = qWeight * q + nnWeight * nn * 5f + rule * 0.001f + deadEndPenalty + rolloutScore * 0.5f if (mixed > bestValue) { bestValue = mixed; best = candidate.copy(qValue = q) } } val action = dirs.indexOfFirst { it == best.d }.coerceAtLeast(0) lastV2State = state; lastV2Action = action v3LayerActive = 4; v3LayerWhy = "L4混合决策(Q+NN+规则)" ai = Snapshot(strategy = "V4混合进化 (Q+NN+记忆)", reason = "规则过滤后,Q表与NN选出最优动作", danger = danger, region = regionNow, spaceRatio = regionNow.toFloat() / max(1, snake.size), tailReachable = tailReachable(snake), foodReachable = foodDistNow >= 0, foodDistance = foodDistNow, hunger = hunger, chosen = best.d, candidates = candidates.map { if (it.d == best.d) it.copy(qValue = qRead(state, action)) else it }, depth = 1, nodes = pool.size, hungerFactor = hungerFactorValue(), regionWeight = wRegion, strategyId = 20, qValue = qRead(state, action), nVisits = qVisit(state, action), forceEatActive = false, forceEatSafe = false, safeFollowMode = false) return v3PostCheck(best.d, legal, state) } private fun hungerFactorValue(): Float = when { hunger >= 60 -> 4f; hunger >= 30 -> 2.5f; hunger >= 15 -> 1.6f; else -> 1f } private fun followTailStep(legal: List): P? { if (snake.size < 2) return null val path = shortestPath(snake.first(), snake.last(), snake, allowTail = true) ?: return null if (path.isEmpty()) return null val step = path.first(); if (legal.none { it.d == step }) return null val sim = simulate(snake.first(), step); if (sim.body.isEmpty()) return null if (!tailReachable(sim.body)) return null val r = freeRegion(sim.body) if (r < 3 && snake.size < total - 5) return null return step } private fun findSafeFoodStep(): P? { val path = shortestPath(snake.first(), food, snake, allowTail = true) ?: return null if (path.isEmpty()) return null val simBody = ArrayDeque(snake); var ate = false for (step in path) { val nh = P(simBody.first().x + step.x, simBody.first().y + step.y) if (!inside(nh)) return null val willEat = nh == food if (simBody.contains(nh) && !(nh == simBody.last() && !willEat)) return null simBody.addFirst(nh); if (!willEat) simBody.removeLast() else ate = true } if (!ate) return null if (!tailReachable(simBody)) return null val len = simBody.size; val freeLeft = total - len if (len > 180 && freeLeft < 4) return null if (len > 150 && freeLeft < 6) return null if (len > 120 && freeLeft < 10) return null if (len > 90 && freeLeft < 14) return null if (len > 60 && freeLeft < 20) return null if (len <= 60 && freeLeft >= 20) { val r = freeRegion(simBody); if (r < max(4, freeLeft / 3)) return null } return path.first() } private fun bestSurvivalStep(legal: List): Candidate { var best = legal.first(); var bestScore = -1e30f val len = snake.size; val tailPos = snake.last(); val bodySet = snake.toHashSet() val tailW = (wTailGood / wTailGood0).coerceIn(0.8f, 1.2f) val spaceW = (wSpace / wSpace0).coerceIn(0.8f, 1.2f) val foodW = (wFoodNear / wFoodNear0).coerceIn(0.8f, 1.2f) for (cand in legal) { val sim = simulate(snake.first(), cand.d); if (sim.body.isEmpty()) continue val nh = sim.body.first(); val r = freeRegion(sim.body) val t = tailReachable(sim.body); val mob = countSafeMoves(sim.body) val freeLeft = total - sim.body.size var sc = 0f if (t) sc += 50000f * tailW else { sc -= 200000f; if (sc > bestScore) { bestScore = sc; best = cand }; continue } sc += r * 250f * spaceW if (freeLeft > 0) sc += (r.toFloat() / freeLeft) * 15000f * spaceW sc += mob * 500f var bodyAdj = 0 for (d in dirs) { val p = P(nh.x + d.x, nh.y + d.y); if (bodySet.contains(p) || sim.body.contains(p)) bodyAdj++ } sc += bodyAdj * 600f val dToTail = distance(nh, tailPos, sim.body, true) if (dToTail >= 0) { sc += 8000f / (dToTail + 1); if (dToTail > 6) sc -= (dToTail - 6) * 150f } val fd = distance(nh, food, sim.body, true) if (fd >= 0) sc += (500f * foodW) / (fd + 1) if (len >= 100) sc += r * 300f val nx = sim.body.first().x; val ny = sim.body.first().y val edge = min(min(nx, cols - 1 - nx), min(ny, rows - 1 - ny)) sc -= max(0, 3 - edge) * 800f val hx0 = snake.first().x; val hy0 = snake.first().y val atEdgeNow = hx0 == 0 || hx0 == cols - 1 || hy0 == 0 || hy0 == rows - 1 if (!atEdgeNow && edge == 0 && snake.size > 20) sc -= 6000f if (nx == 0 || nx == cols - 1 || ny == 0 || ny == rows - 1) sc -= 500f if (sim.ate) sc += 5000f if (sc > bestScore) { bestScore = sc; best = cand } } return best } private fun evaluate(d: P): Candidate { if (!legalDirection(d)) return Candidate(d, -1e9f, "撞墙/身体", false) val sim = simulate(snake.first(), d) val region = freeRegion(sim.body); val tail = tailReachable(sim.body) val foodDist = distance(sim.body.first(), food, sim.body, true) val ate = sim.ate; val mobility = countSafeMoves(sim.body) val hungerFactor = hungerFactorValue() val lenBoost = if (snake.size >= safeFollowLength) (snake.size.toFloat() / safeFollowLength).coerceIn(1f, 3f) else 1f val regionScoreVal = region * wRegion * lenBoost val mobilityScoreVal = mobility * wMobility val tailScoreVal = if (tail) wTailGood * lenBoost else wTailBad * lenBoost var foodScoreVal = if (foodDist >= 0) hungerFactor * aggression * (wFoodNear / (foodDist + 1)) else -450f * hungerFactor if (ate) foodScoreVal += wFoodAte * aggression var eatPenalty = 0f if (ate) { val afterSize = sim.body.size val needSpace = (afterSize * safetyMargin).toInt() + 2 if (!tail || region < needSpace) eatPenalty = -7500f - snake.size * 100f else { val after = ArrayDeque(sim.body) val dx = after.last().x - after.first().x val dy = after.last().y - after.first().y val follow = when { abs(dx) >= abs(dy) && dx != 0 -> P(if (dx > 0) 1 else -1, 0) dy != 0 -> P(0, if (dy > 0) 1 else -1) else -> null } if (follow != null && canSim(after, follow)) { val next = simulateOn(after, follow) val r2 = freeRegion(next.body); val t2 = tailReachable(next.body) if (!t2 || r2 < (afterSize * 1.2f).toInt()) eatPenalty = -4500f - snake.size * 60f } } } val edge = min(min(sim.body.first().x, cols - 1 - sim.body.first().x), min(sim.body.first().y, rows - 1 - sim.body.first().y)) val edgeScoreVal = -max(0, 2 - edge) * wEdge val spacePenalty = max(0f, snake.size * safetyMargin - region.toFloat()) val spaceScoreVal = -spacePenalty * wSpace val totalScore = regionScoreVal + mobilityScoreVal + tailScoreVal + foodScoreVal + edgeScoreVal + spaceScoreVal + eatPenalty val reason = when { ate && eatPenalty < 0 -> "吃完会困死,避开" ate -> "马上吃到食物" tail -> "保持尾巴可达" else -> "扩大可用空间" } return Candidate(d = d, score = totalScore, reason = reason, legal = true, regionScore = regionScoreVal, mobilityScore = mobilityScoreVal, tailScore = tailScoreVal, foodScore = foodScoreVal, edgeScore = edgeScoreVal, spaceScore = spaceScoreVal, region = region, mobility = mobility, tailOk = tail, foodDist = foodDist, ate = ate) } private fun clampW(v: Float, lo: Float, hi: Float): Float = v.coerceIn(lo, hi) private fun adjustWeights(cause: String) { val beforeEdge = wEdge; val beforeTail = wTailGood; val beforeSpace = wSpace; val beforeFood = wFoodNear val beforeAgg = aggression; val beforeSafety = safetyMargin when (cause) { "WALL" -> { wEdge = clampW(wEdge * 1.03f, wEdge0 * 0.9f, wEdge0 * 1.1f); aggression = clampW(aggression * 0.99f, 0.95f, 1.35f) } "SELF" -> { wTailGood = clampW(wTailGood * 1.03f, wTailGood0 * 0.9f, wTailGood0 * 1.1f); wTailBad = clampW(wTailBad * 1.02f, wTailBad0 * 1.1f, wTailBad0 * 0.9f); safetyMargin = clampW(safetyMargin * 1.01f, 1.0f, 1.20f) } "TRAP" -> { wSpace = clampW(wSpace * 1.04f, wSpace0 * 0.9f, wSpace0 * 1.1f); wRegion = clampW(wRegion * 1.02f, wRegion0 * 0.9f, wRegion0 * 1.1f); safetyMargin = clampW(safetyMargin * 1.01f, 1.0f, 1.20f); aggression = clampW(aggression * 1.005f, 0.95f, 1.35f); wFoodAte = clampW(wFoodAte * 1.02f, wFoodAte0 * 0.9f, wFoodAte0 * 1.1f) } "HUNGER" -> { wFoodNear = clampW(wFoodNear * 1.04f, wFoodNear0 * 0.9f, wFoodNear0 * 1.1f); wFoodAte = clampW(wFoodAte * 1.03f, wFoodAte0 * 0.9f, wFoodAte0 * 1.1f); aggression = clampW(aggression * 1.02f, 0.95f, 1.35f); wSpace = clampW(wSpace * 0.98f, wSpace0 * 0.9f, wSpace0 * 1.1f); safetyMargin = clampW(safetyMargin * 0.99f, 1.0f, 1.20f) } } wRegion = wRegion * 0.95f + wRegion0 * 0.05f wMobility = wMobility * 0.95f + wMobility0 * 0.05f wEdge = wEdge * 0.95f + wEdge0 * 0.05f wSpace = wSpace * 0.95f + wSpace0 * 0.05f wFoodNear = wFoodNear * 0.95f + wFoodNear0 * 0.05f wTailGood = wTailGood * 0.95f + wTailGood0 * 0.05f wFoodAte = wFoodAte * 0.95f + wFoodAte0 * 0.05f lastLearnAction = when (cause) { "WALL" -> "边界${"%.0f→%.0f".format(beforeEdge, wEdge)} 攻${"%.2f→%.2f".format(beforeAgg, aggression)}" "SELF" -> "尾+${"%.0f→%.0f".format(beforeTail, wTailGood)} 安全${"%.2f→%.2f".format(beforeSafety, safetyMargin)}" "TRAP" -> "空间${"%.0f→%.0f".format(beforeSpace, wSpace)} 安全${"%.2f→%.2f".format(beforeSafety, safetyMargin)}" "HUNGER" -> "食近${"%.0f→%.0f".format(beforeFood, wFoodNear)} 攻${"%.2f→%.2f".format(beforeAgg, aggression)}" else -> "微调" } } private fun saveTrainingState() { try { val file = context.getFileStreamPath("snake_training.dat") val genCopy: Int; val scoreCopy: Float; val qCopy: FloatArray; val nCopy: IntArray; val popCopy: List synchronized(sharedLock) { genCopy = generation; scoreCopy = bestScoreAllTime qCopy = qV2.copyOf(); nCopy = nV2.copyOf() popCopy = population.map { b -> TinyBrain().apply { System.arraycopy(b.w1, 0, w1, 0, b.w1.size) System.arraycopy(b.w2, 0, w2, 0, b.w2.size) System.arraycopy(b.w3, 0, w3, 0, b.w3.size) } } } DataOutputStream(BufferedOutputStream(FileOutputStream(file))).use { out -> out.writeInt(genCopy); out.writeFloat(scoreCopy) for (brain in popCopy) { for (v in brain.w1) out.writeFloat(v) for (v in brain.w2) out.writeFloat(v) for (v in brain.w3) out.writeFloat(v) } for (q in qCopy) out.writeFloat(q) for (n in nCopy) out.writeInt(n) } } catch (e: Exception) { e.printStackTrace() } } private fun loadTrainingState() { try { val file = context.getFileStreamPath("snake_training.dat") if (!file.exists()) return DataInputStream(BufferedInputStream(FileInputStream(file))).use { input -> generation = input.readInt(); bestScoreAllTime = input.readFloat() for (i in 0 until POPULATION_SIZE) { val brain = TinyBrain() for (j in brain.w1.indices) brain.w1[j] = input.readFloat() for (j in brain.w2.indices) brain.w2[j] = input.readFloat() for (j in brain.w3.indices) brain.w3[j] = input.readFloat() population[i] = brain } for (i in qV2.indices) qV2[i] = input.readFloat() for (i in nV2.indices) nV2[i] = input.readInt() } } catch (e: Exception) { e.printStackTrace(); generation = 0; bestScoreAllTime = 0f for (i in 0 until POPULATION_SIZE) population[i] = TinyBrain() java.util.Arrays.fill(qV2, 0f); java.util.Arrays.fill(nV2, 0) } } private fun saveLearning() { synchronized(sharedLock) { prefs.edit() .putFloat("learn_aggression", aggression).putFloat("learn_safety", safetyMargin) .putFloat("learn_shortcut", shortcutBonus) .putFloat("w_region", wRegion).putFloat("w_mobility", wMobility) .putFloat("w_tail_good", wTailGood).putFloat("w_tail_bad", wTailBad) .putFloat("w_food_near", wFoodNear).putFloat("w_food_ate", wFoodAte) .putFloat("w_edge", wEdge).putFloat("w_space", wSpace) .putString("last_learn_action", lastLearnAction) .putFloat("v3_g_food", v3Genome.foodPriority).putFloat("v3_g_space", v3Genome.spacePriority) .putFloat("v3_g_tail", v3Genome.tailPriority).putFloat("v3_g_danger", v3Genome.dangerAversion) .putFloat("v3_g_loop", v3Genome.loopAversion).putFloat("v3_g_hunger", v3Genome.hungerUrgency) .putInt("v3_lesson_fires", v3LessonFires).putInt("v3_loop_hits", v3LoopHits) .putInt("stat_wall", deathWall).putInt("stat_self", deathSelf) .putInt("stat_trap", deathTrap).putInt("stat_total", totalGames) .putInt("money", money).putInt("high_score", highScore) .apply() } } private fun die(cause: String) { gameOver = true; deathCause = cause v3AnalyzeDeath(cause) when (cause) { "WALL" -> deathWall++; "SELF" -> deathSelf++; "HUNGER" -> deathTrap++; else -> deathTrap++ } adjustWeights(cause); totalGames++; heartbeat() recentScores.addLast(score) while (recentScores.size > 50) recentScores.removeFirst() if (score > bestRecentScore) bestRecentScore = score aggression = when (cause) { "WALL" -> max(0.9f, aggression * 0.99f) "SELF" -> max(0.9f, aggression * 0.995f) "HUNGER" -> min(1.4f, aggression * 1.03f) else -> max(0.9f, aggression * 0.99f) } if (cause == "TRAP") safetyMargin = min(1.20f, safetyMargin * 1.01f) if (cause == "HUNGER") safetyMargin = max(1.0f, safetyMargin * 0.99f) safetyMargin = safetyMargin * 0.95f + 1.08f * 0.05f aggression = aggression * 0.95f + 1.15f * 0.05f safetyMargin = safetyMargin.coerceIn(1.0f, 1.20f) aggression = aggression.coerceIn(0.95f, 1.35f) lastDeathInfo = "死因=$cause 长度=${snake.size} 分=$score 空间=${ai.region} 需求=${(snake.size * safetyMargin).toInt()}" saveLearning() if (trainingMode) { reset(); return } flash = 1f; vibrateDeath(); restartCountdown = 0L; invalidate() } private fun simulate(head: P, d: P): Sim = simulateOn(ArrayDeque(snake), d) private fun simulateOn(src: ArrayDeque

, d: P): Sim { val b = ArrayDeque(src); if (b.isEmpty()) return Sim(b, false) val nh = P(b.first().x + d.x, b.first().y + d.y) if (!inside(nh)) return Sim(b, false) val ate = nh == food if (b.contains(nh) && !(nh == b.last() && !ate)) return Sim(b, false) b.addFirst(nh); if (!ate) b.removeLast(); return Sim(b, ate) } private fun rolloutN(startDir: P, maxSteps: Int): Float { var body = ArrayDeque(snake); var dir = startDir; var score = 0f for (step in 0 until maxSteps) { var bestNext: P? = null; var bestSpace = -1 for (d in dirs) { val nh = P(body.first().x + d.x, body.first().y + d.y) if (!inside(nh)) continue val isTail = nh == body.last() if (body.contains(nh) && !isTail) continue var space = 0 for (sd in dirs) { val sx = nh.x + sd.x; val sy = nh.y + sd.y if (sx in 0 until cols && sy in 0 until rows) { if (!body.any { it.x == sx && it.y == sy }) space++ } } if (space > bestSpace) { bestSpace = space; bestNext = d } } if (bestNext == null) return -50f + step * -10f val sim = simulateOn(body, bestNext) body = sim.body; score += 5f; if (sim.ate) score += 15f; dir = bestNext } return score } private fun simulateForBody(src: ArrayDeque

, d: P, target: P): Sim { val b = ArrayDeque(src); if (b.isEmpty()) return Sim(b, false) val nh = P(b.first().x + d.x, b.first().y + d.y) if (!inside(nh)) return Sim(b, false) val ate = nh == target if (b.contains(nh) && !(nh == b.last() && !ate)) return Sim(b, false) b.addFirst(nh); if (!ate) b.removeLast(); return Sim(b, ate) } private fun canSim(body: ArrayDeque

, d: P): Boolean { if (body.isEmpty()) return false val h = body.first() val nh = P(h.x + d.x, h.y + d.y) if (!inside(nh)) return false val ate = nh == food return !body.contains(nh) || (nh == body.last() && !ate) } private fun legalDirection(d: P): Boolean = d != P(0, 0) && !isReverse(d, dir) && canSim(snake, d) private fun isReverse(a: P, b: P): Boolean = a.x == -b.x && a.y == -b.y private fun directionOfFirst(body: ArrayDeque

): P { if (body.size < 2) return dir return P(body.elementAt(0).x - body.elementAt(1).x, body.elementAt(0).y - body.elementAt(1).y) } private fun inside(p: P): Boolean = p.x in 0 until cols && p.y in 0 until rows private fun shortestPath(start: P, target: P, body: Collection

, allowTail: Boolean): List

? { if (start == target) return emptyList() if (!inside(start) || !inside(target)) return null val visited = tlVisited.get(); val queue = tlQueue.get() val parent = IntArray(total) { -1 }; val parentDir = arrayOfNulls

(total) java.util.Arrays.fill(visited, 0, total, false) for (p in body) { if (!inside(p)) continue; visited[p.y * cols + p.x] = true } if (allowTail && body.isNotEmpty()) { val last = body.last(); if (inside(last)) visited[last.y * cols + last.x] = false } val startIndex = start.y * cols + start.x; val targetIndex = target.y * cols + target.x visited[startIndex] = true var head = 0; var tail = 0; queue[tail++] = startIndex; var found = false while (head < tail) { val curr = queue[head++]; if (curr == targetIndex) { found = true; break } val cx = curr % cols; val cy = curr / cols val candidates = arrayOf( P(0, -1) to if (cy > 0) curr - cols else -1, P(0, 1) to if (cy < rows - 1) curr + cols else -1, P(-1, 0) to if (cx > 0) curr - 1 else -1, P(1, 0) to if (cx < cols - 1) curr + 1 else -1 ) for ((d, ni) in candidates) { if (ni < 0 || ni >= total) continue if (visited[ni]) continue visited[ni] = true; parent[ni] = curr; parentDir[ni] = d if (tail < total) queue[tail++] = ni } } if (!found) return null val steps = ArrayList

(); var cur = targetIndex while (cur != startIndex) { val d = parentDir[cur] ?: return null steps.add(d); cur = parent[cur]; if (cur < 0) return null } steps.reverse(); return steps } private fun freeRegion(body: ArrayDeque

): Int { if (body.isEmpty()) return 0 val visited = tlVisited.get(); val queue = tlQueue.get() java.util.Arrays.fill(visited, 0, total, false) for (p in body) { if (!inside(p)) continue; visited[p.y * cols + p.x] = true } val start = body.first(); if (!inside(start)) return 0 val startIndex = start.y * cols + start.x; visited[startIndex] = true var head = 0; var tail = 0; queue[tail++] = startIndex; var count = 0 while (head < tail) { val curr = queue[head++]; val cx = curr % cols; val cy = curr / cols; count++ if (cx > 0) { val ni = curr - 1; if (!visited[ni]) { visited[ni] = true; if (tail < total) queue[tail++] = ni } } if (cx < cols - 1) { val ni = curr + 1; if (!visited[ni]) { visited[ni] = true; if (tail < total) queue[tail++] = ni } } if (cy > 0) { val ni = curr - cols; if (!visited[ni]) { visited[ni] = true; if (tail < total) queue[tail++] = ni } } if (cy < rows - 1) { val ni = curr + cols; if (!visited[ni]) { visited[ni] = true; if (tail < total) queue[tail++] = ni } } } return count } private fun distance(start: P, target: P, body: Collection

, allowTail: Boolean): Int { if (!inside(start) || !inside(target)) return -1 if (start == target) return 0 val visited = tlVisited.get(); val queue = tlQueue.get() java.util.Arrays.fill(visited, 0, total, false) for (p in body) { if (!inside(p)) continue; visited[p.y * cols + p.x] = true } if (allowTail && body.isNotEmpty()) { val last = body.last(); if (inside(last)) visited[last.y * cols + last.x] = false } val startIndex = start.y * cols + start.x; visited[startIndex] = true var head = 0; var tail = 0; queue[tail++] = startIndex; var dist = 0 while (head < tail) { val layerSize = tail - head repeat(layerSize) { if (head >= tail) return@repeat val curr = queue[head++]; val cx = curr % cols; val cy = curr / cols if (cx == target.x && cy == target.y) return dist if (cx > 0) { val ni = curr - 1; if (!visited[ni]) { visited[ni] = true; if (tail < total) queue[tail++] = ni } } if (cx < cols - 1) { val ni = curr + 1; if (!visited[ni]) { visited[ni] = true; if (tail < total) queue[tail++] = ni } } if (cy > 0) { val ni = curr - cols; if (!visited[ni]) { visited[ni] = true; if (tail < total) queue[tail++] = ni } } if (cy < rows - 1) { val ni = curr + cols; if (!visited[ni]) { visited[ni] = true; if (tail < total) queue[tail++] = ni } } } dist++ } return -1 } private fun countSafeMoves(body: ArrayDeque

): Int = countSafeMovesFor(body, directionOfFirst(body), food) private fun countSafeMovesFor(body: ArrayDeque

, heading: P, target: P): Int { if (body.isEmpty()) return 0 val h = body.first() return dirs.count { d -> if (isReverse(d, heading)) return@count false val nh = P(h.x + d.x, h.y + d.y) if (!inside(nh)) return@count false val ate = nh == target !body.contains(nh) || (nh == body.last() && !ate) } } private fun tailReachable(body: ArrayDeque

): Boolean = if (body.isEmpty()) false else distance(body.first(), body.last(), body, true) >= 0 private fun checkFoodDeadEnd(): Pair { if (snake.isEmpty()) return Pair(false, 10f) val visited = Array(cols) { BooleanArray(rows) } val queue = ArrayDeque

(); queue.addLast(food); visited[food.x][food.y] = true var space = 0 while (!queue.isEmpty()) { val cur = queue.removeFirst(); space++ for (d in dirs) { val nx = cur.x + d.x; val ny = cur.y + d.y if (nx in 0 until cols && ny in 0 until rows && !visited[nx][ny]) { val isBody = snake.any { it.x == nx && it.y == ny } if (!isBody) { visited[nx][ny] = true; queue.addLast(P(nx, ny)) } } } } val ratio = space.toFloat() / snake.size return Pair(ratio < 1.2f, ratio) } private fun calculateDanger(): Int { val region = freeRegion(snake) val ratio = region.toFloat() / max(1, snake.size) val mobility = countSafeMoves(snake) return when { mobility <= 0 -> 5; ratio < 1.5f -> 5; ratio < 2.2f -> 4 ratio < 3.5f -> 3; ratio < 5f -> 2; else -> 1 } } private fun placeFood() { val free = ArrayList

() for (x in 0 until cols) for (y in 0 until rows) { val p = P(x, y); if (!snake.contains(p)) free.add(p) } if (free.isNotEmpty()) food = free[Random.nextInt(free.size)] } private fun spawnFoodEffect(p: P) { repeat(14) { particles.add(Particle(p.x.toFloat(), p.y.toFloat(), Random.nextFloat() - 0.5f, Random.nextFloat() - 0.5f, 1f, Color.YELLOW)) } floats.add(FloatText(p.x.toFloat(), p.y.toFloat(), 1f, "+10")) } private fun updateEffects(dt: Float) { flash = max(0f, flash - dt * 0.035f) v3LessonFlash = max(0f, v3LessonFlash - dt * 0.8f) particles.forEach { it.x += it.vx * dt; it.y += it.vy * dt; it.life -= dt * 0.035f } particles.removeAll { it.life <= 0f } floats.forEach { it.y -= dt * 0.04f; it.life -= dt * 0.025f } floats.removeAll { it.life <= 0f } } override fun onSizeChanged(w: Int, h: Int, oldw: Int, oldh: Int) { val hudReserve = 760f + 292f; val availTop = 12f val availBottom = (h - hudReserve).coerceAtLeast(availTop + h * 0.30f) val availH = availBottom - availTop cell = min(w.toFloat() / cols, availH / rows) ox = (w - cols * cell) / 2f val baseOy = availTop + (availH - rows * cell) / 2f val moveDown = cell * 2.5f; val maxOy = availBottom - rows * cell oy = (baseOy + moveDown).coerceAtMost(maxOy) } override fun onDraw(c: Canvas) { super.onDraw(c) if (reinforceTraining) { c.drawColor(Color.BLACK); text.textAlign = Paint.Align.LEFT; drawTrainingDashboard(c); return } if (trainingMode) { c.drawColor(Color.BLACK); text.textAlign = Paint.Align.LEFT; drawDebug(c); return } c.drawColor(bgColor) gridPaint.color = gridColor; gridPaint.style = Paint.Style.STROKE; gridPaint.strokeWidth = 1f for (i in 0..cols) c.drawLine(ox + i * cell, oy, ox + i * cell, oy + rows * cell, gridPaint) for (i in 0..rows) c.drawLine(ox, oy + i * cell, ox + cols * cell, oy + i * cell, gridPaint) drawFood(c); drawSnake(c); drawParticles(c); drawDebug(c) if (gameOver) drawGameOver(c) } private fun drawFood(c: Canvas) { val cx = ox + (food.x + 0.5f) * cell; val cy = oy + (food.y + 0.5f) * cell foodPaint.color = Color.argb(90, 255, 80, 80); c.drawCircle(cx, cy, cell * 0.38f, foodPaint) foodPaint.color = Color.RED; c.drawCircle(cx, cy, cell * 0.22f, foodPaint) foodPaint.color = Color.WHITE; c.drawCircle(cx, cy, cell * 0.07f, foodPaint) } private fun drawSnake(c: Canvas) { if (snake.isEmpty()) return snakePaint.strokeWidth = max(8f, cell * 0.62f) val list = snake.toList() for (i in 0 until list.size - 1) { val a = list[i]; val b = list[i + 1] snakePaint.color = if (rainbowSkin) hsv[(i * 23 + score) % 360] else bodyColor c.drawLine(ox + (a.x + 0.5f) * cell, oy + (a.y + 0.5f) * cell, ox + (b.x + 0.5f) * cell, oy + (b.y + 0.5f) * cell, snakePaint) } val h = snake.first() headPaint.color = headColor c.drawCircle(ox + (h.x + 0.5f) * cell, oy + (h.y + 0.5f) * cell, cell * 0.36f, headPaint) val ex = when { dir.x > 0 -> 0.12f; dir.x < 0 -> -0.12f; else -> 0f } val ey = when { dir.y > 0 -> 0.12f; dir.y < 0 -> -0.12f; else -> 0f } paint.color = Color.WHITE c.drawCircle(ox + (h.x + 0.5f) * cell + cell * (0.13f + ex), oy + (h.y + 0.5f) * cell + cell * (0.13f + ey), cell * 0.075f, paint) c.drawCircle(ox + (h.x + 0.5f) * cell - cell * (0.13f - ex), oy + (h.y + 0.5f) * cell - cell * (0.13f - ey), cell * 0.075f, paint) } private fun drawParticles(c: Canvas) { particles.forEach { paint.color = Color.argb((it.life * 255).toInt().coerceIn(0, 255), Color.red(it.color), Color.green(it.color), Color.blue(it.color)) c.drawCircle(ox + (it.x + 0.5f) * cell, oy + (it.y + 0.5f) * cell, max(2f, cell * 0.05f), paint) } text.textAlign = Paint.Align.CENTER; text.textSize = cell * 0.3f floats.forEach { text.color = Color.argb((it.life * 255).toInt().coerceIn(0, 255), 255, 215, 0) c.drawText(it.text, ox + (it.x + 0.5f) * cell, oy + (it.y + 0.3f) * cell, text) } } private fun drawDangerGauge(c: Canvas, cx: Float, cy: Float, r: Float, danger: Int) { val arcPaint = Paint(Paint.ANTI_ALIAS_FLAG) arcPaint.style = Paint.Style.STROKE; arcPaint.strokeWidth = 13f; arcPaint.strokeCap = Paint.Cap.ROUND val oval = RectF(cx - r, cy - r, cx + r, cy + r) arcPaint.color = Color.rgb(50, 50, 50); c.drawArc(oval, -90f, 360f, false, arcPaint) arcPaint.color = when (danger) { 1 -> Color.GREEN; 2 -> Color.rgb(150, 255, 80); 3 -> Color.YELLOW 4 -> Color.rgb(255, 150, 0); else -> Color.RED } c.drawArc(oval, -90f, 360f * (danger.coerceIn(1, 5) / 5f), false, arcPaint) } private val prevWeights = FloatArray(8) { 1f } private fun drawWeightBars(c: Canvas, left: Float, top: Float, w: Float, h: Float) { val names = listOf("区域", "机动", "尾+", "尾-", "食近", "食吃", "边界", "空间") val values = floatArrayOf( wRegion / wRegion0, wMobility / wMobility0, wTailGood / wTailGood0, wTailBad / wTailBad0, wFoodNear / wFoodNear0, wFoodAte / wFoodAte0, wEdge / wEdge0, wSpace / wSpace0 ) val barW = w / names.size; val maxRatio = 2.0f names.forEachIndexed { i, name -> val x = left + i * barW; val ratio = values[i].coerceIn(0f, maxRatio) val barH = h * (ratio / maxRatio) barPaint.style = Paint.Style.FILL; barPaint.color = Color.rgb(40, 40, 40) c.drawRect(x + 3f, top, x + barW - 8f, top + h, barPaint) barPaint.color = Color.rgb(100, 100, 100) val baseY = top + h - h * (1f / maxRatio) c.drawLine(x + 3f, baseY, x + barW - 8f, baseY, barPaint) val prevRatio = prevWeights.getOrElse(i) { 1f } val delta = ratio - prevRatio; prevWeights[i] = ratio barPaint.color = when { delta > 0.05f -> Color.rgb(46, 204, 113) delta < -0.05f -> Color.rgb(46, 204, 113) ratio > 1.3f -> Color.rgb(46, 204, 113) ratio > 1.1f -> Color.rgb(241, 196, 15) ratio < 0.8f -> Color.rgb(52, 152, 219) else -> Color.rgb(46, 204, 113) } c.drawRect(x + 3f, top + h - barH, x + barW - 8f, top + h, barPaint) text.textAlign = Paint.Align.CENTER; text.textSize = 12f; text.color = Color.LTGRAY c.drawText(name, x + (barW - 5f) / 2f, top + h + 15f, text) text.color = Color.WHITE c.drawText("${(values[i] * 100f).toInt()}%", x + (barW - 5f) / 2f, top + h + 30f, text) } text.textAlign = Paint.Align.LEFT } private fun drawLearningCurve(c: Canvas, left: Float, top: Float, w: Float, h: Float) { barPaint.style = Paint.Style.FILL; barPaint.color = Color.rgb(25, 25, 25) c.drawRect(left, top, left + w, top + h, barPaint) if (recentScores.size < 2) { text.textAlign = Paint.Align.CENTER; text.textSize = 13f; text.color = Color.GRAY c.drawText("数据不足", left + w / 2f, top + h / 2f, text) text.textAlign = Paint.Align.LEFT; return } val scores = recentScores.toList(); val maxScore = scores.max().coerceAtLeast(1) val stepX = w / (scores.size - 1).coerceAtLeast(1) val areaPath = Path(); areaPath.moveTo(left, top + h) scores.forEachIndexed { i, s -> areaPath.lineTo(left + i * stepX, top + h - h * (s.toFloat() / maxScore)) } areaPath.lineTo(left + w, top + h); areaPath.close() barPaint.color = Color.argb(60, 120, 255, 180); c.drawPath(areaPath, barPaint) val linePath = Path() scores.forEachIndexed { i, s -> val x = left + i * stepX; val y = top + h - h * (s.toFloat() / maxScore) if (i == 0) linePath.moveTo(x, y) else linePath.lineTo(x, y) } val curvePaint = Paint(Paint.ANTI_ALIAS_FLAG) curvePaint.style = Paint.Style.STROKE; curvePaint.strokeWidth = 3f curvePaint.color = Color.rgb(120, 255, 180); c.drawPath(linePath, curvePaint) text.textSize = 12f; text.color = Color.LTGRAY c.drawText("max $maxScore", left + 5f, top + 14f, text) } private fun drawDeathPie(c: Canvas, cx: Float, cy: Float, r: Float) { val totalD = deathWall + deathSelf + deathTrap barPaint.style = Paint.Style.FILL if (totalD == 0) { barPaint.color = Color.rgb(60, 60, 60); c.drawCircle(cx, cy, r, barPaint); return } val oval = RectF(cx - r, cy - r, cx + r, cy + r); var start = -90f val sweepW = 360f * deathWall / totalD if (sweepW > 0f) { barPaint.color = Color.rgb(255, 100, 100); c.drawArc(oval, start, sweepW, true, barPaint); start += sweepW } val sweepS = 360f * deathSelf / totalD if (sweepS > 0f) { barPaint.color = Color.rgb(100, 150, 255); c.drawArc(oval, start, sweepS, true, barPaint); start += sweepS } val sweepT = 360f * deathTrap / totalD if (sweepT > 0f) { barPaint.color = Color.rgb(255, 200, 100); c.drawArc(oval, start, sweepT, true, barPaint) } } private fun drawQHeatmap(c: Canvas, left: Float, top: Float, w: Float, h: Float) { val grid = 10; val cellW = w / grid; val cellH = h / grid barPaint.style = Paint.Style.FILL; barPaint.color = Color.rgb(35, 35, 35) c.drawRect(left, top, left + w, top + h, barPaint) val sampleList = visitedStates.take(100); var maxAbs = 0.001f for (s in sampleList) for (a in 0 until 4) { val qv = qRead(s, a); if (abs(qv) > maxAbs) maxAbs = abs(qv) } for (i in 0 until 100) { val px = left + (i % grid) * cellW; val py = top + (i / grid) * cellH if (i >= sampleList.size) barPaint.color = Color.rgb(30, 30, 35) else { val s = sampleList[i]; var bestQ = -Float.MAX_VALUE for (a in 0 until 4) { val qv = qRead(s, a); if (qv > bestQ) bestQ = qv } val t = (bestQ / maxAbs).coerceIn(-1f, 1f) barPaint.color = if (t >= 0f) Color.rgb((60 * (1f - t)).toInt(), (60 + 195 * t).toInt(), 60) else { val nt = -t; Color.rgb((60 + 195 * nt).toInt(), (60 * (1f - nt)).toInt(), (60 * (1f - nt)).toInt()) } } c.drawRect(px, py, px + cellW - 1f, py + cellH - 1f, barPaint) } } private fun drawV3Panel(c: Canvas, w: Float, mainTop: Float) { if (mainTop < 290f) return val panelH = 272f; val left = (width - w) / 2f; val top = mainTop - 14f - panelH panel.color = Color.argb(232, 0, 0, 0) c.drawRoundRect(left, top, left + w, top + panelH, 22f, 22f, panel) val pulse = 0.55f + 0.45f * kotlin.math.sin(System.currentTimeMillis() / 160f) text.isFakeBoldText = true; text.textSize = 15f; text.color = Color.rgb(255, 200, 100) text.textAlign = Paint.Align.LEFT c.drawText("【V4融合 · AI层级链】", left + 16f, top + 24f, text) text.isFakeBoldText = true; text.textSize = 14f; val goalColor = v3Goal.color text.color = Color.argb((90 + 165 * pulse).toInt().coerceIn(0, 255), (goalColor shr 16) and 0xFF, (goalColor shr 8) and 0xFF, goalColor and 0xFF) c.drawText("目标: ${v3Goal.label}", left + w - 210f, top + 24f, text) text.textSize = 12f text.color = if (v3FusionEnabled) Color.rgb(46, 204, 113) else Color.rgb(120, 120, 120) c.drawText(if (v3FusionEnabled) "融合:开" else "融合:关", left + w - 70f, top + 24f, text) val layerNames = arrayOf("L1安全食物", "L2尾巴", "L3饥饿", "L4混合", "L5反事实", "L6基因") val layerColors = intArrayOf( Color.rgb(46, 204, 113), Color.rgb(155, 89, 182), Color.rgb(230, 126, 34), Color.rgb(52, 152, 219), Color.rgb(231, 76, 60), Color.rgb(241, 196, 15) ) val chipGap = 6f; val chipW = (w - 32f - chipGap * 5f) / 6f val chipY = top + 34f; val chipH = 24f for (i in 0 until 6) { val cx = left + 16f + i * (chipW + chipGap) val active = (v3LayerActive == i + 1) && v3FusionEnabled panel.color = if (active) Color.argb((120 + 135 * pulse).toInt().coerceIn(0, 255), (layerColors[i] shr 16) and 0xFF, (layerColors[i] shr 8) and 0xFF, layerColors[i] and 0xFF) else Color.rgb(30, 30, 40) c.drawRoundRect(cx, chipY, cx + chipW, chipY + chipH, 8f, 8f, panel) text.isFakeBoldText = active; text.textSize = 10f text.color = if (active) Color.WHITE else Color.rgb(140, 140, 150) text.textAlign = Paint.Align.CENTER c.drawText(layerNames[i], cx + chipW / 2f, chipY + chipH / 2f + 4f, text) } text.textAlign = Paint.Align.LEFT text.isFakeBoldText = false; text.textSize = 12f; text.color = Color.rgb(180, 190, 205) val whyLine = if (v3FusionEnabled) { if (v3Regret > 0.01f) "$v3LayerWhy 遗憾=${"%.2f".format(v3Regret)}" else v3LayerWhy } else "V3层级已关闭,运行V2基线决策" c.drawText(whyLine, left + 16f, top + 76f, text) text.isFakeBoldText = true; text.textSize = 13f; text.color = Color.rgb(255, 150, 255) c.drawText("【V3基因柱状图】", left + 16f, top + 98f, text) val geneNames = arrayOf("食", "空", "尾", "危", "循", "饿") val gv = v3Genome.values() val geneColors = intArrayOf( Color.rgb(46, 204, 113), Color.rgb(52, 152, 219), Color.rgb(155, 89, 182), Color.rgb(231, 76, 60), Color.rgb(241, 196, 15), Color.rgb(230, 126, 34) ) val geneAreaLeft = left + 16f; val geneAreaW = w * 0.52f val barGap = 5f; val geneBarW = (geneAreaW - barGap * 5f) / 6f val geneBaseY = top + 158f; val geneMaxH = 44f for (i in 0 until 6) { val bx = geneAreaLeft + i * (geneBarW + barGap) val frac = (gv[i] / 2.6f).coerceIn(0.05f, 1f); val bh = geneMaxH * frac panel.color = Color.rgb(30, 30, 40) c.drawRoundRect(bx, geneBaseY - geneMaxH, bx + geneBarW, geneBaseY, 4f, 4f, panel) panel.color = geneColors[i] c.drawRoundRect(bx, geneBaseY - bh, bx + geneBarW, geneBaseY, 4f, 4f, panel) text.isFakeBoldText = false; text.textSize = 11f; text.color = Color.rgb(160, 165, 175) text.textAlign = Paint.Align.CENTER c.drawText(geneNames[i], bx + geneBarW / 2f, geneBaseY + 12f, text) text.textSize = 9f; text.color = Color.rgb(210, 210, 220) c.drawText(String.format("%.2f", gv[i]), bx + geneBarW / 2f, geneBaseY - bh - 3f, text) } text.textAlign = Paint.Align.LEFT val infoX = left + w * 0.60f text.isFakeBoldText = false; text.textSize = 12f; text.color = Color.rgb(180, 190, 205) c.drawText("失败记忆: ${v3Memory.size()} 条", infoX, top + 118f, text) c.drawText("教训触发: ${v3LessonFires} 次", infoX, top + 136f, text) c.drawText("循环打断: ${v3LoopHits} 次", infoX, top + 154f, text) c.drawText("Beam 命中: $beamSelected/$beamCalls", infoX, top + 172f, text) text.textSize = 12f if (v3LessonFlash > 0.01f) { val a = (v3LessonFlash * 255f).toInt().coerceIn(0, 255) text.isFakeBoldText = true text.color = Color.argb(a, 255, 90 + ((60 * pulse).toInt()), 40) c.drawText("⚠ 新教训已写入: ${v3Memory.worstLessonText()}", left + 16f, top + 196f, text) } else { text.isFakeBoldText = false; text.color = Color.rgb(130, 135, 145) c.drawText("最近教训: ${v3Memory.worstLessonText()}", left + 16f, top + 196f, text) } text.isFakeBoldText = true; text.textSize = 13f; text.color = Color.rgb(255, 150, 255) c.drawText("【V3目标分布饼图】", left + 16f, top + 224f, text) val pieCx = left + w - 64f; val pieCy = top + 206f; val pieR = 42f val totalCnt = v3GoalCounter.sum() if (totalCnt <= 0) { panel.color = Color.rgb(30, 30, 40); c.drawCircle(pieCx, pieCy, pieR, panel) } else { var startAngle = -90f for (g in V3Goal.values()) { val cnt = v3GoalCounter[g.ordinal]; if (cnt <= 0) continue val sweep = cnt * 360f / totalCnt panel.color = g.color c.drawArc(pieCx - pieR, pieCy - pieR, pieCx + pieR, pieCy + pieR, startAngle, sweep, true, panel) startAngle += sweep } panel.color = Color.argb(232, 0, 0, 0); c.drawCircle(pieCx, pieCy, pieR * 0.45f, panel) } text.isFakeBoldText = false; text.textSize = 11f; var ly = top + 242f for (g in V3Goal.values()) { val cnt = v3GoalCounter[g.ordinal] val pct = if (totalCnt > 0) cnt * 100 / totalCnt else 0 text.color = g.color; c.drawText("■", left + 18f, ly, text) text.color = Color.rgb(170, 175, 185) c.drawText("${g.label} $pct%", left + 34f, ly, text) ly += 14f; if (ly > top + panelH - 6f) break } text.isFakeBoldText = false; text.textSize = 14f; text.color = Color.WHITE } private fun drawTrainingDashboard(c: Canvas) { val W = width.toFloat(); val pad = W * 0.022f; var y = pad panel.color = Color.rgb(28, 24, 48) c.drawRoundRect(pad, y, W - pad, y + W * 0.155f, pad * 0.7f, pad * 0.7f, panel) text.textAlign = Paint.Align.LEFT; text.isFakeBoldText = true text.textSize = W * 0.062f; text.color = Color.rgb(150, 230, 255) c.drawText("🧬 第 $generation 代", pad * 1.6f, y + W * 0.068f, text) text.textSize = W * 0.034f text.color = if (evolving || generationTransitioning) Color.rgb(255, 180, 80) else Color.rgb(120, 230, 150) val headState = if (evolving || generationTransitioning) "⚙ 进化繁殖中..." else "▶ 训练运行中" text.textAlign = Paint.Align.RIGHT c.drawText(headState, W - pad * 1.6f, y + W * 0.06f, text) text.textAlign = Paint.Align.LEFT; text.textSize = W * 0.032f; text.color = Color.rgb(200, 200, 220) c.drawText("已分配 $currentAgentIndex/$POPULATION_SIZE 已完成 $completedAgents/$POPULATION_SIZE 总局数 $totalGames", pad * 1.6f, y + W * 0.118f, text) y += W * 0.155f + pad * 0.6f val barH = W * 0.022f panel.color = Color.rgb(50, 48, 66) c.drawRoundRect(pad, y, W - pad, y + barH, barH / 2, barH / 2, panel) val frac = (completedAgents.toFloat() / POPULATION_SIZE).coerceIn(0f, 1f) panel.color = Color.rgb(80, 220, 150) if (frac > 0.01f) c.drawRoundRect(pad, y, pad + (W - 2 * pad) * frac, y + barH, barH / 2, barH / 2, panel) y += barH + pad val cardW = (W - 3 * pad) / 2f; val cardH = W * 0.118f val phaseColor = mapOf( "捕食" to Color.rgb(255, 170, 60), "避险" to Color.rgb(255, 90, 90), "搜索" to Color.rgb(90, 170, 255), "进化" to Color.rgb(200, 130, 255), "等待" to Color.rgb(130, 130, 140), "空闲" to Color.rgb(100, 100, 110), "训练" to Color.rgb(120, 220, 150) ) for (ti in 0 until 8) { val col = ti % 2; val row = ti / 2 val x0 = pad + col * (cardW + pad); val y0 = y + row * (cardH + pad * 0.55f) val st = threadStatus[ti] panel.color = Color.rgb(24, 26, 38) c.drawRoundRect(x0, y0, x0 + cardW, y0 + cardH, pad * 0.5f, pad * 0.5f, panel) paint.color = phaseColor[st.phase] ?: Color.GRAY c.drawCircle(x0 + pad * 0.8f, y0 + pad * 0.9f, pad * 0.32f, paint) text.textSize = W * 0.029f; text.isFakeBoldText = true; text.color = Color.WHITE c.drawText("线程$ti · Agent ${if (st.agentId >= 0) st.agentId else "-"}", x0 + pad * 1.5f, y0 + pad * 1.15f, text) text.textAlign = Paint.Align.RIGHT; text.color = paint.color c.drawText(st.phase, x0 + cardW - pad * 0.7f, y0 + pad * 1.15f, text) text.textAlign = Paint.Align.LEFT; text.textSize = W * 0.045f; text.color = Color.rgb(150, 235, 170) c.drawText("${st.score}", x0 + pad * 0.9f, y0 + cardH - pad * 0.55f, text) text.textSize = W * 0.026f; text.color = Color.rgb(160, 165, 185); text.textAlign = Paint.Align.RIGHT c.drawText("步 ${st.steps}", x0 + cardW - pad * 0.7f, y0 + cardH - pad * 0.6f, text) text.textAlign = Paint.Align.LEFT } y += 4 * cardH + 3 * pad * 0.55f + pad * 0.4f val metrics = arrayOf( "本代最佳" to "%.0f".format(bestScoreThisGen), "历史最佳" to "%.0f".format(bestScoreAllTime), "Q权重" to "%.2f".format(qWeight), "NN权重" to "%.2f".format(nnWeight), "回放" to "${replayBuffer.size}", "ε" to "%.3f".format(v2Epsilon), "学习步" to "$v2LearningSteps", "Beam" to "$beamSelected/$beamCalls" ) val mW = (W - 5 * pad) / 4f; val mH = W * 0.105f for (mi in metrics.indices) { val col = mi % 4; val row = mi / 4 val x0 = pad + col * (mW + pad); val y0 = y + row * (mH + pad * 0.5f) panel.color = Color.rgb(30, 30, 44) c.drawRoundRect(x0, y0, x0 + mW, y0 + mH, pad * 0.4f, pad * 0.4f, panel) text.textSize = W * 0.024f; text.isFakeBoldText = false; text.color = Color.rgb(150, 155, 175) c.drawText(metrics[mi].first, x0 + pad * 0.5f, y0 + pad * 0.95f, text) text.textSize = W * 0.040f; text.isFakeBoldText = true; text.color = Color.WHITE c.drawText(metrics[mi].second, x0 + pad * 0.5f, y0 + mH - pad * 0.5f, text) } y += 2 * mH + pad * 0.5f + pad text.textSize = W * 0.034f; text.isFakeBoldText = true; text.color = Color.rgb(200, 170, 255) c.drawText("🧪 进化繁殖历程", pad * 0.4f, y, text); y += pad * 0.7f val fullHistory = evoHistory.toList() val history = if (fullHistory.size > 10) fullHistory.drop(fullHistory.size - 10) else fullHistory val rowH = W * 0.052f if (history.isEmpty()) { text.textSize = W * 0.028f; text.isFakeBoldText = false; text.color = Color.rgb(140, 140, 155) c.drawText("(等待第一代50只完成后开始进化…)", pad * 0.6f, y + rowH * 0.6f, text) y += rowH } for (rec in history) { text.textSize = W * 0.026f; text.isFakeBoldText = true; text.color = Color.WHITE c.drawText("G${rec.gen}", pad * 0.5f, y + rowH * 0.62f, text) val barX = pad * 2.6f; val barW = W * 0.46f; val bh2 = rowH * 0.42f val total = (rec.eliteN + rec.crossN + rec.breedN).coerceAtLeast(1) var bx = barX val seg = arrayOf(rec.eliteN to Color.rgb(70, 220, 130), rec.crossN to Color.rgb(80, 150, 255), rec.breedN to Color.rgb(255, 160, 70)) for ((cnt, col) in seg) { val wseg = barW * (cnt.toFloat() / total) if (wseg > 0.5f) { panel.color = col; c.drawRect(bx, y, bx + wseg, y + bh2, panel) } bx += wseg } text.textSize = W * 0.025f; text.isFakeBoldText = false; text.color = Color.rgb(180, 240, 190) c.drawText("最佳%.0f".format(rec.bestScore), barX + barW + pad * 0.5f, y + rowH * 0.55f, text) text.color = Color.rgb(230, 200, 140); text.textAlign = Paint.Align.RIGHT c.drawText(rec.personality, W - pad * 0.5f, y + rowH * 0.55f, text) text.textAlign = Paint.Align.LEFT; y += rowH + pad * 0.18f } y += pad * 0.4f text.textSize = W * 0.032f; text.isFakeBoldText = true; text.color = Color.rgb(255, 150, 150) c.drawText("☠ 死因分布", pad * 0.4f, y, text); y += pad * 0.7f val deaths = arrayOf("撞墙" to deathWall, "撞自己" to deathSelf, "被困" to deathTrap) val maxDeath = max(1, max(deathWall, max(deathSelf, deathTrap))) val dRowH = W * 0.05f for ((name, cnt) in deaths) { text.textSize = W * 0.027f; text.isFakeBoldText = false; text.color = Color.WHITE c.drawText(name, pad * 0.6f, y + dRowH * 0.6f, text) val dx0 = pad * 2.4f; val dW = W * 0.55f panel.color = Color.rgb(60, 50, 60) c.drawRect(dx0, y + dRowH * 0.18f, dx0 + dW, y + dRowH * 0.62f, panel) panel.color = Color.rgb(235, 90, 90) val wf = dW * (cnt.toFloat() / maxDeath) if (wf > 0.5f) c.drawRect(dx0, y + dRowH * 0.18f, dx0 + wf, y + dRowH * 0.62f, panel) text.color = Color.rgb(220, 220, 230) c.drawText("$cnt", dx0 + dW + pad * 0.5f, y + dRowH * 0.6f, text) y += dRowH + pad * 0.15f } y += pad * 0.3f val bottom0 = y panel.color = Color.rgb(20, 30, 34) c.drawRoundRect(pad, bottom0, W - pad, height - pad, pad * 0.6f, pad * 0.6f, panel) text.textSize = W * 0.033f; text.isFakeBoldText = true; text.color = Color.rgb(120, 230, 200) val nowDoing = when { evolving || generationTransitioning -> "🧬 正在繁殖下一代:精英保留 + 交叉 + 变异" completedAgents >= POPULATION_SIZE - 2 -> "⏳ 本代即将完成,准备进化" else -> "🐍 8只小蛇并行训练,吃满CPU搜索+学习" } c.drawText(nowDoing, pad * 1.4f, bottom0 + pad * 1.4f, text) text.textSize = W * 0.026f; text.isFakeBoldText = false; text.color = Color.rgb(170, 200, 210) val why = v3LayerWhy c.drawText(why.take(46), pad * 1.4f, bottom0 + pad * 2.6f, text) if (why.length > 46) c.drawText(why.substring(46), pad * 1.4f, bottom0 + pad * 3.6f, text) text.color = Color.rgb(150, 180, 200) c.drawText("目标: $v3Goal 循环命中$v3LoopHits 教训触发$v3LessonFires 已分配$currentAgentIndex 已完成$completedAgents", pad * 1.4f, bottom0 + pad * 4.8f, text) } private fun drawDebug(c: Canvas) { if (aiMode == 0 && !reinforceTraining) return val w = min(width * 0.97f, 720f); val h = 760f val left = (width - w) / 2f; val top = max(12f, height - h - 12f) drawV3Panel(c, w, top) panel.color = Color.argb(232, 0, 0, 0) c.drawRoundRect(left, top, left + w, top + h, 22f, 22f, panel) border.color = Color.argb(60, 46, 204, 113); border.style = Paint.Style.STROKE; border.strokeWidth = 1.5f c.drawRoundRect(left, top, left + w, top + h, 22f, 22f, border) text.textAlign = Paint.Align.LEFT; text.isFakeBoldText = true; text.textSize = 26f; text.color = Color.WHITE c.drawText(if (reinforceTraining) "混合进化中 · 世代 $generation" else ai.strategy, left + 16f, top + 36f, text) text.isFakeBoldText = false; text.textSize = 14f; text.color = Color.YELLOW c.drawText(if (reinforceTraining) "规则过滤 + Q表 + 神经网络 自我迭代" else ai.reason, left + 16f, top + 58f, text) val gaugeCX = left + 58f; val gaugeCY = top + 116f if (reinforceTraining) { val progRatio = completedAgents.toFloat() / POPULATION_SIZE val arcPaint = Paint(Paint.ANTI_ALIAS_FLAG) arcPaint.style = Paint.Style.STROKE; arcPaint.strokeWidth = 13f; arcPaint.strokeCap = Paint.Cap.ROUND val oval = RectF(gaugeCX - 42f, gaugeCY - 42f, gaugeCX + 42f, gaugeCY + 42f) arcPaint.color = Color.rgb(50, 50, 50); c.drawArc(oval, -90f, 360f, false, arcPaint) arcPaint.color = Color.rgb(46, 204, 113); c.drawArc(oval, -90f, 360f * progRatio, false, arcPaint) text.textAlign = Paint.Align.CENTER; text.isFakeBoldText = true; text.textSize = 22f; text.color = Color.WHITE c.drawText("$generation", gaugeCX, gaugeCY + 8f, text) text.isFakeBoldText = false; text.textSize = 11f; text.color = Color.LTGRAY c.drawText("世代", gaugeCX, gaugeCY + 28f, text) } else { drawDangerGauge(c, gaugeCX, gaugeCY, 42f, ai.danger) text.textAlign = Paint.Align.CENTER; text.isFakeBoldText = true; text.textSize = 26f; text.color = Color.WHITE c.drawText("${ai.danger}", gaugeCX, gaugeCY + 9f, text) text.isFakeBoldText = false; text.textSize = 12f; text.color = Color.LTGRAY c.drawText("危险", gaugeCX, gaugeCY + 28f, text) } text.textAlign = Paint.Align.LEFT; text.textSize = 16f; text.color = Color.WHITE c.drawText("局数 $totalGames", left + 118f, top + 86f, text) val avg = if (recentScores.isEmpty()) 0f else recentScores.average().toFloat() c.drawText("近50均分 ${"%.0f".format(avg)} 最佳 $bestRecentScore", left + 118f, top + 110f, text) val threadInfo = if (reinforceTraining) "${trainThreads.size} 线程" else "单局" c.drawText("模式 $threadInfo 主蛇长 ${snake.size} Q覆盖${visitedStates.size}", left + 118f, top + 134f, text) if (reinforceTraining) { barPaint.style = Paint.Style.FILL; var meterY = top + 155f; val meterW = w - 32f text.textSize = 9f; text.color = Color.rgb(200, 180, 255) c.drawText("Q覆盖 ${visitedStates.size}/$V2_STATE_COUNT", left + 16f, meterY + 8f, text) barPaint.color = Color.rgb(30, 30, 40) c.drawRoundRect(left + 16f, meterY + 12f, left + 16f + meterW, meterY + 19f, 3f, 3f, barPaint) barPaint.color = Color.rgb(200, 150, 255) val qCovR = (visitedStates.size.toFloat() / V2_STATE_COUNT).coerceIn(0f, 1f) c.drawRoundRect(left + 16f, meterY + 12f, left + 16f + meterW * qCovR, meterY + 19f, 3f, 3f, barPaint) meterY += 24f text.color = Color.rgb(100, 220, 255) c.drawText("回放 ${replayBuffer.size}/$REPLAY_CAPACITY", left + 16f, meterY + 8f, text) barPaint.color = Color.rgb(30, 30, 40) c.drawRoundRect(left + 16f, meterY + 12f, left + 16f + meterW, meterY + 19f, 3f, 3f, barPaint) barPaint.color = Color.rgb(100, 200, 255) val rpR = replayBuffer.size.toFloat() / REPLAY_CAPACITY c.drawRoundRect(left + 16f, meterY + 12f, left + 16f + meterW * rpR, meterY + 19f, 3f, 3f, barPaint) meterY += 24f text.color = Color.rgb(241, 196, 15) c.drawText("ε ${"%.2f".format(v2Epsilon)}", left + 16f, meterY + 8f, text) barPaint.color = Color.rgb(30, 30, 40) c.drawRoundRect(left + 16f, meterY + 12f, left + 16f + meterW, meterY + 19f, 3f, 3f, barPaint) barPaint.color = Color.rgb(241, 196, 15) c.drawRoundRect(left + 16f, meterY + 12f, left + 16f + meterW * v2Epsilon, meterY + 19f, 3f, 3f, barPaint) meterY += 24f text.color = Color.rgb(200, 200, 200) c.drawText("Q ${"%.2f".format(qWeight)} / NN ${"%.2f".format(nnWeight)}", left + 16f, meterY + 8f, text) barPaint.color = Color.rgb(30, 30, 40) c.drawRoundRect(left + 16f, meterY + 12f, left + 16f + meterW, meterY + 19f, 3f, 3f, barPaint) barPaint.color = Color.rgb(231, 76, 60) c.drawRoundRect(left + 16f, meterY + 12f, left + 16f + meterW * qWeight, meterY + 19f, 3f, 3f, barPaint) barPaint.color = Color.rgb(46, 204, 113) c.drawRoundRect(left + 16f + meterW * qWeight, meterY + 12f, left + 16f + meterW, meterY + 19f, 3f, 3f, barPaint) } val wbY = top + 250f text.isFakeBoldText = true; text.textSize = 14f; text.color = Color.rgb(255, 200, 100) c.drawText("【权重柱状图】", left + 16f, wbY, text) text.isFakeBoldText = false drawWeightBars(c, left + 16f, wbY + 8f, w - 32f, 66f) val learnY = top + 290f text.isFakeBoldText = true; text.textSize = 14f; text.color = Color.rgb(255, 150, 255) c.drawText("【V2学习】", left + 16f, learnY, text) text.isFakeBoldText = false; text.color = Color.WHITE c.drawText("训练中", left + 118f, learnY, text) val barStartX = left + 118f; val barEndX = left + w - 16f; val barW = barEndX - barStartX var barY = top + 318f if (!reinforceTraining) { text.isFakeBoldText = true; text.textSize = 14f; text.color = Color.rgb(255, 200, 100) c.drawText("【四方向安全 / Q】", left + 16f, barY, text) text.isFakeBoldText = false; barY += 10f val dirNames = listOf("UP", "DN", "LF", "RT") ai.candidates.forEachIndexed { idx, cand -> text.textSize = 18f text.color = if (!cand.legal) Color.GRAY else if (cand.d == ai.chosen) Color.CYAN else Color.WHITE c.drawText(dirNames[idx], left + 16f, barY + 19f, text) if (!cand.legal) { text.textSize = 14f; text.color = Color.GRAY c.drawText("非法", barStartX, barY + 19f, text) } else { val regionRatio = if (snake.isNotEmpty()) cand.region.toFloat() / snake.size else 0f val safeScore = when { cand.tailOk && regionRatio >= 1.5f -> 1.0f cand.tailOk && regionRatio >= 1.0f -> 0.75f regionRatio >= 1.0f -> 0.5f regionRatio >= 0.6f -> 0.3f else -> 0.15f } barPaint.color = Color.rgb(40, 40, 40); barPaint.style = Paint.Style.FILL c.drawRoundRect(barStartX, barY, barEndX, barY + 24f, 4f, 4f, barPaint) val barColor = when { safeScore >= 0.9f -> Color.rgb(46, 204, 113) safeScore >= 0.7f -> Color.rgb(241, 196, 15) safeScore >= 0.5f -> Color.rgb(230, 126, 34) else -> Color.rgb(231, 76, 60) } barPaint.color = barColor c.drawRoundRect(barStartX, barY, barStartX + barW * safeScore, barY + 24f, 4f, 4f, barPaint) text.textSize = 13f; text.color = Color.WHITE c.drawText("空间${cand.region} 尾${if (cand.tailOk) "Y" else "N"} 食${if (cand.foodDist < 0) "∞" else cand.foodDist} Q${"%.2f".format(cand.qValue)}", barStartX + 8f, barY + 18f, text) } barY += 28f } } else barY = top + 318f val chartY = barY + 20f; val chartH = 72f val curveX = left + 16f; val curveW = (w - 32f) * 0.55f text.isFakeBoldText = true; text.textSize = 14f; text.color = Color.rgb(120, 255, 180) c.drawText("【历史分数】", curveX, chartY, text) text.isFakeBoldText = false drawLearningCurve(c, curveX, chartY + 8f, curveW, chartH) val pieCX = left + w - 130f; val pieCY = chartY + chartH / 2f + 14f; val pieR = 34f text.isFakeBoldText = true; text.textSize = 14f; text.color = Color.rgb(255, 150, 150) c.drawText("【死亡统计】", pieCX - 52f, chartY, text) text.isFakeBoldText = false drawDeathPie(c, pieCX, pieCY, pieR) text.textSize = 13f; text.color = Color.rgb(255, 100, 100) c.drawText("W $deathWall", pieCX + 42f, pieCY - 16f, text) text.color = Color.rgb(100, 150, 255); c.drawText("S $deathSelf", pieCX + 42f, pieCY + 4f, text) text.color = Color.rgb(255, 200, 100); c.drawText("T $deathTrap", pieCX + 42f, pieCY + 24f, text) val heatY = chartY + chartH + 30f text.textAlign = Paint.Align.LEFT; text.isFakeBoldText = true; text.textSize = 14f; text.color = Color.rgb(200, 180, 255) c.drawText("【V2 Q表热度】", left + 16f, heatY, text) text.isFakeBoldText = false drawQHeatmap(c, left + 16f, heatY + 8f, w - 32f, 66f) val infoY = heatY + 8f + 66f + 22f var visited = 0; var sumQ = 0f; var cntQ = 0 val sampleStep = max(1, V2_STATE_COUNT / 400) for (s in 0 until V2_STATE_COUNT step sampleStep) for (a in 0 until 4) { val idx = qIndex(s, a); if (nV2[idx] > 0) { visited++; sumQ += qV2[idx]; cntQ++ } } val avgQ = if (cntQ > 0) sumQ / cntQ else 0f text.textSize = 14f; text.color = Color.WHITE; text.textAlign = Paint.Align.LEFT if (reinforceTraining) { c.drawText("世代 $generation 已分配 $currentAgentIndex/$POPULATION_SIZE 已完成 $completedAgents/$POPULATION_SIZE Q覆盖${visitedStates.size}", left + 16f, infoY, text) c.drawText("本代最佳 ${"%.0f".format(bestScoreThisGen)} 历史最佳 ${"%.0f".format(bestScoreAllTime)}", left + 16f, infoY + 18f, text) c.drawText("Q表权重 ${"%.2f".format(qWeight)} 神经网络权重 ${"%.2f".format(nnWeight)}", left + 16f, infoY + 36f, text) c.drawText("回放${replayBuffer.size}/$REPLAY_CAPACITY ε${"%.2f".format(v2Epsilon)} 步${v2LearningSteps}", left + 16f, infoY + 54f, text) val avgBeamNodes = if (beamCalls > 0) beamTotalNodes.toFloat() / beamCalls else 0f c.drawText("Beam 命中$beamSelected 调用$beamCalls 成功$beamSuccess 节点$beamTotalNodes 均$avgBeamNodes", left + 16f, infoY + 72f, text) val learnStatus = when { generation < 3 -> "🧬 初始化种群,随机试错中..." v2Epsilon > 0.15f -> "🔍 探索阶段:尝试新策略" v2Epsilon < 0.05f && nnWeight >= 0.45f -> "🧠 收敛阶段:神经网络主导" bestScoreThisGen > bestScoreAllTime * 0.95f && bestScoreAllTime > 0 -> "🚀 突破中!分数上升" bestScoreThisGen < bestScoreAllTime * 0.3f && bestScoreAllTime > 1000 -> "⚡ 瓶颈期,等待变异突破" else -> "📊 正常进化中..." } text.textSize = 13f; text.isFakeBoldText = true; text.color = Color.rgb(100, 220, 255) c.drawText(learnStatus, left + 16f, infoY + 92f, text) if (deadEndPredicted) { text.color = Color.rgb(255, 80, 80) c.drawText("⚠️ 死局预判!食物周围空间仅 ${"%.1f".format(foodSpaceRatio)}x 蛇长", left + 16f, infoY + 112f, text) } else { text.color = Color.rgb(80, 200, 120) c.drawText("✅ 食物空间 ${"%.1f".format(foodSpaceRatio)}x 蛇长", left + 16f, infoY + 112f, text) } text.color = if (rolloutActive) Color.rgb(180, 180, 255) else Color.rgb(100, 100, 120) c.drawText(if (rolloutActive) "🔮 前瞻模拟${rolloutSteps}步(蛇长${snake.size})" else "🔮 前瞻模拟:蛇短不启用", left + 16f, infoY + 130f, text) text.isFakeBoldText = false; text.color = Color.WHITE; text.textSize = 14f } else { c.drawText("局数 $totalGames 近50均分 ${"%.0f".format(avg)} 最佳 $bestRecentScore", left + 16f, infoY, text) c.drawText("V2覆盖 $visited 平均Q ${"%.2f".format(avgQ)} 学习步 $v2LearningSteps", left + 16f, infoY + 18f, text) c.drawText("ε ${"%.3f".format(v2Epsilon)} 攻击x${"%.2f".format(aggression)} 安全x${"%.2f".format(safetyMargin)}", left + 16f, infoY + 36f, text) } if (gameOver && !reinforceTraining) { text.color = Color.RED; text.isFakeBoldText = true; text.textSize = 16f c.drawText("死亡原因:$deathCause", left + 16f, infoY + 62f, text) text.isFakeBoldText = false } val btnW = 160f; val btnH = 42f val btnLeft = left + w - btnW - 16f; val btnTop = infoY + 48f reinforceButtonRect.set(btnLeft, btnTop, btnLeft + btnW, btnTop + btnH) val btnColor = if (reinforceTraining) Color.rgb(231, 76, 60) else Color.rgb(46, 204, 113) panel.color = btnColor; c.drawRoundRect(reinforceButtonRect, 12f, 12f, panel) text.textAlign = Paint.Align.CENTER; text.isFakeBoldText = true; text.textSize = 18f; text.color = Color.WHITE c.drawText(if (reinforceTraining) "停止进化" else "神经进化", reinforceButtonRect.centerX(), reinforceButtonRect.centerY() + 7f, text) text.isFakeBoldText = false; text.textAlign = Paint.Align.LEFT } private fun drawGameOver(c: Canvas) { paint.color = Color.argb(150, 0, 0, 0) c.drawRect(0f, 0f, width.toFloat(), height.toFloat(), paint) text.textAlign = Paint.Align.CENTER; text.isFakeBoldText = true text.textSize = 30f; text.color = Color.WHITE c.drawText("GAME OVER", width / 2f, height / 2f - 45, text) text.textSize = 15f; text.isFakeBoldText = false c.drawText("分数 $score 最高 $highScore 长度 ${snake.size}", width / 2f, height / 2f - 15, text) text.color = Color.YELLOW c.drawText("点击屏幕重新开始", width / 2f, height / 2f + 20, text) text.color = Color.LTGRAY; text.textSize = 11f c.drawText(lastDeathInfo, width / 2f, height / 2f + 48, text) } private var touchStartX = 0f private var touchStartY = 0f override fun onTouchEvent(e: MotionEvent): Boolean { when (e.action) { MotionEvent.ACTION_DOWN -> { touchStartX = e.x; touchStartY = e.y; return true } MotionEvent.ACTION_UP, MotionEvent.ACTION_CANCEL -> { if (reinforceButtonRect.contains(e.x, e.y)) { val now = System.currentTimeMillis() if (now - lastReinforceTap > 400L) { lastReinforceTap = now; setReinforceTraining(!reinforceTraining) } return true } if (reinforceTraining) return true if (gameOver) { reset(); return true } if (aiMode != 0) return true if (snake.isEmpty()) return true val dx = e.x - touchStartX; val dy = e.y - touchStartY val useSwipe = abs(dx) > 24f || abs(dy) > 24f val d = if (useSwipe) { if (abs(dx) > abs(dy)) P(if (dx > 0) 1 else -1, 0) else P(0, if (dy > 0) 1 else -1) } else { val hx = ox + (snake.first().x + 0.5f) * cell val hy = oy + (snake.first().y + 0.5f) * cell val rdx = e.x - hx; val rdy = e.y - hy if (abs(rdx) > abs(rdy)) P(if (rdx > 0) 1 else -1, 0) else P(0, if (rdy > 0) 1 else -1) } if (!isReverse(d, dir)) { queue.clear(); queue.add(d) } return true } else -> return true } } override fun onDetachedFromWindow() { stopParallelTraining(); stopWatchDog() super.onDetachedFromWindow() bgm?.stop() try { toneGen?.release() } catch (_: Throwable) {} try { aiPool.shutdownNow() } catch (_: Throwable) {} saveTrainingState() } private inner class TrainGame( seed: Long, private val runId: Long, private val statusIndex: Int = 0 ) { val rng = Random(seed) val gSnake = ArrayDeque

() var gDir = P(1, 0) var gFood = P(7, 7) var gScore = 0 var gHunger = 0 var gLastFreeRegion = 0f var gCombo = 0 var gSteps = 0 var gOver = false var lastDeathCause = "UNKNOWN" val vis = BooleanArray(total) val que = IntArray(total + 4) val par = IntArray(total) val parDir = arrayOfNulls

(total) private var currentAgentId = -1 var gLastAction = -1 fun playOneGame(agentId: Int, expectedGenerationToken: Long): Boolean { currentAgentId = agentId if (agentId !in 0 until POPULATION_SIZE) { Log.e(LOG_TAG, "INVALID_AGENT_ID=$agentId"); return false } if (generationToken != expectedGenerationToken) { Log.w(LOG_TAG, "STALE_GAME_START agent=$agentId expected=$expectedGenerationToken actual=$generationToken") return false } gSnake.clear(); gSnake.add(P(7, 7)); gSnake.add(P(6, 7)); gSnake.add(P(5, 7)) gDir = P(1, 0); gScore = 0; gHunger = 0; gCombo = 0; gSteps = 0; gOver = false; gLastAction = -1 lastDeathCause = "UNKNOWN" gLastFreeRegion = gFreeRegion(gSnake).toFloat(); gPlaceFood() val maxIterations = 8000; var iter = 0 try { while (!gOver && isTrainingRunActive(runId) && generationToken == expectedGenerationToken && !Thread.currentThread().isInterrupted && iter < maxIterations) { gStep(); iter++ } } catch (t: Throwable) { Log.e(LOG_TAG, "TrainGame agent=$agentId crashed", t) if (isTrainingRunActive(runId) && generationToken == expectedGenerationToken) { lastDeathCause = "EXCEPTION"; gOver = true try { gDie(lastDeathCause) } catch (inner: Throwable) { Log.e(LOG_TAG, "gDie exception failed", inner) } return true } return false } if (!isTrainingRunActive(runId) || generationToken != expectedGenerationToken || Thread.currentThread().isInterrupted) return false if (!gOver) { lastDeathCause = "TIMEOUT"; gOver = true gScore = max(0, gScore - 500) try { gDie(lastDeathCause) } catch (inner: Throwable) { Log.e(LOG_TAG, "gDie timeout failed", inner) } return true } try { gDie(lastDeathCause) } catch (inner: Throwable) { Log.e(LOG_TAG, "gDie normal failed", inner) } return true } fun gPlaceFood() { val free = ArrayList

(total) for (x in 0 until cols) for (y in 0 until rows) { val p = P(x, y); if (!gSnake.contains(p)) free.add(p) } if (free.isNotEmpty()) gFood = free[rng.nextInt(free.size)] } fun gStep() { heartbeat() val state = gBuildState() val mv = gChooseMove(state) val action = dirs.indexOfFirst { it == mv }.coerceAtLeast(0) gLastAction = action if (!gIsReverse(mv, gDir)) gDir = mv val nh = P(gSnake.first().x + gDir.x, gSnake.first().y + gDir.y) if (!gInside(nh)) { gTerminal(state, action, DEATH_WALL); lastDeathCause = "WALL"; gOver = true; return } val ate = nh == gFood; val tail = gSnake.last() if (gSnake.contains(nh) && !(nh == tail && !ate)) { val testSnake = ArrayDeque(gSnake); testSnake.addFirst(nh); if (!ate) testSnake.removeLast() val reg = gFreeRegion(testSnake) val cause = if (reg < max(2, (gSnake.size * safetyMargin).toInt())) "TRAP" else "SELF" gTerminal(state, action, if (cause == "TRAP") DEATH_TRAP else DEATH_SELF) lastDeathCause = cause; gOver = true; return } gSnake.addFirst(nh); var reward = REWARD_STEP if (ate) { gScore += 10 + min(gCombo, 30) * 3 + gSnake.size; gCombo++; gHunger = 0; gPlaceFood() reward += REWARD_FOOD + gCombo * 0.06f } else { gSnake.removeLast(); gHunger++; gCombo = max(0, gCombo - 1) val gCurFree = gFreeRegion(gSnake).toFloat(); val gSpaceDelta = gCurFree - gLastFreeRegion gLastFreeRegion = gCurFree; reward += gSpaceDelta * REWARD_SPACE_DELTA if (gTailReachable(gSnake)) reward += REWARD_TAIL if (gCalculateDanger() >= 4) reward += REWARD_DANGER reward += -(gHunger * gHunger * 0.001f).coerceAtMost(0.15f) } gSteps++ if (statusIndex in threadStatus.indices) { val st = threadStatus[statusIndex] st.score = gScore; st.steps = gSteps st.phase = if (ate) "捕食" else if (gCalculateDanger() >= 3) "避险" else "搜索" } if (gHunger >= hungerKillLimit) { gTerminal(state, action, DEATH_HUNGER); lastDeathCause = "HUNGER"; gOver = true; return } val nextState = gBuildState(); val nextMask = gLegalMask() qUpdate(state, action, reward, nextState, nextMask, false) } private fun gTerminal(state: Int, action: Int, reward: Float) { qUpdate(state, action, reward, state, 0, true) } private fun beamSearchBestMove(): Pair? { if (!BEAM_ENABLED) return null if (gSnake.isEmpty()) return null if (gSnake.size < BEAM_MIN_LENGTH) return null beamCalls++ try { val depth = when { gSnake.size >= 50 -> BEAM_DEPTH_MAX gSnake.size >= BEAM_LONG_LENGTH -> BEAM_DEPTH_LONG else -> BEAM_DEPTH_SHORT }.coerceAtMost(BEAM_DEPTH_MAX) val width = if (gSnake.size >= BEAM_LONG_LENGTH) 4 else BEAM_WIDTH val initialMoves = dirs .map { d -> gEvaluate(d) } .filter { it.legal } .sortedByDescending { it.score } .take(width) if (initialMoves.isEmpty()) return null var beam = ArrayList() var nodesExpanded = 0 val startHeads = IntArray(4) { -1 } for (candidate in initialMoves) { val node = simulateBeamMove(gSnake, gDir, gFood, candidate.d, candidate.d, startHeads) if (node != null) { beam.add(node) nodesExpanded++ } } if (beam.isEmpty()) return null var depthStep = 0 while (depthStep < depth - 1 && beam.isNotEmpty()) { val next = ArrayList() var limitReached = false for (node in beam) { if (node.dead || limitReached) continue val validMoves = dirs.filter { !gIsReverse(it, node.dir) } val moves = ArrayList() for (mv in validMoves) { val simulated = simulateBeamMove(node.body, node.dir, node.food, mv, node.firstAction, node.recentHeadCells) if (simulated != null) { moves.add(simulated) } } val topMoves = moves.sortedByDescending { it.score }.take(width) next.addAll(topMoves) nodesExpanded += topMoves.size if (nodesExpanded >= BEAM_MAX_NODES) { limitReached = true } } if (next.isEmpty()) { beam = ArrayList() break } beam = ArrayList(next.sortedByDescending { it.score }.take(width)) depthStep++ } if (beam.isEmpty()) return null val best = beam.filter { !it.dead }.maxByOrNull { it.score } ?: beam.maxByOrNull { it.score } ?: return null beamSuccess++ beamTotalNodes += nodesExpanded beamBestScore = best.score return Pair(best.firstAction, best.score) } catch (t: Throwable) { Log.e(LOG_TAG, "Beam search failed", t) return null } } private fun simulateBeamMove( sourceBody: ArrayDeque

, sourceDir: P, targetFood: P, move: P, firstAction: P, prevHeads: IntArray ): BeamNode? { if (sourceBody.isEmpty()) return null if (gIsReverse(move, sourceDir)) return null val body = ArrayDeque

() for (p in sourceBody) body.addLast(p) val head = body.first() val nextHead = P(head.x + move.x, head.y + move.y) if (!gInside(nextHead)) { return BeamNode(body, move, targetFood, -100f, 1, false, true, 0f, firstAction, prevHeads) } val ate = nextHead == targetFood val hitBody = body.contains(nextHead) if (hitBody && !(nextHead == body.last() && !ate)) { return BeamNode(body, move, targetFood, -100f, 1, false, true, 0f, firstAction, prevHeads) } body.addFirst(nextHead) if (!ate) body.removeLast() val region = gFreeRegion(body) val ratio = region.toFloat() / max(1, body.size) val tailOk = gTailReachable(body) val safeMoves = gCountSafeMoves(body) val foodDistance = gDistance(body.first(), targetFood, body, true).let { if (it < 0) 999 else it } val danger = calculateBeamDanger(body, move, targetFood) var score = 0f score += BEAM_SURVIVAL_WEIGHT score += ratio * BEAM_SPACE_WEIGHT if (tailOk) { score += BEAM_TAIL_WEIGHT } else if (body.size > 8) { score -= BEAM_TAIL_WEIGHT * 1.5f } if (ate) { score += 12f * BEAM_FOOD_WEIGHT val afterRegion = gFreeRegion(body) val afterTail = gTailReachable(body) val afterSafe = gCountSafeMoves(body) if (afterRegion < body.size * 0.8f) score -= 15f if (!afterTail && body.size > 12) score -= 12f if (afterSafe <= 1) score -= 8f } else { val foodProgress = 1f / (foodDistance + 1f) score += foodProgress * 8f * BEAM_FOOD_WEIGHT } score += safeMoves * 0.5f score -= danger * BEAM_DANGER_WEIGHT if (ratio < 0.7f && body.size > 8) { score -= 8f } if (!tailOk && body.size > 12) { score -= 7f } val nextCell = nextHead.y * cols + nextHead.x var loopHits = 0 for (h in prevHeads) if (h == nextCell) loopHits++ if (loopHits > 0) { score -= BEAM_LOOP_WEIGHT * loopHits if (loopHits >= 2) score -= BEAM_LOOP_WEIGHT } val newHeads = IntArray(4) { -1 } newHeads[0] = nextCell for (i in 1..3) newHeads[i] = prevHeads[i - 1] return BeamNode(body, move, targetFood, score, 1, ate, false, loopHits.toFloat(), firstAction, newHeads) } private fun calculateBeamDanger(body: ArrayDeque

, heading: P, target: P): Float { if (body.isEmpty()) return 5f val head = body.first() var danger = 0f for (d in dirs) { if (gIsReverse(d, heading)) continue val np = P(head.x + d.x, head.y + d.y) if (!gInside(np)) { danger += 1f; continue } val ate = np == target if (body.contains(np) && !(np == body.last() && !ate)) danger += 1f } val safeMoves = gCountSafeMoves(body) if (safeMoves <= 1) danger += 2f if (safeMoves == 0) danger += 5f return danger.coerceIn(0f, 5f) } fun gChooseMove(state: Int): P { val cands = dirs.map { gEvaluate(it) } val legal = cands.filter { it.legal } if (legal.isEmpty()) return gDir if (legal.size == 1) return legal.first().d val safeFood = gFindSafeFoodStep() val tailStep = if (gHunger < gHungerForceEat()) gFollowTailStep(legal) else null val threshold = gHungerForceEat() val candidatePool = when { gHunger >= threshold -> { val safeFoodCandidates = legal.filter { it.tailOk && it.foodDist >= 0 } if (safeFoodCandidates.isNotEmpty()) safeFoodCandidates else legal } else -> { val shielded = legal.filter { val ratio = it.region.toFloat() / max(1, gSnake.size) it.tailOk || ratio >= 1.0f || (gSnake.size < 15 && ratio >= 0.75f) } if (shielded.isNotEmpty()) shielded else legal } } val boosted = candidatePool.map { c -> var extra = 0f if (safeFood != null && c.d == safeFood) extra += 3.0f if (tailStep != null && c.d == tailStep) extra += 1.5f c.copy(score = c.score + extra) } return gSelectAction(state, boosted).d } private fun gSelectAction(state: Int, legal: List): Candidate { if (legal.size == 1) return legal.first() if (rng.nextFloat() < currentEpsilon()) { val safe = legal.filter { it.tailOk || it.region >= max(5, gSnake.size / 2) } return if (safe.isNotEmpty()) safe[rng.nextInt(safe.size)] else legal[rng.nextInt(legal.size)] } if (currentAgentId !in 0 until POPULATION_SIZE) { Log.e(LOG_TAG, "Invalid currentAgentId=$currentAgentId") return legal.first() } val brain = population[currentAgentId] val nnInputs = buildInputs(gSnake.first(), gFood, gSnake) val nnVals = brain.think(nnInputs) val v3TrainCtx: Long = run { val h0 = gSnake.first() val foodDxBucket = ((gFood.x - h0.x) / 3).coerceIn(-5, 5) val foodDyBucket = ((gFood.y - h0.y) / 3).coerceIn(-5, 5) val regionBucket = (gFreeRegion(gSnake) / 4).coerceIn(0, 60) val lenBucket = (gSnake.size / 8).coerceIn(0, 30) val headingBucket = when (gDir) { P(0, -1) -> 0; P(0, 1) -> 1; P(-1, 0) -> 2; else -> 3 } val dangerBucket = gCalculateDanger().coerceIn(1, 5) val tailBucket = if (gTailReachable(gSnake)) 1 else 0 var hh = 1125899906842597L hh = hh * 31 + h0.x; hh = hh * 31 + h0.y hh = hh * 31 + foodDxBucket; hh = hh * 31 + foodDyBucket hh = hh * 31 + headingBucket; hh = hh * 31 + dangerBucket hh = hh * 31 + regionBucket; hh = hh * 31 + lenBucket; hh = hh * 31 + tailBucket hh xor v3TrainCtxSalt } val de = gCheckFoodDeadEnd() deadEndPredicted = de.first; foodSpaceRatio = de.second var deadEndPenalty = 0f; if (de.first) deadEndPenalty = -5f val rSteps = when { gSnake.size > 60 -> 5; gSnake.size > 25 -> 3; else -> 0 } rolloutActive = rSteps > 0; rolloutSteps = rSteps val beamResult = if (BEAM_ENABLED && gSnake.size >= BEAM_MIN_LENGTH) beamSearchBestMove() else null val beamMove = beamResult?.first val beamValid = beamMove != null && legal.any { it.d == beamMove } var best = legal.first(); var bestValue = -Float.MAX_VALUE for (candidate in legal) { val a = dirs.indexOfFirst { it == candidate.d }; if (a < 0) continue val q = qRead(state, a); val nn = nnVals[a]; val rule = candidate.score val rolloutScore = if (rSteps > 0) gRolloutN(candidate.d, rSteps) else 0f val v3TrainPenalty = v3Memory.penaltyFor(v3TrainCtx, a) * (if (v3FusionEnabled) 45f else 0f) val qNorm = (q / 10f).coerceIn(-1f, 1f) val nnNorm = kotlin.math.tanh(nn * 0.25f).coerceIn(-1f, 1f) val ruleNorm = kotlin.math.tanh(rule / 1000f).coerceIn(-1f, 1f) val rolloutNorm = kotlin.math.tanh(rolloutScore / 30f).coerceIn(-1f, 1f) val memoryNorm = kotlin.math.tanh(v3TrainPenalty / 10f).coerceIn(-1f, 1f) val deadNorm = deadEndPenalty.coerceIn(-1f, 0f) val beamNorm = if (beamValid && candidate.d == beamMove) { kotlin.math.tanh(beamResult!!.second / 20f).coerceIn(-1f, 1f) } else 0f val mixed = qWeight * qNorm + nnWeight * nnNorm + 0.30f * ruleNorm + 0.20f * rolloutNorm + 0.25f * memoryNorm + 0.20f * deadNorm + BEAM_WEIGHT * beamNorm if (mixed > bestValue) { bestValue = mixed; best = candidate.copy(qValue = q) } } if (beamValid && best.d == beamMove) beamSelected++ if (beamCalls > 0L && beamCalls % 100L == 0L && currentAgentId == 0) { Log.d(LOG_TAG, "Beam: calls=$beamCalls success=$beamSuccess selected=$beamSelected nodes=$beamTotalNodes best=$beamBestScore") } return best } private fun gCheckFoodDeadEnd(): Pair { if (gSnake.isEmpty()) return Pair(false, 10f) val gVisited = BooleanArray(total); val queue = ArrayDeque() val start = gFood.y * cols + gFood.x; if (start !in gVisited.indices) return Pair(true, 0f) gVisited[start] = true; queue.addLast(start); var space = 0 while (queue.isNotEmpty()) { val curr = queue.removeFirst(); space++ val x = curr % cols; val y = curr / cols for (d in dirs) { val nx = x + d.x; val ny = y + d.y if (nx !in 0 until cols || ny !in 0 until rows) continue val index = ny * cols + nx; if (gVisited[index]) continue val p = P(nx, ny); if (gSnake.contains(p)) continue gVisited[index] = true; queue.addLast(index) } } val ratio = space.toFloat() / max(1, gSnake.size) return Pair(ratio < 1.2f, ratio) } private fun gRolloutN(startDir: P, maxSteps: Int): Float { var body = ArrayDeque(gSnake); var dir = startDir; var score = 0f for (step in 0 until maxSteps) { if (body.isEmpty()) return score - 50f var bestNext: P? = null; var bestSpace = -1 for (d in dirs) { val nh = P(body.first().x + d.x, body.first().y + d.y) if (!gInside(nh)) continue val isTail = nh == body.last(); if (body.contains(nh) && !isTail) continue var space = 0 for (sd in dirs) { val sx = nh.x + sd.x; val sy = nh.y + sd.y if (sx in 0 until cols && sy in 0 until rows) if (!body.any { it.x == sx && it.y == sy }) space++ } if (space > bestSpace) { bestSpace = space; bestNext = d } } if (bestNext == null) return score - 50f val sim = gSimulateOn(body, bestNext); body = sim.body val region = gFreeRegion(body); val tailOk = gTailReachable(body); val mobility = gCountSafeMoves(body) val spaceRatio = region.toFloat() / max(1, body.size) score += spaceRatio.coerceIn(0f, 6f) * 0.8f score += mobility * 0.5f if (tailOk) score += 1.5f else if (body.size > 10) score -= 3.0f if (sim.ate) score += 8f val fd = gDistance(body.first(), gFood, body, true) if (fd >= 0) score += 3.0f / (fd + 1) score *= 0.92f dir = bestNext } return score } private fun gBuildState(): Int = buildState(gSnake, gDir, gFood, gHunger) private fun gLegalMask(): Int = legalActionMask(gSnake, gDir, gFood) fun gHungerForceEat(): Int = when { gSnake.size < 20 -> 50; gSnake.size < 35 -> 70; gSnake.size < 55 -> 100; gSnake.size < 80 -> 140; else -> 200 } fun gFollowTailStep(legal: List): P? { if (gSnake.size < 2) return null val path = gShortestPath(gSnake.first(), gSnake.last(), gSnake, true) ?: return null if (path.isEmpty()) return null val step = path.first(); if (legal.none { it.d == step }) return null val sim = gSimulate(step); if (sim.body.isEmpty()) return null if (!gTailReachable(sim.body)) return null return step } fun gFindSafeFoodStep(): P? { val path = gShortestPath(gSnake.first(), gFood, gSnake, true) ?: return null if (path.isEmpty()) return null val simBody = ArrayDeque(gSnake); var ate = false for (step in path) { val nh = P(simBody.first().x + step.x, simBody.first().y + step.y) if (!gInside(nh)) return null val willEat = nh == gFood if (simBody.contains(nh) && !(nh == simBody.last() && !willEat)) return null simBody.addFirst(nh); if (!willEat) simBody.removeLast() else ate = true } if (!ate) return null; if (!gTailReachable(simBody)) return null val len = simBody.size; val freeLeft = total - len if (len > 180 && freeLeft < 4) return null if (len > 150 && freeLeft < 6) return null if (len > 120 && freeLeft < 10) return null if (len > 90 && freeLeft < 14) return null if (len > 60 && freeLeft < 20) return null return path.first() } fun gEvaluate(d: P): Candidate { if (!gLegalDirection(d)) return Candidate(d, -1e9f, "非法", false) val sim = gSimulate(d); if (sim.body.isEmpty()) return Candidate(d, -1e9f, "非法", false) val region = gFreeRegion(sim.body); val tail = gTailReachable(sim.body) val foodDist = gDistance(sim.body.first(), gFood, sim.body, true) val ate = sim.ate; val mobility = gCountSafeMoves(sim.body) val hungerFactor = when { gHunger >= 60 -> 4f; gHunger >= 30 -> 2.5f; gHunger >= 15 -> 1.6f; else -> 1f } val regionScore = region * wRegion; val mobilityScore = mobility * wMobility val tailScore = if (tail) wTailGood else wTailBad var foodScore = if (foodDist >= 0) hungerFactor * aggression * (wFoodNear / (foodDist + 1)) else -450f * hungerFactor if (ate) foodScore += wFoodAte * aggression var eatPenalty = 0f if (ate) { val afterSize = sim.body.size; val needSpace = (afterSize * safetyMargin).toInt() + 2 if (!tail || region < needSpace) eatPenalty = -7500f - gSnake.size * 100f else { val after = ArrayDeque(sim.body) val dx = after.last().x - after.first().x; val dy = after.last().y - after.first().y val follow = when { abs(dx) >= abs(dy) && dx != 0 -> P(if (dx > 0) 1 else -1, 0); dy != 0 -> P(0, if (dy > 0) 1 else -1); else -> null } if (follow != null && canSim(after, follow)) { val next = simulateOn(after, follow) val r2 = freeRegion(next.body); val t2 = tailReachable(next.body) if (!t2 || r2 < (afterSize * 1.2f).toInt()) eatPenalty = -4500f - gSnake.size * 60f } } } val edge = min(min(sim.body.first().x, cols - 1 - sim.body.first().x), min(sim.body.first().y, rows - 1 - sim.body.first().y)) val edgeScore = -max(0, 2 - edge) * wEdge val spacePenalty = max(0f, gSnake.size * safetyMargin - region.toFloat()) val spaceScore = -spacePenalty * wSpace val totalScore = regionScore + mobilityScore + tailScore + foodScore + edgeScore + spaceScore + eatPenalty return Candidate(d, totalScore, "", true, regionScore = regionScore, mobilityScore = mobilityScore, tailScore = tailScore, foodScore = foodScore, edgeScore = edgeScore, spaceScore = spaceScore, region = region, mobility = mobility, tailOk = tail, foodDist = foodDist, ate = ate) } fun gSimulate(d: P): Sim = gSimulateOn(ArrayDeque(gSnake), d) fun gSimulateOn(src: ArrayDeque

, d: P): Sim { val b = ArrayDeque(src); if (b.isEmpty()) return Sim(b, false) val nh = P(b.first().x + d.x, b.first().y + d.y) if (!gInside(nh)) return Sim(b, false) val ate = nh == gFood if (b.contains(nh) && !(nh == b.last() && !ate)) return Sim(b, false) b.addFirst(nh); if (!ate) b.removeLast(); return Sim(b, ate) } fun gLegalDirection(d: P): Boolean { if (d == P(0, 0)) return false if (gIsReverse(d, gDir)) return false if (gSnake.isEmpty()) return false val nh = P(gSnake.first().x + d.x, gSnake.first().y + d.y) if (!gInside(nh)) return false val ate = nh == gFood return !gSnake.contains(nh) || (nh == gSnake.last() && !ate) } fun gIsReverse(a: P, b: P): Boolean = a.x == -b.x && a.y == -b.y fun gInside(p: P): Boolean = p.x in 0 until cols && p.y in 0 until rows fun gFreeRegion(body: ArrayDeque

): Int { if (body.isEmpty()) return 0 java.util.Arrays.fill(vis, 0, total, false) for (p in body) { if (!gInside(p)) continue; vis[p.y * cols + p.x] = true } val start = body.first(); if (!gInside(start)) return 0 val si = start.y * cols + start.x; vis[si] = true var head = 0; var tail = 0; que[tail++] = si; var count = 0 while (head < tail) { val curr = que[head++]; val cx = curr % cols; val cy = curr / cols; count++ if (cx > 0) { val ni = curr - 1; if (!vis[ni]) { vis[ni] = true; if (tail < total) que[tail++] = ni } } if (cx < cols - 1) { val ni = curr + 1; if (!vis[ni]) { vis[ni] = true; if (tail < total) que[tail++] = ni } } if (cy > 0) { val ni = curr - cols; if (!vis[ni]) { vis[ni] = true; if (tail < total) que[tail++] = ni } } if (cy < rows - 1) { val ni = curr + cols; if (!vis[ni]) { vis[ni] = true; if (tail < total) que[tail++] = ni } } } return count } fun gDistance(start: P, target: P, body: Collection

, allowTail: Boolean): Int { if (!gInside(start) || !gInside(target)) return -1 if (start == target) return 0 java.util.Arrays.fill(vis, 0, total, false) for (p in body) { if (!gInside(p)) continue; vis[p.y * cols + p.x] = true } if (allowTail && body.isNotEmpty()) { val last = body.last(); if (gInside(last)) vis[last.y * cols + last.x] = false } val si = start.y * cols + start.x; vis[si] = true var head = 0; var tail = 0; que[tail++] = si; var dist = 0 while (head < tail) { val layer = tail - head repeat(layer) { if (head >= tail) return@repeat val curr = que[head++]; val cx = curr % cols; val cy = curr / cols if (cx == target.x && cy == target.y) return dist if (cx > 0) { val ni = curr - 1; if (!vis[ni]) { vis[ni] = true; if (tail < total) que[tail++] = ni } } if (cx < cols - 1) { val ni = curr + 1; if (!vis[ni]) { vis[ni] = true; if (tail < total) que[tail++] = ni } } if (cy > 0) { val ni = curr - cols; if (!vis[ni]) { vis[ni] = true; if (tail < total) que[tail++] = ni } } if (cy < rows - 1) { val ni = curr + cols; if (!vis[ni]) { vis[ni] = true; if (tail < total) que[tail++] = ni } } } dist++ } return -1 } fun gCountSafeMoves(body: ArrayDeque

): Int { if (body.isEmpty()) return 0 val h = body.first() val prevDir = if (body.size < 2) gDir else P(body.elementAt(0).x - body.elementAt(1).x, body.elementAt(0).y - body.elementAt(1).y) return dirs.count { d -> val nh = P(h.x + d.x, h.y + d.y) gInside(nh) && (!body.contains(nh) || (nh == body.last() && nh != gFood)) && !gIsReverse(d, prevDir) } } fun gTailReachable(body: ArrayDeque

): Boolean = if (body.isEmpty()) false else gDistance(body.first(), body.last(), body, true) >= 0 fun gShortestPath(start: P, target: P, body: Collection

, allowTail: Boolean): List

? { if (start == target) return emptyList() if (!gInside(start) || !gInside(target)) return null java.util.Arrays.fill(vis, 0, total, false) for (p in body) { if (!gInside(p)) continue; val idx = p.y * cols + p.x; vis[idx] = true } if (allowTail && body.isNotEmpty()) { val last = body.last(); if (gInside(last)) { val li = last.y * cols + last.x; vis[li] = false } } val si = start.y * cols + start.x; val ti = target.y * cols + target.x; vis[si] = true java.util.Arrays.fill(par, 0, total, -1) for (i in 0 until total) parDir[i] = null var head = 0; var tail = 0; que[tail++] = si; var found = false while (head < tail) { val curr = que[head++]; val cx = curr % cols; val cy = curr / cols if (cy > 0) { val ni = curr - cols; if (!vis[ni]) { vis[ni] = true; par[ni] = curr; parDir[ni] = P(0, -1); if (tail < total) que[tail++] = ni } } if (cy < rows - 1) { val ni = curr + cols; if (!vis[ni]) { vis[ni] = true; par[ni] = curr; parDir[ni] = P(0, 1); if (tail < total) que[tail++] = ni } } if (cx > 0) { val ni = curr - 1; if (!vis[ni]) { vis[ni] = true; par[ni] = curr; parDir[ni] = P(-1, 0); if (tail < total) que[tail++] = ni } } if (cx < cols - 1) { val ni = curr + 1; if (!vis[ni]) { vis[ni] = true; par[ni] = curr; parDir[ni] = P(1, 0); if (tail < total) que[tail++] = ni } } } if (!found) return null val steps = ArrayList

(); var cur = ti while (cur != si) { val d = parDir[cur] ?: return null; steps.add(d); cur = par[cur]; if (cur < 0) return null } steps.reverse(); return steps } fun gCalculateDanger(): Int { val region = gFreeRegion(gSnake); val ratio = region.toFloat() / max(1, gSnake.size) val mobility = gCountSafeMoves(gSnake) return when { mobility <= 0 -> 5; ratio < 1.5f -> 5; ratio < 2.2f -> 4 ratio < 3.5f -> 3; ratio < 5f -> 2; else -> 1 } } fun gDie(cause: String) { var shouldSave = false synchronized(sharedLock) { when (cause) { "WALL" -> deathWall++; "SELF" -> deathSelf++; else -> deathTrap++ } totalGames++; recentScores.addLast(gScore) while (recentScores.size > 50) recentScores.removeFirst() if (gScore > bestRecentScore) bestRecentScore = gScore val index = currentAgentId if (index in 0 until POPULATION_SIZE) { currentScores[index] = gScore.toFloat() if (gScore.toFloat() > bestScoreThisGen) bestScoreThisGen = gScore.toFloat() if (gScore.toFloat() > bestScoreAllTime) bestScoreAllTime = gScore.toFloat() } else { Log.e(LOG_TAG, "gDie invalid agent index=$index") } v2Episodes++; shouldSave = totalGames % 200 == 0 } try { if (v3FusionEnabled && gSnake.isNotEmpty()) v3TrainRecordLesson(cause, gSnake.first(), gTailReachable(gSnake), gSnake.size, gLastAction) } catch (t: Throwable) { Log.e(LOG_TAG, "v3TrainRecordLesson failed", t) } adjustWeights(cause) safetyMargin = (safetyMargin * 0.95f + 1.08f * 0.05f).coerceIn(1.0f, 1.20f) aggression = (aggression * 0.95f + 1.15f * 0.05f).coerceIn(0.95f, 1.35f) if (shouldSave) { try { saveLearning() } catch (_: Throwable) {} } } } private class BgmPlayer { private var audioTrack: AudioTrack? = null @Volatile private var playing = false private var thread: Thread? = null fun start() { if (playing) return playing = true thread = Thread { try { val sr = 22050; val pcm = generateMelody(sr) val track = AudioTrack.Builder() .setAudioAttributes(AudioAttributes.Builder().setUsage(AudioAttributes.USAGE_GAME).setContentType(AudioAttributes.CONTENT_TYPE_MUSIC).build()) .setAudioFormat(AudioFormat.Builder().setEncoding(AudioFormat.ENCODING_PCM_16BIT).setSampleRate(sr).setChannelMask(AudioFormat.CHANNEL_OUT_MONO).build()) .setBufferSizeInBytes(pcm.size * 2).setTransferMode(AudioTrack.MODE_STATIC).build() track.write(pcm, 0, pcm.size) if (Build.VERSION.SDK_INT >= 23) track.setLoopPoints(0, pcm.size, -1) audioTrack = track if (playing) track.play() } catch (_: Throwable) {} }.also { it.start() } } fun stop() { playing = false try { audioTrack?.pause() } catch (_: Throwable) {} try { audioTrack?.flush() } catch (_: Throwable) {} try { audioTrack?.release() } catch (_: Throwable) {} audioTrack = null; thread = null } private fun generateMelody(sampleRate: Int): ShortArray { val N = 0f; val E5 = 659.25f; val G5 = 783.99f; val C6 = 1046.50f val D5 = 587.33f; val F5 = 698.46f; val A5 = 880.00f val C5 = 523.25f; val E4 = 329.63f; val G4 = 392.00f val B4 = 493.88f; val A4 = 440.00f val notes = listOf( E5 to 180, G5 to 180, C6 to 180, G5 to 180, E5 to 180, G5 to 180, C6 to 260, N to 100, D5 to 180, F5 to 180, A5 to 180, F5 to 180, D5 to 180, F5 to 180, A5 to 260, N to 100, E5 to 180, G5 to 180, C6 to 180, G5 to 180, E5 to 180, G5 to 180, C6 to 180, E4 to 180, F5 to 180, A5 to 180, C6 to 180, A5 to 180, G5 to 260, D5 to 260, C5 to 420, N to 260, C5 to 180, E5 to 180, G5 to 180, E5 to 180, A4 to 180, C5 to 180, E5 to 180, C5 to 180, G4 to 180, B4 to 180, D5 to 180, G5 to 180, E5 to 220, D5 to 220, C5 to 420, N to 260, E5 to 180, D5 to 180, C5 to 180, D5 to 180, E5 to 260, G5 to 260, C6 to 400, N to 200 ) val out = ArrayList() for ((freq, durMs) in notes) { val n = durMs * sampleRate / 1000 if (freq == N || freq <= 0f) repeat(n) { out.add(0) } else { val period = (sampleRate / freq).toInt().coerceAtLeast(1) for (i in 0 until n) { val phase = (i % period) / period.toFloat() val t = i.toFloat() / n val env = when { t < 0.05f -> t / 0.05f; t > 0.70f -> (1f - t) / 0.30f; else -> 1f }.coerceIn(0f, 1f) val v = (if (phase < 0.5f) 1f else -1f) * env * 0.06f out.add((v * Short.MAX_VALUE).toInt().coerceIn(Short.MIN_VALUE.toInt(), Short.MAX_VALUE.toInt()).toShort()) } } } return out.toShortArray() } } }