Subversion Repositories SmartDukaan

Rev

Rev 37482 | Blame | Compare with Previous | Last modification | View Log | RSS feed

package com.smartdukaan.cron.scheduled;

import com.spice.profitmandi.dao.model.ImeiActivationTimestampModel;
import com.spice.profitmandi.dao.repository.fofo.ActivatedImeiRepository;
import org.apache.logging.log4j.LogManager;
import org.apache.logging.log4j.Logger;
import org.springframework.beans.factory.annotation.Autowired;
import com.smartdukaan.cron.monitored.ImeiActivationGauges;
import org.springframework.stereotype.Component;

import java.util.ArrayList;
import java.util.Arrays;
import java.util.List;
import java.util.stream.Collectors;

@Component
public class StandAlone {

        @Autowired
        private OppoImeiActivationService oppoImeiActivationService;

        @Autowired
        private RealmeImeiActivationService realmeImeiActivationService;

        @Autowired
        private MotorolaImeiActivationService motorolaImeiActivationService;

        @Autowired
        private ActivatedImeiRepository activatedImeiRepository;

        @Autowired
        private ImeiActivationGauges gauges;

        private static final Logger LOGGER = LogManager.getLogger(StandAlone.class);

        /**
         * ZERO, deliberately, and it is not an off-by-one.
         *
         * The pool query keeps rows with createTimestamp < now().atStartOfDay().minusDays(DAYS).
         * At DAYS=1 an imei answered today is measured against YESTERDAY midnight, so it is
         * not due tomorrow either -- it comes back the day after, a two-day cadence. DAYS=0
         * measures against this morning's midnight, which is what "a full run each day"
         * actually means. Oppo and realme ran at 1 before this change.
         *
         * Within a pass the snapshot already guarantees one ask per imei, so this constant
         * only governs the gap BETWEEN passes.
         */
        private static final int DAYS = 0;

        /** Imeis per browser session. Recycles the driver; does NOT re-query the pool. */
        private static final int CHUNK = 25;

        /** Safety stop so a mis-set DAYS cannot pull an unbounded list into memory. */
        private static final int POOL_CAP = 10000;

        /** The browser lane's rotation, in turn order. */
        private static final List<String> BROWSER_BRANDS = Arrays.asList("Oppo", "Realme");

        /**
         * Round-robin cursor across BROWSER_BRANDS.
         *
         * Not an AtomicInteger: this lane is a single @Scheduled(fixedDelay) job, so Spring
         * never runs two of its ticks at once and the increment has nothing to race with.
         * volatile only for visibility -- consecutive ticks are serialised but land on
         * whichever scheduled-task-pool thread is free, and a stale read here would break the
         * alternation. floorMod keeps it correct when it eventually overflows.
         */
        private volatile int turn = 0;

        /**
         * Oppo and Realme share ONE thread and one work queue, ONE CHUNK PER TICK.
         *
         * Called every 20 seconds. Each call takes the next brand in the rotation, asks the
         * pool query for its next CHUNK, runs exactly that chunk, and returns. When a brand's
         * pool comes back empty its turn is skipped and the other brand keeps the lane; when
         * both are empty the tick does nothing at all and stays silent until midnight.
         *
         * Nothing else in this class runs a browser, so with a single caller there is exactly
         * one ChromeDriver alive at any moment. Peak concurrent drivers is what triggers the
         * OOM killer on this box, not average driver-seconds, and a tick model does not change
         * that number -- it only changes how the same driver-seconds are laid out in the day.
         *
         * ⚠ THE POOL QUERY IS THE CURSOR. There is no snapshot and no in-memory position. An
         * imei leaves the pool because a row was stamped for it, which is why every asked imei
         * must be stamped whatever the outcome -- see restUnanswered on the two services. An
         * imei that is asked and not stamped is handed straight back on the next tick, twenty
         * seconds later, and the lane stops advancing.
         *
         * That is also what replaces the snapshot the previous version relied on. The snapshot
         * existed to stop the re-ask loop measured on 29-Aug -- realme issued 4,524 requests
         * against 1,004 distinct imeis, 4.5 asks each, 78% of the day's budget spent re-asking,
         * because a failed lookup wrote no row and was eligible again five minutes later.
         * Stamping closes that at the source instead, and costs nothing the snapshot was buying:
         * a failure still waits until tomorrow, DAYS=0 still means one ask per imei per day.
         *
         * What the snapshot cost, and this does not: a restart threw the day away. The pass
         * held one thread for 20+ hours and kept its position only in that thread's stack, so
         * the 12:03 restart on 09-Sep forfeited roughly 4,000 lookups and the whole afternoon,
         * and the funnel had reported last_finish = -1 for three days running. A tick loses at
         * most the chunk in flight.
         *
         * Sizing, measured on prod 2026-08-31: oppo 4,133 and realme 2,118 imeis at 10.2s and
         * 14.2s each is 20.1 hours of a single thread. Add 20s per chunk and it is 21.4 hours --
         * it fits, but only at those per-imei costs. Measured again on 09-Sep the lane was at
         * ~21s per imei, twice the sizing, which needs ~35 hours and does not fit. So expect the
         * pool NOT to clear on a bad day: the lane will still be working at midnight, beginDay
         * will log the truncation, and the tail rolls over. The tick model makes that visible
         * and survivable; it does not create capacity. The lever for capacity is DAYS=1, which
         * halves the daily load by asking each brand every other day.
         *
         * ⚠ Realme's ceiling is a REQUEST-VOLUME ceiling, not a CPU one. realme.com stops
         * serving the captcha widget as the day's request count climbs: measured 920/day ->
         * 0.3% canvas timeouts, 3,467/day -> 28%, 4,524/day -> 75%, resetting at midnight --
         * while oppo on the same box, same driver count, same widget vendor, at 4,350/day had
         * ZERO timeouts across all 24 hours, on a box loaded at 0.9 of 6 cores. Realme's own
         * canvas wait is 15s against oppo's 8s, so the longer wait is the one expiring. Spacing
         * the same volume across the day does not move that number -- total daily requests is
         * what the far end counts -- so judge this change on dates written and on whether the
         * day clears, not on timeout count.
         */
        public void checkBrowserImeiActivation() {
                for (int attempt = 0; attempt < BROWSER_BRANDS.size(); attempt++) {
                        String brand = BROWSER_BRANDS.get(Math.floorMod(turn++, BROWSER_BRANDS.size()));
                        List<String> chunk = dueNow(brand);
                        if (chunk.isEmpty()) {
                                // Cleared for today. Say so once, then let every later tick pass in silence:
                                // at 20 seconds a line per idle tick is 4,320 a day, per brand.
                                gauges.endDay(brand);
                                continue;
                        }
                        runBrand(brand, chunk, batchFor(brand));
                        gauges.churned(brand, chunk.size());
                        return;
                }
        }

        /**
         * This brand's next chunk, and the day's denominator on the first tick after midnight.
         *
         * The full pool is fetched once a day purely to have something to measure progress
         * against -- POOL_CAP rows instead of CHUNK, one extra query per brand per day, the
         * same query the old midnight pass ran. Its head doubles as that tick's chunk, so the
         * sizing costs no extra work. Every later tick asks for CHUNK and nothing more.
         */
        private List<String> dueNow(String brand) {
                if (gauges.needsDayStart(brand)) {
                        List<String> pool = pendingFor(brand, POOL_CAP);
                        gauges.beginDay(brand, pool.size());
                        return pool.size() <= CHUNK ? pool : new ArrayList<>(pool.subList(0, CHUNK));
                }
                return pendingFor(brand, CHUNK);
        }

        private ImeiBatch batchFor(String brand) {
                return "Oppo".equals(brand)
                                ? oppoImeiActivationService::updateActivationDate
                                : realmeImeiActivationService::updateActivationDate;
        }

        /**
         * One brand's turn. Wrapped so a failure in the first brand still lets the second
         * one run -- these are separate sites and separate driver sessions, and a realme
         * outage must not cost oppo its whole tick.
         */
        private void runBrand(String brand, List<String> imeis, ImeiBatch batch) {
                if (imeis.isEmpty()) {
                        return;
                }
                LOGGER.info("{} imeis {}", brand, imeis);
                try {
                        batch.run(imeis);
                } catch (Exception e) {
                        gauges.error(brand);
                        LOGGER.error("{} activation batch failed, continuing with the next brand", brand, e);
                }
        }

        @FunctionalInterface
        private interface ImeiBatch {
                void run(List<String> imeis) throws Exception;
        }

        /**
         * The next `maxResults` imeis due for one brand, newest query wins -- there is no
         * cached list anywhere, so this is the lane's only notion of position.
         *
         * The secondary/tertiary split is an artefact of there being two join paths to a
         * serial (transaction.lineitem vs fofo.fofo_line_item), not two kinds of work: both
         * funnel into the same updateActivationDate -> checkWarranty -> saveActivation path.
         * They were separate jobs with separate batch sizes, which is what made the split
         * visible at all. See interleave for why they are merged by turns rather than joined
         * end to end.
         *
         * ⚠ Neither named query has an ORDER BY, so rows arrive in whatever order MySQL
         * returns and each tick's chunk is an arbitrary slice of what is still due. That is
         * survivable but wasteful: hit rate per lookup is 8.4% for oppo stock sold within
         * 180 days against 97.5% for stock sold over a year ago, so an `order by
         * o.billingTimestamp asc` in the two named queries would drain the productive cohort
         * first. It is a dao change and deliberately not made here.
         *
         * The brand-generic call is used for every brand; the realme-named copy of it has
         * been deleted, it was a verbatim duplicate down to the named query.
         */
        private List<String> pendingFor(String brand, int maxResults) {
                List<String> secondary = serials(
                                activatedImeiRepository.selectImeiActivationPendingByBrand(brand, DAYS, maxResults));
                List<String> tertiary = serials(
                                activatedImeiRepository.selectImeiActivationPendingByBrandTertiary(brand, DAYS, maxResults));

                List<String> pool = interleave(secondary, tertiary).stream()
                                .distinct()
                                .limit(maxResults)
                                .collect(Collectors.toList());
                if (maxResults == POOL_CAP && pool.size() >= POOL_CAP) {
                        LOGGER.warn("{} pool hit the {} cap -- today's due count is a floor, not the true total",
                                        brand, POOL_CAP);
                }
                return pool;
        }

        private static List<String> serials(List<ImeiActivationTimestampModel> rows) {
                return rows.stream().map(ImeiActivationTimestampModel::getSerialNumber).collect(Collectors.toList());
        }

        /**
         * Take from both queues by turns rather than concatenating them.
         *
         * Plain concatenation was correct while a pass snapshotted everything and walked it to
         * the end -- ordering could not starve anything that was going to be reached anyway.
         * A chunk-at-a-time lane has no such guarantee: it stops wherever midnight finds it,
         * and on the measured per-imei cost it often will not reach the end. Concatenated,
         * oppo's 163 tertiary serials sit behind 3,819 secondary ones and would be asked only
         * on a day that fully cleared -- so the smaller queue would go months untouched.
         *
         * The two are disjoint by construction (the secondary query excludes anything carrying
         * a FofoLineItem), so distinct() downstream is insurance rather than a fix.
         */
        private static List<String> interleave(List<String> first, List<String> second) {
                List<String> merged = new ArrayList<>(first.size() + second.size());
                for (int i = 0; i < Math.max(first.size(), second.size()); i++) {
                        if (i < first.size()) {
                                merged.add(first.get(i));
                        }
                        if (i < second.size()) {
                                merged.add(second.get(i));
                        }
                }
                return merged;
        }

        /**
         * Motorola: secondary + tertiary in ONE browser session, same shape as the
         * oppo/realme combined jobs.
         *
         * Cadence: the pool query defers an imei for `days` after each attempt
         * (saveActivation bumps createTimestamp even when no date came back), so
         * days=2 gives the requested "retry everything every two days".
         *
         * Sizing: the pending pool measured 1,534 (1,182 secondary + 352 tertiary).
         * At the ~10-14s/imei the oppo and realme jobs measure, 60 per invocation is
         * roughly 12 minutes of driver time, and clearing 1,534 inside 48h needs
         * about 26 invocations -- i.e. an OS cron entry every 90 minutes, with
         * headroom. Do NOT schedule it inside the oppo/realme window: each driver
         * tree costs ~850MB and this box has been OOM-killed twice with tomcat the
         * victim, so peak concurrent drivers is the number that matters.
         */
        public void checkMotorolaImeiStatusCombined() throws Exception {
                List<String> secondary = activatedImeiRepository.selectImeiActivationPendingByBrand("Motorola", 2, 30)
                                .stream().map(ImeiActivationTimestampModel::getSerialNumber).collect(Collectors.toList());
                List<String> tertiary = activatedImeiRepository.selectImeiActivationPendingByBrandTertiary("Motorola", 2, 30)
                                .stream().map(ImeiActivationTimestampModel::getSerialNumber).collect(Collectors.toList());
                LOGGER.info("Motorola secondary imeis {}", secondary);
                LOGGER.info("Motorola tertiary imeis {}", tertiary);
                List<String> all = new ArrayList<>(secondary);
                all.addAll(tertiary);
                all = all.stream().distinct().collect(Collectors.toList());
                if (all.isEmpty()) {
                        LOGGER.info("Motorola: nothing pending, not starting a browser");
                        return;
                }
                motorolaImeiActivationService.updateActivationDate(all);
        }

}