Subversion Repositories SmartDukaan

Rev

Rev 37482 | Show entire file | Ignore whitespace | Details | Blame | Last modification | View Log | RSS feed

Rev 37482 Rev 37565
Line 7... Line 7...
7
import org.springframework.beans.factory.annotation.Autowired;
7
import org.springframework.beans.factory.annotation.Autowired;
8
import com.smartdukaan.cron.monitored.ImeiActivationGauges;
8
import com.smartdukaan.cron.monitored.ImeiActivationGauges;
9
import org.springframework.stereotype.Component;
9
import org.springframework.stereotype.Component;
10
 
10
 
11
import java.util.ArrayList;
11
import java.util.ArrayList;
-
 
12
import java.util.Arrays;
12
import java.util.List;
13
import java.util.List;
13
import java.util.stream.Collectors;
14
import java.util.stream.Collectors;
14
 
15
 
15
@Component
16
@Component
16
public class StandAlone {
17
public class StandAlone {
Line 50... Line 51...
50
	private static final int CHUNK = 25;
51
	private static final int CHUNK = 25;
51
 
52
 
52
	/** Safety stop so a mis-set DAYS cannot pull an unbounded list into memory. */
53
	/** Safety stop so a mis-set DAYS cannot pull an unbounded list into memory. */
53
	private static final int POOL_CAP = 10000;
54
	private static final int POOL_CAP = 10000;
54
 
55
 
-
 
56
	/** The browser lane's rotation, in turn order. */
-
 
57
	private static final List<String> BROWSER_BRANDS = Arrays.asList("Oppo", "Realme");
-
 
58
 
55
	/**
59
	/**
56
	 * Oppo and Realme share ONE thread and one work queue.
60
	 * Round-robin cursor across BROWSER_BRANDS.
57
	 *
61
	 *
-
 
62
	 * Not an AtomicInteger: this lane is a single @Scheduled(fixedDelay) job, so Spring
58
	 * Replaces the separate oppo() and realme() jobs. Nothing else in this class runs a
63
	 * never runs two of its ticks at once and the increment has nothing to race with.
59
	 * browser, so with a single caller there is exactly one ChromeDriver alive at any
64
	 * volatile only for visibility -- consecutive ticks are serialised but land on
-
 
65
	 * whichever scheduled-task-pool thread is free, and a stale read here would break the
60
	 * moment -- down from four. Peak concurrent drivers is what triggers the OOM killer
66
	 * alternation. floorMod keeps it correct when it eventually overflows.
-
 
67
	 */
61
	 * on this box, not average driver-seconds.
68
	private volatile int turn = 0;
-
 
69
 
-
 
70
	/**
-
 
71
	 * Oppo and Realme share ONE thread and one work queue, ONE CHUNK PER TICK.
62
	 *
72
	 *
63
	 * ONE FULL PASS PER DAY, and no imei is asked twice inside a pass. Both pools are
73
	 * Called every 20 seconds. Each call takes the next brand in the rotation, asks the
-
 
74
	 * pool query for its next CHUNK, runs exactly that chunk, and returns. When a brand's
64
	 * snapshotted before any browser starts and the pass walks those lists to the end;
75
	 * pool comes back empty its turn is skipped and the other brand keeps the lane; when
65
	 * it never re-queries. That is the whole fix for the re-ask loop:
76
	 * both are empty the tick does nothing at all and stays silent until midnight.
66
	 *
77
	 *
-
 
78
	 * Nothing else in this class runs a browser, so with a single caller there is exactly
67
	 *   A failed lookup writes no row (dateMap.put is only reached on the success path),
79
	 * one ChromeDriver alive at any moment. Peak concurrent drivers is what triggers the
68
	 *   so createTimestamp is not bumped, so the imei was still eligible on the next tick
80
	 * OOM killer on this box, not average driver-seconds, and a tick model does not change
69
	 *   five minutes later. Measured 29-Aug: realme issued 4,524 requests against just
81
	 * that number -- it only changes how the same driver-seconds are laid out in the day.
-
 
82
	 *
-
 
83
	 * ⚠ THE POOL QUERY IS THE CURSOR. There is no snapshot and no in-memory position. An
70
	 *   1,004 distinct imeis -- 4.5 asks each, 78% of the day's budget spent re-asking --
84
	 * imei leaves the pool because a row was stamped for it, which is why every asked imei
71
	 *   while oppo, which rarely fails, sat at 1.03. More requests hardened the block,
85
	 * must be stamped whatever the outcome -- see restUnanswered on the two services. An
-
 
86
	 * imei that is asked and not stamped is handed straight back on the next tick, twenty
72
	 *   which caused more failures, which caused more requests.
87
	 * seconds later, and the lane stops advancing.
73
	 *
88
	 *
74
	 * A snapshotted pass bounds that structurally: a failure costs one retry TOMORROW,
89
	 * That is also what replaces the snapshot the previous version relied on. The snapshot
-
 
90
	 * existed to stop the re-ask loop measured on 29-Aug -- realme issued 4,524 requests
-
 
91
	 * against 1,004 distinct imeis, 4.5 asks each, 78% of the day's budget spent re-asking,
75
	 * never one in five minutes, no matter how the far end misbehaves.
92
	 * because a failed lookup wrote no row and was eligible again five minutes later.
-
 
93
	 * Stamping closes that at the source instead, and costs nothing the snapshot was buying:
-
 
94
	 * a failure still waits until tomorrow, DAYS=0 still means one ask per imei per day.
76
	 *
95
	 *
77
	 * Sizing, measured on prod 2026-08-31 for a pass starting at midnight: oppo 4,133 and
96
	 * What the snapshot cost, and this does not: a restart threw the day away. The pass
78
	 * realme 2,118 imeis, which at 10.2s and 14.2s each is 11.7h + 8.4h = 20.1 hours of a
-
 
79
	 * single thread -- an 84% duty cycle. It fits, but there is NO slack. If per-imei
97
	 * held one thread for 20+ hours and kept its position only in that thread's stack, so
80
	 * seconds regress the pass will not finish and the tail rolls into the next day.
98
	 * the 12:03 restart on 09-Sep forfeited roughly 4,000 lookups and the whole afternoon,
81
	 * Watch the "Daily pass finished" line: counts short of the "starting" line are the
99
	 * and the funnel had reported last_finish = -1 for three days running. A tick loses at
82
	 * signal, and the lever is DAYS=1 (halves the work) rather than a second thread.
100
	 * most the chunk in flight.
83
	 *
101
	 *
-
 
102
	 * Sizing, measured on prod 2026-08-31: oppo 4,133 and realme 2,118 imeis at 10.2s and
84
	 * Excluding stock billed today or yesterday is correctness, not capacity -- measured,
103
	 * 14.2s each is 20.1 hours of a single thread. Add 20s per chunk and it is 21.4 hours --
85
	 * it trims 30 imeis from oppo and 9 from realme. A handset billed in the last 48
104
	 * it fits, but only at those per-imei costs. Measured again on 09-Sep the lane was at
86
	 * hours has essentially never been activated yet, and the cohort data agrees: the
105
	 * ~21s per imei, twice the sizing, which needs ~35 hours and does not fit. So expect the
-
 
106
	 * pool NOT to clear on a bad day: the lane will still be working at midnight, beginDay
-
 
107
	 * will log the truncation, and the tail rolls over. The tick model makes that visible
87
	 * sold-within-180-days cohort returns a date on 2.4% (vivo) to 8.4% (oppo) of
108
	 * and survivable; it does not create capacity. The lever for capacity is DAYS=1, which
88
	 * lookups, against 43-97% for stock sold over a year ago.
109
	 * halves the daily load by asking each brand every other day.
89
	 *
110
	 *
90
	 * ⚠ Realme's ceiling is a REQUEST-VOLUME ceiling, not a CPU one. realme.com stops
111
	 * ⚠ Realme's ceiling is a REQUEST-VOLUME ceiling, not a CPU one. realme.com stops
91
	 * serving the captcha widget as the day's request count climbs: measured 920/day ->
112
	 * serving the captcha widget as the day's request count climbs: measured 920/day ->
92
	 * 0.3% canvas timeouts, 3,467/day -> 28%, 4,524/day -> 75%, resetting at midnight --
113
	 * 0.3% canvas timeouts, 3,467/day -> 28%, 4,524/day -> 75%, resetting at midnight --
93
	 * while oppo on the same box, same driver count, same widget vendor, at 4,350/day had
114
	 * while oppo on the same box, same driver count, same widget vendor, at 4,350/day had
94
	 * ZERO timeouts across all 24 hours, on a box loaded at 0.9 of 6 cores. Realme's own
115
	 * ZERO timeouts across all 24 hours, on a box loaded at 0.9 of 6 cores. Realme's own
95
	 * canvas wait is 15s against oppo's 8s, so the longer wait is the one expiring. A
116
	 * canvas wait is 15s against oppo's 8s, so the longer wait is the one expiring. Spacing
96
	 * daily pass puts realme near 2,127 requests/day, between the 920/day point where it
-
 
97
	 * was healthy and the 3,467/day point where it was 28% blocked -- so expect SOME
117
	 * the same volume across the day does not move that number -- total daily requests is
98
	 * blocking and judge the change on dates written, not on timeout count. What the
118
	 * what the far end counts -- so judge this change on dates written and on whether the
99
	 * pass guarantees is that blocking can no longer feed itself.
-
 
100
	 *
-
 
101
	 * Oppo and realme alternate chunk by chunk, so whichever pool still has work keeps
-
 
102
	 * the thread busy once the other is exhausted.
119
	 * day clears, not on timeout count.
103
	 */
120
	 */
104
	public void checkBrowserImeiActivation() {
121
	public void checkBrowserImeiActivation() {
105
		List<String> oppo = pendingFor("Oppo");
122
		for (int attempt = 0; attempt < BROWSER_BRANDS.size(); attempt++) {
106
		List<String> realme = pendingFor("Realme");
123
			String brand = BROWSER_BRANDS.get(Math.floorMod(turn++, BROWSER_BRANDS.size()));
107
		gauges.beginPass("Oppo", oppo.size());
124
			List<String> chunk = dueNow(brand);
108
		gauges.beginPass("Realme", realme.size());
125
			if (chunk.isEmpty()) {
109
 
-
 
110
		try {
-
 
111
			if (oppo.isEmpty() && realme.isEmpty()) {
126
				// Cleared for today. Say so once, then let every later tick pass in silence:
112
				LOGGER.info("Oppo and Realme: nothing due today, not starting a browser");
127
				// at 20 seconds a line per idle tick is 4,320 a day, per brand.
113
				return;
128
				gauges.endDay(brand);
114
			}
-
 
115
 
-
 
116
			int oppoDone = 0;
129
				continue;
117
			int realmeDone = 0;
-
 
118
			while (oppoDone < oppo.size() || realmeDone < realme.size()) {
-
 
119
				oppoDone += runChunk("Oppo", oppo, oppoDone, oppoImeiActivationService::updateActivationDate);
-
 
120
				realmeDone += runChunk("Realme", realme, realmeDone, realmeImeiActivationService::updateActivationDate);
-
 
121
			}
130
			}
122
		} finally {
-
 
123
			// In a finally so a pass killed part-way still publishes what it managed.
-
 
124
			// A truncated pass is exactly the case worth alerting on, so it must not be
-
 
125
			// the case that silently reports nothing.
131
			runBrand(brand, chunk, batchFor(brand));
126
			gauges.endPass("Oppo");
132
			gauges.churned(brand, chunk.size());
127
			gauges.endPass("Realme");
133
			return;
128
		}
134
		}
129
	}
135
	}
130
 
136
 
131
	/**
137
	/**
132
	 * One driver session's worth of one brand, taken from a list that was snapshotted
-
 
133
	 * before the pass began. Returns how many were consumed so the caller can advance.
138
	 * This brand's next chunk, and the day's denominator on the first tick after midnight.
134
	 *
139
	 *
135
	 * Chunking exists only to recycle the browser -- a driver held open for the whole
140
	 * The full pool is fetched once a day purely to have something to measure progress
136
	 * pass would leak memory on a box that has been OOM-killed twice, and a crash would
141
	 * against -- POOL_CAP rows instead of CHUNK, one extra query per brand per day, the
137
	 * cost the entire pool rather than CHUNK imeis. It deliberately does NOT re-query:
142
	 * same query the old midnight pass ran. Its head doubles as that tick's chunk, so the
138
	 * re-querying between chunks is what produced the 4.5 lookups per imei per day that
143
	 * sizing costs no extra work. Every later tick asks for CHUNK and nothing more.
139
	 * this rewrite removes.
-
 
140
	 */
144
	 */
141
	private int runChunk(String brand, List<String> pool, int from, ImeiBatch batch) {
145
	private List<String> dueNow(String brand) {
142
		if (from >= pool.size()) {
146
		if (gauges.needsDayStart(brand)) {
-
 
147
			List<String> pool = pendingFor(brand, POOL_CAP);
143
			return 0;
148
			gauges.beginDay(brand, pool.size());
-
 
149
			return pool.size() <= CHUNK ? pool : new ArrayList<>(pool.subList(0, CHUNK));
144
		}
150
		}
145
		List<String> chunk = pool.subList(from, Math.min(from + CHUNK, pool.size()));
-
 
146
		runBrand(brand, chunk, batch);
151
		return pendingFor(brand, CHUNK);
-
 
152
	}
-
 
153
 
147
		gauges.churned(brand, chunk.size());
154
	private ImeiBatch batchFor(String brand) {
148
		return chunk.size();
155
		return "Oppo".equals(brand)
-
 
156
				? oppoImeiActivationService::updateActivationDate
-
 
157
				: realmeImeiActivationService::updateActivationDate;
149
	}
158
	}
150
 
159
 
151
	/**
160
	/**
152
	 * One brand's turn. Wrapped so a failure in the first brand still lets the second
161
	 * One brand's turn. Wrapped so a failure in the first brand still lets the second
153
	 * one run -- these are separate sites and separate driver sessions, and a realme
162
	 * one run -- these are separate sites and separate driver sessions, and a realme
Line 170... Line 179...
170
	private interface ImeiBatch {
179
	private interface ImeiBatch {
171
		void run(List<String> imeis) throws Exception;
180
		void run(List<String> imeis) throws Exception;
172
	}
181
	}
173
 
182
 
174
	/**
183
	/**
175
	 * Everything due for one brand today, as a single list.
184
	 * The next `maxResults` imeis due for one brand, newest query wins -- there is no
-
 
185
	 * cached list anywhere, so this is the lane's only notion of position.
176
	 *
186
	 *
177
	 * The secondary/tertiary split is an artefact of there being two join paths to a
187
	 * The secondary/tertiary split is an artefact of there being two join paths to a
178
	 * serial (transaction.lineitem vs fofo.fofo_line_item), not two kinds of work: both
188
	 * serial (transaction.lineitem vs fofo.fofo_line_item), not two kinds of work: both
179
	 * funnel into the same updateActivationDate -> checkWarranty -> saveActivation path.
189
	 * funnel into the same updateActivationDate -> checkWarranty -> saveActivation path.
180
	 * They were separate jobs with separate batch sizes, which is what made the split
190
	 * They were separate jobs with separate batch sizes, which is what made the split
181
	 * visible at all. A pass walks the whole list to the end, so nothing can be starved
191
	 * visible at all. See interleave for why they are merged by turns rather than joined
182
	 * by ordering and a plain concatenation is enough.
192
	 * end to end.
183
	 *
193
	 *
-
 
194
	 * ⚠ Neither named query has an ORDER BY, so rows arrive in whatever order MySQL
184
	 * The pools are disjoint by construction -- the secondary query excludes anything
195
	 * returns and each tick's chunk is an arbitrary slice of what is still due. That is
-
 
196
	 * survivable but wasteful: hit rate per lookup is 8.4% for oppo stock sold within
185
	 * with a FofoLineItem -- so distinct() is cheap insurance, not a fix for a known
197
	 * 180 days against 97.5% for stock sold over a year ago, so an `order by
-
 
198
	 * o.billingTimestamp asc` in the two named queries would drain the productive cohort
-
 
199
	 * first. It is a dao change and deliberately not made here.
-
 
200
	 *
186
	 * overlap. The brand-generic call is used for every brand; the realme-named copy of
201
	 * The brand-generic call is used for every brand; the realme-named copy of it has
187
	 * it has been deleted, it was a verbatim duplicate down to the named query.
202
	 * been deleted, it was a verbatim duplicate down to the named query.
188
	 */
203
	 */
189
	private List<String> pendingFor(String brand) {
204
	private List<String> pendingFor(String brand, int maxResults) {
190
		List<String> pool = new ArrayList<>();
205
		List<String> secondary = serials(
191
		pool.addAll(activatedImeiRepository.selectImeiActivationPendingByBrand(brand, DAYS, POOL_CAP)
206
				activatedImeiRepository.selectImeiActivationPendingByBrand(brand, DAYS, maxResults));
192
				.stream().map(ImeiActivationTimestampModel::getSerialNumber).collect(Collectors.toList()));
207
		List<String> tertiary = serials(
193
		pool.addAll(activatedImeiRepository.selectImeiActivationPendingByBrandTertiary(brand, DAYS, POOL_CAP)
208
				activatedImeiRepository.selectImeiActivationPendingByBrandTertiary(brand, DAYS, maxResults));
-
 
209
 
-
 
210
		List<String> pool = interleave(secondary, tertiary).stream()
-
 
211
				.distinct()
-
 
212
				.limit(maxResults)
194
				.stream().map(ImeiActivationTimestampModel::getSerialNumber).collect(Collectors.toList()));
213
				.collect(Collectors.toList());
195
		if (pool.size() >= POOL_CAP) {
214
		if (maxResults == POOL_CAP && pool.size() >= POOL_CAP) {
196
			LOGGER.warn("{} pool hit the {} cap -- the pass will not cover everything due today", brand, POOL_CAP);
215
			LOGGER.warn("{} pool hit the {} cap -- today's due count is a floor, not the true total",
-
 
216
					brand, POOL_CAP);
-
 
217
		}
-
 
218
		return pool;
-
 
219
	}
-
 
220
 
-
 
221
	private static List<String> serials(List<ImeiActivationTimestampModel> rows) {
-
 
222
		return rows.stream().map(ImeiActivationTimestampModel::getSerialNumber).collect(Collectors.toList());
-
 
223
	}
-
 
224
 
-
 
225
	/**
-
 
226
	 * Take from both queues by turns rather than concatenating them.
-
 
227
	 *
-
 
228
	 * Plain concatenation was correct while a pass snapshotted everything and walked it to
-
 
229
	 * the end -- ordering could not starve anything that was going to be reached anyway.
-
 
230
	 * A chunk-at-a-time lane has no such guarantee: it stops wherever midnight finds it,
-
 
231
	 * and on the measured per-imei cost it often will not reach the end. Concatenated,
-
 
232
	 * oppo's 163 tertiary serials sit behind 3,819 secondary ones and would be asked only
-
 
233
	 * on a day that fully cleared -- so the smaller queue would go months untouched.
-
 
234
	 *
-
 
235
	 * The two are disjoint by construction (the secondary query excludes anything carrying
-
 
236
	 * a FofoLineItem), so distinct() downstream is insurance rather than a fix.
-
 
237
	 */
-
 
238
	private static List<String> interleave(List<String> first, List<String> second) {
-
 
239
		List<String> merged = new ArrayList<>(first.size() + second.size());
-
 
240
		for (int i = 0; i < Math.max(first.size(), second.size()); i++) {
-
 
241
			if (i < first.size()) {
-
 
242
				merged.add(first.get(i));
-
 
243
			}
-
 
244
			if (i < second.size()) {
-
 
245
				merged.add(second.get(i));
-
 
246
			}
197
		}
247
		}
198
		return pool.stream().distinct().collect(Collectors.toList());
248
		return merged;
199
	}
249
	}
200
 
250
 
201
	/**
251
	/**
202
	 * Motorola: secondary + tertiary in ONE browser session, same shape as the
252
	 * Motorola: secondary + tertiary in ONE browser session, same shape as the
203
	 * oppo/realme combined jobs.
253
	 * oppo/realme combined jobs.