fix root search load data and infinity load data also in graph + add search in graph to flask

This commit is contained in:
Jieyab89
2026-08-08 08:37:29 +07:00
parent 30c25fa1bf
commit 28316a5e8e
4 changed files with 121 additions and 12 deletions
+4
View File
@@ -267,6 +267,10 @@ Sentiment Analysis
<img width="2553" height="1225" alt="Image" src="https://github.com/user-attachments/assets/0b10340a-1503-434a-9015-1719e169a664" />
Follower and Following
<img width="2552" height="1226" alt="image" src="https://github.com/user-attachments/assets/d61d4b5d-f4eb-4b98-af59-7a77bf896f86" />
# Help
About SnowflakeID -> Twitter userid : https://en.wikipedia.org/wiki/Snowflake_ID
+21 -7
View File
@@ -284,7 +284,7 @@ def _filter_by_date(items: list, from_date: str, to_date: str) -> list:
def _multi_source_search(
query: str, count: int, from_date: str = "", to_date: str = "", cursor: str | None = None,
) -> tuple[list, str | None]:
) -> tuple[list, str | None, dict]:
"""Fans out across every source in parallel. cursor (if given) is an
opaque JSON object of {source: source_cursor} built from a previous
call's returned cursor — each key present in it is a source that still
@@ -293,7 +293,15 @@ def _multi_source_search(
(cursor=None); every load-more page after that is cookie/wayback/cse only.
A cursor value that doesn't parse as a JSON object is treated as "no
cursor" (first page) rather than raising — same tolerant-of-garbage-input
posture as the rest of this file's client-supplied-field handling."""
posture as the rest of this file's client-supplied-field handling.
Third return value is {source: error_message} for any source that failed
this round (missing creds, network blip, quota hit, ...) — a source
failing shouldn't sink the others, but silently dropping it also leaves
the caller unable to tell "this source ran dry" apart from "this source
is broken right now," which matters most on a load-more page where the
UI would otherwise just look like that source stopped contributing for
no reason."""
twitter_query = _apply_date_operators(query, from_date, to_date)
try:
@@ -322,6 +330,7 @@ def _multi_source_search(
results = []
next_cursor_parts = {}
source_errors = {}
with ThreadPoolExecutor(max_workers=len(jobs)) as pool:
futures = {key: pool.submit(fn) for key, fn in jobs.items()}
for key in ("cookie", "xquik", "wayback", "cse"): # deterministic display order
@@ -329,14 +338,15 @@ def _multi_source_search(
continue
try:
data, next_c = futures[key].result()
except Exception:
except Exception as e:
source_errors[key] = str(e)
continue # a source failing (missing creds, network, ...) shouldn't sink the others
results.extend(_tag_source(data, SOURCE_LABELS[key]))
if next_c:
next_cursor_parts[key] = next_c
next_cursor = json.dumps(next_cursor_parts) if next_cursor_parts else None
return _filter_by_date(results, from_date, to_date), next_cursor
return _filter_by_date(results, from_date, to_date), next_cursor, source_errors
# Whitelist: only proxy Twitter's video CDN to prevent SSRF
@@ -401,7 +411,8 @@ def run_tool():
"error": "Server is busy — max concurrent requests reached. Please try again shortly.",
}), 429
next_cursor = None # stays None for tools/modes that don't paginate
next_cursor = None # stays None for tools/modes that don't paginate
source_errors = None # multi_source_search only — {source: error} for lanes that failed this page
try:
if tool_type == "tweet_search_extractor":
@@ -479,7 +490,7 @@ def run_tool():
for label, val in (("dateFrom", from_date), ("dateTo", to_date)):
if val and not _valid_date8(val):
return jsonify({"ok": False, "error": f"{label} must be an 8-digit date (YYYYMMDD)"}), 400
data, next_cursor = _multi_source_search(query, count=count, from_date=from_date, to_date=to_date, cursor=cursor)
data, next_cursor, source_errors = _multi_source_search(query, count=count, from_date=from_date, to_date=to_date, cursor=cursor)
else:
return jsonify({"ok": False, "error": f"Unknown toolType: {tool_type}"}), 400
@@ -492,7 +503,10 @@ def run_tool():
data = _stamp_fetched_at(data)
data = _stamp_tweet_url(data)
return jsonify({"ok": True, "data": data, "nextCursor": next_cursor})
resp = {"ok": True, "data": data, "nextCursor": next_cursor}
if source_errors:
resp["sourceErrors"] = source_errors
return jsonify(resp)
except (XquikError, CookieClientError, WaybackError, GoogleCSEError) as e:
return jsonify({"ok": False, "error": str(e)}), 400
+69 -4
View File
@@ -570,6 +570,7 @@
<input id="dateFromInput" type="text" placeholder="From YYYYMMDD" title="Optional start date" maxlength="8" style="width:100px;display:none">
<input id="dateToInput" type="text" placeholder="To YYYYMMDD" title="Optional end date" maxlength="8" style="width:100px;display:none">
<input id="countInput" type="number" value="20" min="1" max="2000" title="Result count">
<input id="graphFilterInput" type="text" placeholder="Filter graph…" spellcheck="false" autocomplete="off" style="display:none;width:160px" title="Show only nodes whose data matches this text">
<span id="nodeCount">0 nodes</span>
<button id="btnSearch" class="btn btn-primary">Search</button>
<button id="btnArchiveAll" class="btn btn-ghost">Archive All</button>
@@ -908,6 +909,9 @@ function init() {
expandNode('following_explorer');
});
document.getElementById('btnLoadMoreSearch').addEventListener('click', expandSearch);
document.getElementById('graphFilterInput').addEventListener('input', function (e) {
filterGraph(e.target.value);
});
updatePlaceholder();
updateModeVisibility();
@@ -959,13 +963,59 @@ function setStatus(msg, isError, persist) {
}
}
// ── Graph filter ─────────────────────────────────────────────────────────────
// Same idea as the "Search results…" box on the Search page, applied to the
// canvas instead of a card list: typing narrows the graph down to only the
// nodes whose data actually matches, so a big graph doesn't have to be
// visually scanned node-by-node to find one account or keyword.
var graphFilterQuery = '';
function flatText(obj) {
if (obj == null) return '';
if (typeof obj !== 'object') return String(obj);
return Object.values(obj).map(flatText).join(' ');
}
// Root search nodes are graph structure, not a result themselves — always
// kept visible so a filter narrowing the results underneath one doesn't also
// disconnect that whole subtree from view.
function applyGraphFilter() {
var q = graphFilterQuery.toLowerCase().trim();
cy.batch(function () {
if (!q) {
cy.elements().show();
return;
}
cy.nodes().forEach(function (node) {
var data = node.data();
var matches = data.type === 'search'
|| (data.label || '').toLowerCase().includes(q)
|| flatText(data.raw).toLowerCase().includes(q);
if (matches) node.show(); else node.hide();
});
cy.edges().forEach(function (edge) {
if (edge.source().visible() && edge.target().visible()) edge.show(); else edge.hide();
});
});
}
function filterGraph(query) {
graphFilterQuery = query;
updateNodeCount();
}
function updateNodeCount() {
applyGraphFilter(); // re-applied here too, so newly added nodes honor an already-active filter
var n = cy.nodes().length;
var visN = cy.nodes(':visible').length;
var selN = cy.nodes(':selected').length;
var label = n + ' node' + (n !== 1 ? 's' : '');
var label = (graphFilterQuery.trim() && visN !== n)
? (visN + ' of ' + n + ' node' + (n !== 1 ? 's' : ''))
: (n + ' node' + (n !== 1 ? 's' : ''));
if (selN > 1) label += ' · ' + selN + ' selected';
document.getElementById('nodeCount').textContent = label;
document.getElementById('emptyHint').classList.toggle('hidden', n > 0);
document.getElementById('graphFilterInput').style.display = n > 0 ? '' : 'none';
}
// ── API helpers ───────────────────────────────────────────────────────────────
@@ -1012,7 +1062,20 @@ async function apiFetch(body) {
if (res.status === 429 && json.retryAfter) err.retryAfter = json.retryAfter;
throw err;
}
return { items: json.data, nextCursor: json.nextCursor || null };
return { items: json.data, nextCursor: json.nextCursor || null, sourceErrors: json.sourceErrors || null };
}
var SOURCE_LABEL_MAP = { cookie: 'Twitter Cookie', xquik: 'Xquik API', wayback: 'Wayback Machine', cse: 'Google CSE' };
// multi_source_search only — a lane running dry (no more matches) and a lane
// failing (network blip, quota hit) both just look like "fewer new nodes"
// otherwise, so this spells out which lane and why rather than leaving it silent.
function sourceErrorNote(sourceErrors) {
if (!sourceErrors || !Object.keys(sourceErrors).length) return '';
var parts = Object.keys(sourceErrors).map(function (key) {
return (SOURCE_LABEL_MAP[key] || key) + ': ' + sourceErrors[key];
});
return ' — ' + parts.join(' · ');
}
// ── Cookie/Wayback cooldown (mirrors the server's 5s-per-source throttle) ──
@@ -1212,7 +1275,7 @@ async function runSearch() {
var added = addNodes(items, function (item) { return resolveNodeType(tool, item); }, searchId);
runLayout(true, true);
updateNodeCount();
setStatus('Done — ' + items.length + ' result(s) (' + added + ' new nodes)');
setStatus('Done — ' + items.length + ' result(s) (' + added + ' new nodes)' + sourceErrorNote(result.sourceErrors), false, !!result.sourceErrors);
} catch (e) {
setStatus('Error: ' + e.message, true);
} finally {
@@ -1384,7 +1447,7 @@ async function expandSearch() {
var added = addNodes(items, function (item) { return resolveNodeType(data.tool, item); }, selectedNode.id());
runLayout(false, false);
updateNodeCount();
setStatus('Loaded more — ' + items.length + ' result(s) (' + added + ' new nodes)');
setStatus('Loaded more — ' + items.length + ' result(s) (' + added + ' new nodes)' + sourceErrorNote(result.sourceErrors), false, !!result.sourceErrors);
selectedNode.data('searchCursor', result.nextCursor || null);
selectedNode.data('searchExhausted', !result.nextCursor);
@@ -1750,6 +1813,8 @@ function clearGraph() {
cy.elements().remove();
allItems.length = 0;
hidePanel();
graphFilterQuery = '';
document.getElementById('graphFilterInput').value = '';
updateNodeCount();
graphArchivedId = null;
document.getElementById('btnArchiveAll').textContent = 'Archive All';
+27 -1
View File
@@ -895,6 +895,7 @@ const PAGINATED_TOOLS = new Set([
'wayback_archive_search', 'multi_source_search',
]);
const THROTTLE_SECONDS = 5;
const SOURCE_LABEL_MAP = { cookie: 'Twitter Cookie', xquik: 'Xquik API', wayback: 'Wayback Machine', cse: 'Google CSE' };
let nextCursor = null;
let loadingMore = false;
@@ -1575,7 +1576,17 @@ async function doLoadMore(sources) {
currentData = (Array.isArray(currentData) ? currentData : [currentData]).concat(newItems);
appendCards(newItems, searchInput.value);
nextCursor = json.nextCursor || null;
loadMoreStatus.classList.remove('visible');
// multi_source_search: a source can legitimately run out (Wayback CDX
// has a finite match list, Google CSE is hard-capped well under the
// daily quota) or fail transiently (network blip, quota hit) on any
// given page — without this, that source just silently stops
// contributing and it looks like a bug rather than an explainable gap.
if (json.sourceErrors && Object.keys(json.sourceErrors).length) {
showSourceErrors(json.sourceErrors);
} else {
loadMoreStatus.classList.remove('visible');
}
} catch (e) {
loadMoreStatus.textContent = `Load more failed: ${e}`;
} finally {
@@ -1584,6 +1595,21 @@ async function doLoadMore(sources) {
}
}
// Shows which multi_source_search lanes didn't contribute this page and
// why, for a few seconds, then clears — a source running dry (no more
// matches) and a source failing (network/quota) look identical from the
// item count alone, so both get spelled out rather than left silent.
function showSourceErrors(errors) {
const lines = Object.entries(errors)
.map(([key, msg]) => `${esc(SOURCE_LABEL_MAP[key] || key)}: ${esc(msg)}`);
loadMoreStatus.classList.add('visible');
loadMoreStatus.innerHTML = `<span class="load-more-note">${lines.join(' · ')}</span>`;
clearTimeout(showSourceErrors._t);
showSourceErrors._t = setTimeout(() => {
loadMoreStatus.classList.remove('visible');
}, 6000);
}
// root: null (the browser viewport) — .output has `overflow: auto` but is
// never actually height-constrained (body/.layout only set min-height), so
// it never becomes a real scroll container; the page/viewport is what