{
  "schema_version": "reality.hiring-project-analysis/v1",
  "subject": "Xingyu (Alex) Shen",
  "analysis_date": "2026-08-31",
  "purpose": "Human screening support for an illustrative product-minded AI systems engineer rubric.",
  "decision_readiness": {
    "screening_decision": "READY_FOR_HUMAN_DECISION",
    "offer_or_rejection_decision": "NOT_READY",
    "reason": "The packet contains inspectable work evidence and role-relevant interview targets, but does not independently verify contribution scope, employment, degree conferral or job performance."
  },
  "rubric": [
    "systems and orchestration depth",
    "product delivery and usable artifacts",
    "applied ML or data-pipeline depth",
    "research communication and evaluation discipline"
  ],
  "method": {
    "tinyfish": "The recorded consented run selected and fetched the candidate GitHub, portfolio, Clade, arXiv, Duke, DKU and Scholar sources.",
    "reality": "Reality verified URL/hash provenance, reviewed candidate attachment, known claim citations and explicit limits for 7/7 released sources.",
    "analysis": "Human-reviewed, source-bound synthesis. Evidence-depth labels describe the visible artifact, not the person and not an automatic hiring score.",
    "minimax": "The recorded MiniMax-M3 draft organized source-bound claims behind schema and human release gates; this project-depth appendix was human-reviewed against the linked primary sources."
  },
  "projects": [
    {
      "project": "Clade",
      "links": {
        "repository": "https://github.com/shenxingy/Clade",
        "portfolio": "https://alexshen.dev/en/projects"
      },
      "what_it_does": "Provider-neutral delivery control plane for coding agents with native Claude Code and Codex surfaces, an MCP bridge, evidence and verifier controls, correction learning, delivery state and an optional orchestrator.",
      "observable_evidence": [
        "Public repository under the exact candidate-supplied GitHub account.",
        "Repository exposes runtime adapters, plugins, MCP, orchestrator, tests, documentation, security and release material.",
        "The recorded TinyFish Fetch bound 13,202 characters and 274 links from this repository."
      ],
      "evidence_depth": "SUBSTANTIAL_OBSERVABLE_SYSTEM",
      "role_relevance": "HIGH for AI infrastructure, developer tooling and product-minded systems work.",
      "not_proven": "Sole authorship, production adoption, exact personal contribution, code quality or business impact.",
      "interview_question": "Walk through one end-to-end trust or delivery path, identify the hardest failure mode, and show the exact code and test you personally owned."
    },
    {
      "project": "GPT4o-Receipt",
      "links": {
        "paper": "https://arxiv.org/abs/2603.11442",
        "dataset": "https://huggingface.co/datasets/Scam-AI/gpt4o-receipt"
      },
      "what_it_does": "Public benchmark and human study for AI-generated receipt forensics comparing authentic and generated receipts across multimodal models and human annotators.",
      "observable_evidence": [
        "arXiv lists Alex Shen as one of nine authors.",
        "The paper reports 1,235 receipt images, five evaluated multimodal models and a 30-annotator study.",
        "The paper is 12 pages with seven figures and seven tables and links a public evaluation framework and results."
      ],
      "evidence_depth": "SUBSTANTIAL_RESEARCH_ARTIFACT",
      "role_relevance": "HIGH for document forensics, evaluation design and applied multimodal ML.",
      "not_proven": "Individual contribution scope, experiment ownership or research quality attributable to this candidate alone.",
      "interview_question": "Which part of dataset construction or evaluation did you own, what failure mode changed the study design, and how did you validate it?"
    },
    {
      "project": "VoxBlink-CN",
      "links": {
        "repository": "https://github.com/shenxingy/VoxBlink-CN",
        "institutional_record": "https://ugstudies.dukekunshan.edu.cn/2023-srs-poster-session/"
      },
      "what_it_does": "Large-scale Chinese audio-visual speaker-verification dataset assembled through avatar-based and clustering-based video processing pipelines.",
      "observable_evidence": [
        "The DKU SRS page names Xingyu Shen with the VoxBlink-CN project and mentor Ming Li.",
        "The repository reports 41,634 speakers, 789,321 videos, 11,385 hours and 3,641,941 utterances.",
        "The public pipeline documents collection, face/lip tracking, embeddings, active-speaker detection, overlap removal and clustering."
      ],
      "evidence_depth": "SUBSTANTIAL_DATA_PIPELINE_ARTIFACT",
      "role_relevance": "HIGH for large-scale data systems, computer vision/audio and quality-control pipelines.",
      "not_proven": "Exact code ownership, personal contribution percentage, dataset rights/compliance or independent reproduction.",
      "interview_question": "Show one pipeline stage you owned, its precision/recall tradeoff, the largest data-quality failure and the monitoring used to catch it."
    },
    {
      "project": "RTVis",
      "links": {
        "repository": "https://github.com/RTVis/RTVis",
        "institutional_record": "https://scholars.duke.edu/publication/1623120"
      },
      "what_it_does": "Research-trend visualization toolkit with author co-occurrence, citation, word-frequency and theme-river views plus local and Docker deployment.",
      "observable_evidence": [
        "The Duke institutional publication record names Xingyu Shen with RTVis.",
        "The public repository contains Python preprocessing and application code, a demo dataset, Dockerfile and docker-compose deployment.",
        "The repository presents four distinct research-orientation visualizations and cross-platform installation instructions."
      ],
      "evidence_depth": "FUNCTIONAL_RESEARCH_TOOL",
      "role_relevance": "MODERATE for data applications and research tooling; useful but visibly narrower than Clade or VoxBlink-CN.",
      "not_proven": "Exact contribution, maintained production use, user impact or current code quality.",
      "interview_question": "Open one visualization, explain the data model and performance bottleneck, and identify what you would redesign for production use."
    }
  ],
  "screening_summary": {
    "strongest_visible_match": "AI systems/orchestration plus large-scale data and forensic evaluation artifacts.",
    "secondary_signal": "Research communication and deployable data tooling across several public artifacts.",
    "material_unknowns": [
      "personal contribution and ownership for each collaborative artifact",
      "production reliability and real user or business impact",
      "current employment and dates if material",
      "degree conferral if material"
    ],
    "human_next_step": "Use the linked artifacts to choose whether to advance to a structured technical interview; if advancing, ask the project-specific ownership questions before any offer decision."
  },
  "official_record_route": {
    "activated_for_this_subject": false,
    "rule": "DOJ, FBI or IC3 search requires a separate lawful alias, case number, official notice or authorized prior-incident anchor. It never runs because of a sparse footprint, common name or no-hit."
  },
  "automatic_score_or_recommendation": false
}
