{
    "archive_path": "archive/1765982737.730339",
    "base_url": "kevinchen.co/blog/cant-stop-plagiarism-in-computer-science",
    "basename": "",
    "bookmarked_date": "2025-12-17 14:45",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/kevinchen.co/blog/cant-stop-plagiarism-in-computer-science",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=kevinchen.co",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "kevinchen.co",
    "downloaded_at": "2025-12-17T14:45:44.615005+00:00",
    "downloaded_datestr": "2025-12-17 14:45",
    "extension": "",
    "hash": "8C8BRGYATCN2BKQ3BJ4S",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-17T14:47:55.187670+00:00",
                "index_texts": null,
                "output": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/']' timed out after 60 seconds",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:46:55.035706+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2025-12-17T14:46:16.464194+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:46:05.399382+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=kevinchen.co"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-17T14:45:48.205933+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:45:44.884521+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-17T14:45:48.646459+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:45:48.297166+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2025-12-17T14:46:43.578931+00:00",
                "index_texts": [
                    "Why we still can\u2019t stop plagiarism in undergraduate computer science \u2014 Kevin Chen (https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/) (/feed.xml) (/css/main-v5.css)  (/) Kevin Chen  (/) Home (/projects/) Projects (/gallery/) Photos (/blog/) Blog    (/blog/cant-stop-plagiarism-in-computer-science/) Why we still can\u2019t stop plagiarism in undergraduate computer science  Lessons on incentives and scaling from two years as a teaching assistant 22 Mar 2018   Imagine that you\u2019re hired to work at your local public library. As an eagle-eyed\ncheckout clerk, you soon realize that half the patrons leave without actually\nchecking out their books! This leaves everyone else scratching their heads when\nthe catalog doesn\u2019t match the shelves. But conveniently, the library has an\nunused anti-theft alarm sitting in the back room. There\u2019s just one problem: your supervisor, though sympathetic to your cause,\ndoesn\u2019t want you using the alarm. You see, people take books for many reasons,\nand not all of them are malicious. Besides, who cares if the shelves are in\ndisarray? They\u2019ve been like that for years, but everyone who works at the\nlibrary still gets their paychecks the all the same.  If you want to set up the\nalarm, you\u2019ll have to do so on your own time, in addition to your other\nresponsibilities. In many undergraduate computer science programs, this is the absurd reality we\nface when trying to combat plagiarism. Everyone agrees plagiarism is wrong.\nEveryone wishes they could stop it. Everyone has access to the tools that find\nit. But no one seems willing to take any action. Why bother? The most important goal is to keep the course fair for students who do honest\nwork. Instructors must assign grades that accurately reflect performance. A\nstudent who grapples with a problem \u2014 becoming a stronger programmer in the\nprocess \u2014 should never receive a lower grade than one who copies and pastes. Finally, as educators, we also hope that the accused student can learn difficult\nlessons about ethical behavior in the classroom rather than the workplace. Understanding the scope of plagiarism I\u2019ve been a teaching assistant in a lower-division computer science course for\nthe past two years. Each semester, in a class of 200 to 300 students, we\ntypically discover 20 to 40 blatant cases of plagiarism on homework. And because\nof the nature of our process, even more cases go undetected. Here\u2019s how it\nworks: We begin by uploading our students\u2019 code to an online (https://en.wikipedia.org/wiki/Plagiarism_detection) plagiarism detection\ntool . The tool compares the submissions with each other and with\nour massive back catalog of previous solutions, flagging pairs of similar\nprograms. We\u2019ll manually review nearly 1,000 of these pairs over the course of\nthe semester, throwing out all but the most suspicious cases. These cases end up\nrepresenting about 100 students. Then, we apply another filter, keeping only the cases that contain\nindisputable evidence \u2014 for example, hundreds of lines copied right down to\nthe last whitespace error. We have virtually eliminated false positives at this\npoint. (In the course\u2019s entire history, only one such example exists.) That means we aren\u2019t catching cases where there\u2019s plausible deniability. And we\ncertainly can\u2019t find students who transform someone else\u2019s homework beyond\nrecognition. (They\u2019ve demonstrated a better grasp of programming than your\naverage copy-paste-rename job, but still haven\u2019t learned what the homework was\ntrying to teach.) Because our process heavily favors (https://en.wikipedia.org/wiki/Precision_and_recall) precision over recall ,\nthe 20 to 40 cases at the end represent a lower bound on plagiarism in the\ncourse. This is corroborated by the results from Lisa Yan et al, the authors of\na new plagiarism tool named TMOSS. In a Stanford course, 43 percent of true\npositives detected by TMOSS wouldn\u2019t have been discovered by a traditional tool\nsimilar to ours.(See Footnote 1) 1   1   The large number of students suspected of plagiarism isn\u2019t unique to our course.\nIn fact, if a university or course doesn\u2019t have comparable numbers, it most\nlikely reflects their (lack of) plagiarism detection, rather than the actual\nrate of plagiarism. According to a 2017 New York Times report,(See Footnote 2) 2   2  a course at UC Berkeley\nfound about 1 in 7 students in violation of policies on copying code. At\nStanford, one course suspected up to 20\u00a0percent of its students. And in spring\n2017, Harvard\u2019s CS50 reported nearly 60 of 600 students for cheating. The aftermath What happens after an instructor confronts a student about plagiarism? Some admit their mistake right away. Others deny it for awhile before coming\nclean. And every semester, there will be a few conversations that go something\nlike this: \u201cThe TAs think you copied a few homeworks. You need to tell me what\u2019s going\non.\u201d \u201cI didn\u2019t cheat. I don\u2019t know what you\u2019re talking about.\u201d \u201cLook, Bob, you have three screens of code that are exactly the same as\nAlice\u2019s, except she called the parameter searchQuery and you renamed it to info_strings in your final commit.\u201d \u201cThose similarities are a coincidence that happens fairly often in\nprogramming. Stop harassing me!\u201d *files complaint with department*   Trying to gaslight the computer science department about computer science makes\nno sense, but let\u2019s move on. The long tail of case resolution times. After confronting students,\nconversations like the above take up the vast majority of the teaching staff\u2019s\ntime. A small but vocal minority of students hope to avoid consequences by\ndragging out their cases. They send endless emails to the teaching staff, sometimes even getting parents\ninvolved. They appeal to the relevant offices in the university, claiming that\nin computer programming, independently writing hundreds of identical lines of\ncode is a common occurrence. To refute these claims, the teaching staff might\nthen be asked to provide a nontechnical explanation of the similarities,\nsometimes months after the semester has ended. Combined with the time it takes to find cases in the first place, the time sink\nalone is enough to discourage many instructors from looking. But don\u2019t worry,\nbecause there\u2019s more: Lack of support from the university. When there\u2019s enough noise,\nadministrators in high places start getting spooked. No one wants to end up (https://www.nytimes.com/2017/05/29/us/computer-science-cheating.html) on\nthe front page of the New York Times  . So rather than offer their\nsupport and influence, they do nothing. Teaching staffs are left to deal with\nthe issue using their own time and resources. Becoming the bad guy. At the end of it all, students take out their anger in\ncourse evaluations. Instructors become \u201cterrible human beings\u201d and worse for\ndaring to find out why so many programs written by different people have the\nexact same bugs. These reviews matter: the hiring process for faculty positions often requires\ncandidates to submit their past student evaluations. And like any other form of\nanonymous online abuse, repeatedly reading personal attacks and other vitriol\nabout yourself as part of your job carries a large mental health cost.\n(Imagine if your boss outsourced a portion of your quarterly performance review\nto a panel of YouTube commenters and Twitter trolls.) On a more personal level, it hurts to know which students are plagiarizing. We teach because we want to help people learn. It might be a little\ndisappointing when, for instance, a student comes to office hours at the end of\na semester not knowing how to compile programs. But it\u2019s far worse to find out\nthat they\u2019d asked for compilation help because they were trying to submit a\nfriend\u2019s solution. Competing against inaction It\u2019s clear that instructors who choose to pursue plagiarism cases encounter a\nwide variety of disincentives, ranging from slightly unpleasant to truly\ndreadful. But on an individual level, nothing bad really happens to those who\ndon\u2019t bother confronting plagiarism directly. Research-track faculty can still\npublish their papers. And with fewer negative reviews, teaching-track lecturers\nmight even improve their chances of being hired in the future. So it should come as no surprise that most professors avoid doing anything about\nplagiarism directly. Many creative solutions I attended a (https://easychair.org/smart-program/SIGCSE2018/2018-02-22.html#session:21605) discussion on plagiarism at this year\u2019s (/blog/sigcse-2018-notes/) SIGCSE , a computer science education conference. It opened my\neyes to all the ideas faculty have to avoid facing plagiarism directly. Some of them simply don\u2019t generalize, like creating new homework assignments\nfrom scratch each semester. People who bring this up have likely never asked an\ninstructor of an introductory course whether they\u2019d like to spend large chunks\nof time rushing out assignments in exchange for also reducing their quality. And\nit still wouldn\u2019t address plagiarism among students in the same semester. But what really bothered me were all the moralistic solutions. Making students sign an honor pledge. Offering leniency to those who turn\nthemselves in during the 72-hour \u201cregret period\u201d after the deadline. Or even\ninviting students to email the instructor at any time if they are about to\ncheat, so that the instructor can talk them out of it one-on-one. These are great ideas \u2014 for supporting students in courses with large\nenrollments. But they cannot be the only solution to combating plagiarism,\nbecause they do nothing to address the underlying incentive structure: Benefit: Copying and pasting someone else\u2019s code saves a ton of\ndevelopment time, and ensures a higher grade. Good grades might be tied to\ndesirable outcomes like graduation, financial aid, and job prospects. Cost: Zero! If the teaching staff doesn\u2019t look for plagiarism directly,\nstudents know they\u2019ll never be caught.  Empirical evidence shows that this is true. Our students sign an honor pledge at\nthe beginning of the semester, but we did not see any change in the rate of\nplagiarism. Even for Harvard\u2019s CS50, the birthplace of the \u201cregret period,\u201d\nDavid Malan reports that it \u201chas not materially impacted CS50\u2019s number of\ncases.\u201d(See Footnote 3) 3   3   Finding our way out Student incentives come from faculty Student incentives are set by the faculty: for example, if the instructor\nassigns a larger weight to some assignment, students usually spend more time\nworking on it. This means instructors have a lot of influence here! They can\ncombat plagiarism by making it costly. If students knew that there would be a consistent, non-zero cost to\nplagiarizing, many of them wouldn\u2019t do it. Copying homework would no longer be\nin their best interest. To have an effect, the policy would have to be\nimplemented consistently across all courses, beginning with the introductory\ncourses, which play the role of communicating the department\u2019s standards to new\nstudents. Achieving this level of uniformity means we need to provide a strong incentive\nevery instructor \u2014 not just the ones who care the most \u2014 to enforce the\nrules. Fixing faculty incentives University administrators should communicate their support. Instructors\nshould know that, not only will they suffer no retaliation, but that the\nuniversity encourages them to enforce university policies. This might require\nadministrators to acknowledge the inconvenient truth of widespread plagiarism. Next, we need to reduce the cost of looking for plagiarism. Efficiently deploying plagiarism detection software. Most instructors I\u2019ve\nspoken with use (https://theory.stanford.edu/~aiken/moss/) MOSS (if they use automated detection at all).\nHowever, each teaching staff must learn to use the software on their own,\nsometimes writing additional code to format the inputs or aggregate the results.\nWe\u2019ve somehow managed to turn plagiarism software \u2014 which should be written\nonce and deployed widely at zero marginal cost \u2014 into something that is very\nexpensive to use! Instead of deploying the software on a per-class basis, the university should\npay to integrate it into their learning management systems. The software could\neven automatically run on all programming assignments by default. Spreading the workload among more TAs. Universities can further decrease the\nburden by increasing the TA headcounts of courses that enforce plagiarism rules.\nThis ensures that time spent combing through the results of the software doesn\u2019t\ncome at the cost of teaching. Improving the quality of results from plagiarism detection software. The\nsoftware returns more false positives as class sizes increase, since the number\nof pairs of students grows quadratically. But because MOSS is proprietary, the\ncommunity\u2019s open-source development revolves around creating tools that wrap\nMOSS, rather than improving the core algorithm. Anyone who has an idea for\nimproving the algorithm must first reimplement all existing functionality\nstarting from the pseudocode in the paper.(See Footnote 4) 4   4  There is an opportunity\nto have a big impact by creating a free and open source alternative. We won\u2019t be able to fix everything. For example, personal attacks in course\nevaluations may be interleaved with genuine criticisms, making them difficult\nfor moderators to separate. I\u2019m not smart enough to have all the answers. These are just some of my ideas\nfor moving incentives in the right direction. It\u2019s not even clear which ideas\nwill work. But one thing\u2019s for certain: students will learn to care when their\nteachers start caring. Thanks to (https://www.ngomez.me/) Nelson Gomez , Joshua Zweig, John Hui, Vivian Shen,\nEdward Wang, and two anonymous readers for their feedback and discussion.  Lisa Yan, Nick McKeown, Mehran Sahami, and Chris Piech. (https://dl.acm.org/citation.cfm?id=3159490) TMOSS:\nUsing Intermediate Assignment Work to Understand Excessive Collaboration\nin Large Classes . SIGCSE 2018. \u21a9   Jess Bidgood and Jeremy B. Merrill. (https://www.nytimes.com/2017/05/29/us/computer-science-cheating.html) As Computer Coding Classes\nSwell, So Does Cheating . The New York Times. 29 May 2017. \u21a9   David J. Malan. (https://medium.com/@cs50/teaching-academic-honesty-in-cs50-965653520caf) Teaching Academic Honesty in CS50 .\n6 March 2018. \u21a9   Saul Schleimer, Daniel S. Wilkerson, and Alex Aiken. (https://theory.stanford.edu/~aiken/publications/papers/sigmod03.pdf) Winnowing: local algorithms for document fingerprinting . SIGMOD 2003. \u21a9     Thanks for reading! If you\u2019re enjoying my (/blog/) writing , I\u2019d love to send you infrequent notifications for new posts via my (https://buttondown.email/kevinchen) newsletter . You\u2019ll receive the full text of each post, plus occasional bonus content. (Email address) (Subscribe)  You can also follow me on Twitter ((https://twitter.com/kevinchen) @kevinchen ) or (/feed.xml) subscribe via RSS .  Scroll to Top (/blog/) All Posts     The opinions expressed on this website are my own. They do not represent those of my employer. (/about/) About this site      "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:46:43.522425+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2025-12-17T14:46:54.921998+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:46:47.744533+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2025-12-17T14:46:43.361590+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:46:38.591893+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpz2aal8sn",
                    "https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2025-12-17T14:46:22.394421+00:00",
                "index_texts": [
                    "Imagine that you\u2019re hired to work at your local public library. As an eagle-eyed\ncheckout clerk, you soon realize that half the patrons leave without actually\nchecking out their books! This leaves everyone else scratching their heads when\nthe catalog doesn\u2019t match the shelves. But conveniently, the library has an\nunused anti-theft alarm sitting in the back room.\n\nThere\u2019s just one problem: your supervisor, though sympathetic to your cause,\ndoesn\u2019t want you using the alarm. You see, people take books for many reasons,\nand not all of them are malicious. Besides, who cares if the shelves are in\ndisarray? They\u2019ve been like that for years, but everyone who works at the\nlibrary still gets their paychecks the all the same.  If you want to set up the\nalarm, you\u2019ll have to do so on your own time, in addition to your other\nresponsibilities.\n\nIn many undergraduate computer science programs, this is the absurd reality we\nface when trying to combat plagiarism. Everyone agrees plagiarism is wrong.\nEveryone wishes they could stop it. Everyone has access to the tools that find\nit. But no one seems willing to take any action.\n\nWhy bother?\n\nThe most important goal is to keep the course fair for students who do honest\nwork. Instructors must assign grades that accurately reflect performance. A\nstudent who grapples with a problem \u2014 becoming a stronger programmer in the\nprocess \u2014 should never receive a lower grade than one who copies and pastes.\n\nFinally, as educators, we also hope that the accused student can learn difficult\nlessons about ethical behavior in the classroom rather than the workplace.\n\nUnderstanding the scope of plagiarism\n\nI\u2019ve been a teaching assistant in a lower-division computer science course for\nthe past two years. Each semester, in a class of 200 to 300 students, we\ntypically discover 20 to 40 blatant cases of plagiarism on homework. And because\nof the nature of our process, even more cases go undetected. Here\u2019s how it\nworks:\n\nWe begin by uploading our students\u2019 code to an online plagiarism detection\ntool. The tool compares the submissions with each other and with\nour massive back catalog of previous solutions, flagging pairs of similar\nprograms. We\u2019ll manually review nearly 1,000 of these pairs over the course of\nthe semester, throwing out all but the most suspicious cases. These cases end up\nrepresenting about 100 students.\n\nThen, we apply another filter, keeping only the cases that contain\nindisputable evidence \u2014 for example, hundreds of lines copied right down to\nthe last whitespace error. We have virtually eliminated false positives at this\npoint. (In the course\u2019s entire history, only one such example exists.)\n\nThat means we aren\u2019t catching cases where there\u2019s plausible deniability. And we\ncertainly can\u2019t find students who transform someone else\u2019s homework beyond\nrecognition. (They\u2019ve demonstrated a better grasp of programming than your\naverage copy-paste-rename job, but still haven\u2019t learned what the homework was\ntrying to teach.)\n\nBecause our process heavily favors precision over recall,\nthe 20 to 40 cases at the end represent a lower bound on plagiarism in the\ncourse. This is corroborated by the results from Lisa Yan et al, the authors of\na new plagiarism tool named TMOSS. In a Stanford course, 43 percent of true\npositives detected by TMOSS wouldn\u2019t have been discovered by a traditional tool\nsimilar to ours.\n1\n\nThe large number of students suspected of plagiarism isn\u2019t unique to our course.\nIn fact, if a university or course doesn\u2019t have comparable numbers, it most\nlikely reflects their (lack of) plagiarism detection, rather than the actual\nrate of plagiarism.\n\nAccording to a 2017 New York Times report,\n2 a course at UC Berkeley\nfound about 1 in 7 students in violation of policies on copying code. At\nStanford, one course suspected up to 20\u00a0percent of its students. And in spring\n2017, Harvard\u2019s CS50 reported nearly 60 of 600 students for cheating.\n\nThe aftermath\n\nWhat happens after an instructor confronts a student about plagiarism?\n\nSome admit their mistake right away. Others deny it for awhile before coming\nclean. And every semester, there will be a few conversations that go something\nlike this:\n\n\n  \u201cThe TAs think you copied a few homeworks. You need to tell me what\u2019s going\non.\u201d\n\n  \u201cI didn\u2019t cheat. I don\u2019t know what you\u2019re talking about.\u201d\n\n  \u201cLook, Bob, you have three screens of code that are exactly the same as\nAlice\u2019s, except she called the parameter searchQuery and you renamed it to\ninfo_strings in your final commit.\u201d\n\n  \u201cThose similarities are a coincidence that happens fairly often in\nprogramming. Stop harassing me!\u201d *files complaint with department*\n\n\nTrying to gaslight the computer science department about computer science makes\nno sense, but let\u2019s move on.\n\nThe long tail of case resolution times. After confronting students,\nconversations like the above take up the vast majority of the teaching staff\u2019s\ntime. A small but vocal minority of students hope to avoid consequences by\ndragging out their cases.\n\nThey send endless emails to the teaching staff, sometimes even getting parents\ninvolved. They appeal to the relevant offices in the university, claiming that\nin computer programming, independently writing hundreds of identical lines of\ncode is a common occurrence. To refute these claims, the teaching staff might\nthen be asked to provide a nontechnical explanation of the similarities,\nsometimes months after the semester has ended.\n\nCombined with the time it takes to find cases in the first place, the time sink\nalone is enough to discourage many instructors from looking. But don\u2019t worry,\nbecause there\u2019s more:\n\nLack of support from the university. When there\u2019s enough noise,\nadministrators in high places start getting spooked. No one wants to end up on\nthe front page of the New York Times. So rather than offer their\nsupport and influence, they do nothing. Teaching staffs are left to deal with\nthe issue using their own time and resources.\n\nBecoming the bad guy. At the end of it all, students take out their anger in\ncourse evaluations. Instructors become \u201cterrible human beings\u201d and worse for\ndaring to find out why so many programs written by different people have the\nexact same bugs.\n\nThese reviews matter: the hiring process for faculty positions often requires\ncandidates to submit their past student evaluations. And like any other form of\nanonymous online abuse, repeatedly reading personal attacks and other vitriol\nabout yourself as part of your job carries a large mental health cost.\n(Imagine if your boss outsourced a portion of your quarterly performance review\nto a panel of YouTube commenters and Twitter trolls.)\n\nOn a more personal level, it hurts to know which students are plagiarizing.\nWe teach because we want to help people learn. It might be a little\ndisappointing when, for instance, a student comes to office hours at the end of\na semester not knowing how to compile programs. But it\u2019s far worse to find out\nthat they\u2019d asked for compilation help because they were trying to submit a\nfriend\u2019s solution.\n\nCompeting against inaction\n\nIt\u2019s clear that instructors who choose to pursue plagiarism cases encounter a\nwide variety of disincentives, ranging from slightly unpleasant to truly\ndreadful. But on an individual level, nothing bad really happens to those who\ndon\u2019t bother confronting plagiarism directly. Research-track faculty can still\npublish their papers. And with fewer negative reviews, teaching-track lecturers\nmight even improve their chances of being hired in the future.\n\nSo it should come as no surprise that most professors avoid doing anything about\nplagiarism directly.\n\nMany creative solutions\n\nI attended a discussion on plagiarism at this year\u2019s\nSIGCSE, a computer science education conference. It opened my\neyes to all the ideas faculty have to avoid facing plagiarism directly.\n\nSome of them simply don\u2019t generalize, like creating new homework assignments\nfrom scratch each semester. People who bring this up have likely never asked an\ninstructor of an introductory course whether they\u2019d like to spend large chunks\nof time rushing out assignments in exchange for also reducing their quality. And\nit still wouldn\u2019t address plagiarism among students in the same semester.\n\nBut what really bothered me were all the moralistic solutions.\n\nMaking students sign an honor pledge. Offering leniency to those who turn\nthemselves in during the 72-hour \u201cregret period\u201d after the deadline. Or even\ninviting students to email the instructor at any time if they are about to\ncheat, so that the instructor can talk them out of it one-on-one.\n\nThese are great ideas \u2014 for supporting students in courses with large\nenrollments. But they cannot be the only solution to combating plagiarism,\nbecause they do nothing to address the underlying incentive structure:\n\n\n  Benefit: Copying and pasting someone else\u2019s code saves a ton of\ndevelopment time, and ensures a higher grade. Good grades might be tied to\ndesirable outcomes like graduation, financial aid, and job prospects.\n  Cost: Zero! If the teaching staff doesn\u2019t look for plagiarism directly,\nstudents know they\u2019ll never be caught.\n\n\nEmpirical evidence shows that this is true. Our students sign an honor pledge at\nthe beginning of the semester, but we did not see any change in the rate of\nplagiarism. Even for Harvard\u2019s CS50, the birthplace of the \u201cregret period,\u201d\nDavid Malan reports that it \u201chas not materially impacted CS50\u2019s number of\ncases.\u201d\n3\n\nFinding our way out\n\nStudent incentives come from faculty\n\nStudent incentives are set by the faculty: for example, if the instructor\nassigns a larger weight to some assignment, students usually spend more time\nworking on it. This means instructors have a lot of influence here! They can\ncombat plagiarism by making it costly.\n\nIf students knew that there would be a consistent, non-zero cost to\nplagiarizing, many of them wouldn\u2019t do it. Copying homework would no longer be\nin their best interest. To have an effect, the policy would have to be\nimplemented consistently across all courses, beginning with the introductory\ncourses, which play the role of communicating the department\u2019s standards to new\nstudents.\n\nAchieving this level of uniformity means we need to provide a strong incentive\nevery instructor \u2014 not just the ones who care the most \u2014 to enforce the\nrules.\n\nFixing faculty incentives\n\nUniversity administrators should communicate their support. Instructors\nshould know that, not only will they suffer no retaliation, but that the\nuniversity encourages them to enforce university policies. This might require\nadministrators to acknowledge the inconvenient truth of widespread plagiarism.\n\nNext, we need to reduce the cost of looking for plagiarism.\n\nEfficiently deploying plagiarism detection software. Most instructors I\u2019ve\nspoken with use MOSS (if they use automated detection at all).\nHowever, each teaching staff must learn to use the software on their own,\nsometimes writing additional code to format the inputs or aggregate the results.\nWe\u2019ve somehow managed to turn plagiarism software \u2014 which should be written\nonce and deployed widely at zero marginal cost \u2014 into something that is very\nexpensive to use!\n\nInstead of deploying the software on a per-class basis, the university should\npay to integrate it into their learning management systems. The software could\neven automatically run on all programming assignments by default.\n\nSpreading the workload among more TAs. Universities can further decrease the\nburden by increasing the TA headcounts of courses that enforce plagiarism rules.\nThis ensures that time spent combing through the results of the software doesn\u2019t\ncome at the cost of teaching.\n\nImproving the quality of results from plagiarism detection software. The\nsoftware returns more false positives as class sizes increase, since the number\nof pairs of students grows quadratically. But because MOSS is proprietary, the\ncommunity\u2019s open-source development revolves around creating tools that wrap\nMOSS, rather than improving the core algorithm. Anyone who has an idea for\nimproving the algorithm must first reimplement all existing functionality\nstarting from the pseudocode in the paper.\n4 There is an opportunity\nto have a big impact by creating a free and open source alternative.\n\nWe won\u2019t be able to fix everything. For example, personal attacks in course\nevaluations may be interleaved with genuine criticisms, making them difficult\nfor moderators to separate.\n\n\n\nI\u2019m not smart enough to have all the answers. These are just some of my ideas\nfor moving incentives in the right direction. It\u2019s not even clear which ideas\nwill work. But one thing\u2019s for certain: students will learn to care when their\nteachers start caring.\n\nThanks to Nelson Gomez, Joshua Zweig, John Hui, Vivian Shen,\nEdward Wang, and two anonymous readers for their feedback and discussion.\n\n\n\n\n\n\n    Thanks for reading! If you\u2019re enjoying my writing, I\u2019d love to send you infrequent notifications for new posts via my newsletter. You\u2019ll receive the full text of each post, plus occasional bonus content.\n\n    \n\n    You can also follow me on Twitter (@kevinchen) or subscribe via RSS.\n\n\n\n\n    Scroll to Top\n    All Posts"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:46:18.290562+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-17T14:46:16.699004+00:00",
                "index_texts": null,
                "output": "Why we still can\u2019t stop plagiarism in undergraduate computer science \u2014 Kevin Chen",
                "pwd": "/data/archive/1765982737.730339",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-17T14:46:16.670706+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/']' timed out after 60 seconds",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "Why we still can\u2019t stop plagiarism in undergraduate computer science \u2014 Kevin Chen",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1765982737.730339",
    "newest_archive_date": "2025-12-17T14:46:55.035706+00:00",
    "num_failures": 1,
    "num_outputs": 8,
    "oldest_archive_date": "2025-12-17T14:45:44.884521+00:00",
    "path": "/blog/cant-stop-plagiarism-in-computer-science/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KCPCC5BZ778C525501S60YYS",
    "snapshot_id": "41b05b4d-6774-4c52-9145-caf5f2607bd9",
    "sources": [
        "/data/sources/1765982736-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1765982737.730339",
    "title": "Why we still can\u2019t stop plagiarism in undergraduate computer science \u2014 Kevin Chen",
    "url": "https://kevinchen.co/blog/cant-stop-plagiarism-in-computer-science/"
}